o
    í6WjšM  ã                   @  s*  d Z ddlmZ dZdgZddlmZ ddlZddlm	Z	m
Z
mZmZmZmZmZmZmZmZmZ ddlmZmZmZmZmZmZ dd	lmZmZ dd
lmZm Z m!Z!m"Z" ddl#m$Z$ ersddl%m&Z& ddlm'Z' ddl(m)Z)m*Z*m+Z+ dZ,e
ee-e-f e-e-gdf Z.G dd„ deeƒZ/G dd„ de!ƒZ0dS )zCUse the HTMLParser library to parse HTML files that aren't too bad.é    )ÚannotationsÚMITÚHTMLParserTreeBuilder)Ú
HTMLParserN)ÚAnyÚCallableÚcastÚDictÚIterableÚListÚOptionalÚTYPE_CHECKINGÚTupleÚTypeÚUnion)ÚAttributeDictÚCDataÚCommentÚDeclarationÚDoctypeÚProcessingInstruction)ÚEntitySubstitutionÚUnicodeDammit)ÚDetectsXMLParsedAsHTMLÚHTMLÚHTMLTreeBuilderÚSTRICT©ÚParserRejectedMarkup)ÚBeautifulSoup)ÚNavigableString)Ú	_EncodingÚ
_EncodingsÚ
_RawMarkupzhtml.parserc                   @  sæ   e Zd ZU dZded< dZded< 	 edœd;dd„Zded< ded< ded< d<dd„Zd=dd„Z	d>d?dd „Z	d>d@d"d#„Z
dAd%d&„Ze d'¡Ze d(¡ZedBd+d,„ƒZdCd-d.„ZdCd/d0„ZdAd1d2„ZdDd4d5„ZdAd6d7„ZdAd8d9„Zd:S )EÚBeautifulSoupHTMLParserÚreplaceÚstrÚREPLACEÚignoreÚIGNORE©Úon_duplicate_attributeÚsoupr   Úargsr   r+   ú&Union[str, _DuplicateAttributeHandler]Úkwargsc                O  s@   || _ || _|jj| _tj| g|¢R i |¤Ž g | _|  ¡  d S ©N)r,   r+   ÚbuilderÚattribute_dict_classr   Ú__init__Úalready_closed_empty_elementÚ_initialize_xml_detector)Úselfr,   r+   r-   r/   © r7   úf/home/esfera/Documents/content_generation/venv/lib/python3.10/site-packages/bs4/builder/_htmlparser.pyr3   U   s   
	z BeautifulSoupHTMLParser.__init__z	List[str]r4   ÚmessageÚreturnÚNonec                 C  s   t |ƒ‚r0   r   )r6   r9   r7   r7   r8   Úerrorp   s   zBeautifulSoupHTMLParser.errorÚtagÚattrsúList[Tuple[str, Optional[str]]]c                 C  s"   | j ||dd� | j|dd� dS )z…Handle an incoming empty-element tag.

        html.parser only calls this method when the markup looks like
        <tag/>.
        F)Úhandle_empty_element©Úcheck_already_closedN)Úhandle_starttagÚhandle_endtag)r6   r=   r>   r7   r7   r8   Úhandle_startendtag€   s   z*BeautifulSoupHTMLParser.handle_startendtagTr@   Úboolc                 C  sô   |   ¡ }|D ]3\}}|du rd}||v r5| j}|| jkrq|d| jfv r)|||< qtt|ƒ}||||ƒ q|||< q| jjjrF|  	¡ \}}	nd }}	| jj
|dd|||	d�}
|
durl|
jrl|rl| j|dd� | j |¡ | jdu rx|  |¡ dS dS )zÓHandle an opening tag, e.g. '<tag>'

        :param handle_empty_element: True if this tag is known to be
            an empty-element tag (i.e. there is not expected to be any
            closing tag).
        NÚ )Ú
sourcelineÚ	sourceposFrA   )r2   r+   r)   r'   r   Ú_DuplicateAttributeHandlerr,   r1   Ústore_line_numbersÚgetposrC   Úis_empty_elementrD   r4   ÚappendÚ_root_tag_nameÚ_root_tag_encountered)r6   r=   r>   r@   Ú	attr_dictÚkeyÚvalueÚon_duperH   rI   ÚtagObjr7   r7   r8   rC   ”   s2   




ÿ

ÿz'BeautifulSoupHTMLParser.handle_starttagrB   c                 C  s.   |r|| j v r| j  |¡ dS | j |¡ dS )zìHandle a closing tag, e.g. '</tag>'

        :param tag: A tag name.
        :param check_already_closed: True if this tag is expected to
           be the closing portion of an empty-element tag,
           e.g. '<tag></tag>'.
        N)r4   Úremover,   rD   )r6   r=   rB   r7   r7   r8   rD   Ò   s   	z%BeautifulSoupHTMLParser.handle_endtagÚdatac                 C  s   | j  |¡ dS )z4Handle some textual data that shows up between tags.N)r,   Úhandle_data©r6   rW   r7   r7   r8   rX   ä   s   z#BeautifulSoupHTMLParser.handle_dataz^([0-9]+)(.*)z^([0-9a-f]+)(.*)ÚnameúTuple[str, bool, str]c           	      C  sÀ   d}d}d}d}| j }| d¡s| d¡r |dd… }d}| j}d}zt||ƒ}W n! tyJ   | |¡}|durHt| ¡ d	 |ƒ}| ¡ d }Y nw |du rTd}|}nt |¡\}}|||fS )
a°  Convert a numeric character reference into an actual character.

        :param name: The number of the character reference, as
          obtained by html.parser

        :return: A 3-tuple (dereferenced, replacement_added,
          extra_data). `dereferenced` is the dereferenced character
          reference, or the empty string if there was no
          reference. `replacement_added` is True if the reference
          could only be dereferenced by replacing content with U+FFFD
          REPLACEMENT CHARACTER. `extra_data` is a portion of data
          following the character reference, which was deemed to be
          normal data and not part of the reference at all.
        rG   Fé
   ÚxÚXé   Né   r   )	Ú&_DECIMAL_REFERENCE_WITH_FOLLOWING_DATAÚ
startswithÚ"_HEX_REFERENCE_WITH_FOLLOWING_DATAÚintÚ
ValueErrorÚsearchÚgroupsr   Únumeric_character_reference)	ÚclsrZ   ÚdereferencedÚreplacement_addedÚ
extra_dataÚbaseÚregÚ	real_nameÚmatchr7   r7   r8   Ú(_dereference_numeric_character_referenceë   s0   
€ñ
z@BeautifulSoupHTMLParser._dereference_numeric_character_referencec                 C  sH   |   |¡\}}}|rd| j_|dur|  |¡ |dur"|  |¡ dS dS )z×Handle a numeric character reference by converting it to the
        corresponding Unicode character and treating it as textual
        data.

        :param name: Character number, possibly in hexadecimal.
        TN)rq   r,   Úcontains_replacement_charactersrX   )r6   rZ   rj   rk   rl   r7   r7   r8   Úhandle_charref"  s   
ÿz&BeautifulSoupHTMLParser.handle_charrefc                 C  s0   t j |¡}|dur|}nd| }|  |¡ dS )zÈHandle a named entity reference by converting it to the
        corresponding Unicode character(s) and treating it as textual
        data.

        :param name: Name of the entity reference.
        Nz&%s)r   ÚHTML_ENTITY_TO_CHARACTERÚgetrX   )r6   rZ   Ú	characterrW   r7   r7   r8   Úhandle_entityref1  s
   z(BeautifulSoupHTMLParser.handle_entityrefc                 C  s&   | j  ¡  | j  |¡ | j  t¡ dS )zOHandle an HTML comment.

        :param data: The text of the comment.
        N)r,   ÚendDatarX   r   rY   r7   r7   r8   Úhandle_commentD  s   
z&BeautifulSoupHTMLParser.handle_commentÚdeclc                 C  s6   | j  ¡  |tdƒd… }| j  |¡ | j  t¡ dS )zYHandle a DOCTYPE declaration.

        :param data: The text of the declaration.
        zDOCTYPE N)r,   rx   ÚlenrX   r   )r6   rz   r7   r7   r8   Úhandle_declM  s   
z#BeautifulSoupHTMLParser.handle_declc                 C  sN   |  ¡  d¡rt}|tdƒd… }nt}| j ¡  | j |¡ | j |¡ dS )z{Handle a declaration of unknown type -- probably a CDATA block.

        :param data: The text of the declaration.
        zCDATA[N)Úupperrb   r   r{   r   r,   rx   rX   )r6   rW   ri   r7   r7   r8   Úunknown_declW  s   
z$BeautifulSoupHTMLParser.unknown_declc                 C  s0   | j  ¡  | j  |¡ |  |¡ | j  t¡ dS )z\Handle a processing instruction.

        :param data: The text of the instruction.
        N)r,   rx   rX   Ú_document_might_be_xmlr   rY   r7   r7   r8   Ú	handle_pif  s   

z!BeautifulSoupHTMLParser.handle_piN)r,   r   r-   r   r+   r.   r/   r   )r9   r&   r:   r;   )r=   r&   r>   r?   r:   r;   )T)r=   r&   r>   r?   r@   rF   r:   r;   )r=   r&   rB   rF   r:   r;   )rW   r&   r:   r;   )rZ   r&   r:   r[   )rZ   r&   r:   r;   )rz   r&   r:   r;   )Ú__name__Ú
__module__Ú__qualname__r'   Ú__annotations__r)   r3   r<   rE   rC   rD   rX   ÚreÚcompilera   rc   Úclassmethodrq   rs   rw   ry   r|   r~   r€   r7   r7   r7   r8   r$   >   s2   
 ü

ü>



6


	

r$   c                      s”   e Zd ZU dZdZded< dZded< eZded< ee	e
gZd	ed
< ded< dZded< 		d&d'‡ fdd„Z			d(d)dd „Zefd*d$d%„Z‡  ZS )+r   z—A Beautiful soup `bs4.builder.TreeBuilder` that uses the
    :py:class:`html.parser.HTMLParser` parser, found in the Python
    standard library.

    FrF   Úis_xmlTÚ	picklabler&   ÚNAMEzIterable[str]Úfeaturesz$Tuple[Iterable[Any], Dict[str, Any]]Úparser_argsÚTRACKS_LINE_NUMBERSNúOptional[Iterable[Any]]Úparser_kwargsúOptional[Dict[str, Any]]r/   r   c                   sp   t ƒ }dD ]}||v r| |¡}|||< qtt| ƒjdi |¤Ž |p#g }|p'i }| |¡ d|d< ||f| _dS )a‚  Constructor.

        :param parser_args: Positional arguments to pass into
            the BeautifulSoupHTMLParser constructor, once it's
            invoked.
        :param parser_kwargs: Keyword arguments to pass into
            the BeautifulSoupHTMLParser constructor, once it's
            invoked.
        :param kwargs: Keyword arguments for the superclass constructor.
        r*   FÚconvert_charrefsNr7   )ÚdictÚpopÚsuperr   r3   ÚupdaterŒ   )r6   rŒ   r�   r/   Úextra_parser_kwargsÚargrS   ©Ú	__class__r7   r8   r3   ‚  s   
€
zHTMLParserTreeBuilder.__init__Úmarkupr#   Úuser_specified_encodingúOptional[_Encoding]Údocument_declared_encodingÚexclude_encodingsúOptional[_Encodings]r:   úDIterable[Tuple[str, Optional[_Encoding], Optional[_Encoding], bool]]c                 c  s€   � t |tƒr|dddfV  dS g }|r| |¡ g }|r!| |¡ t|||d|d�}|jdu r3tdƒ‚|j|j|j|jfV  dS )a2  Run any preliminary steps necessary to make incoming markup
        acceptable to the parser.

        :param markup: Some markup -- probably a bytestring.
        :param user_specified_encoding: The user asked to try this encoding.
        :param document_declared_encoding: The markup itself claims to be
            in this encoding.
        :param exclude_encodings: The user asked _not_ to try any of
            these encodings.

        :yield: A series of 4-tuples: (markup, encoding, declared encoding,
             has undergone character replacement)

            Each 4-tuple represents a strategy for parsing the document.
            This TreeBuilder uses Unicode, Dammit to convert the markup
            into Unicode, so the ``markup`` element of the tuple will
            always be a string.
        NFT)Úknown_definite_encodingsÚuser_encodingsÚis_htmlrž   zPCould not convert input to Unicode, and html.parser will not accept bytestrings.)	Ú
isinstancer&   rN   r   Úunicode_markupr   Úoriginal_encodingÚdeclared_html_encodingrr   )r6   rš   r›   r�   rž   r¡   r¢   Údammitr7   r7   r8   Úprepare_markup   s4   €


û
ÿ
üz$HTMLParserTreeBuilder.prepare_markupÚ_parser_classútype[BeautifulSoupHTMLParser]r;   c              
   C  s€   | j \}}t|tƒsJ ‚| jdusJ ‚|| jg|¢R i |¤Ž}z| |¡ | ¡  W n ty: } zt|ƒ‚d}~ww g |_dS )z®
        :param markup: The markup to feed into the parser.
        :param _parser_class: An HTMLParser subclass to use. This is only intended for use in unit tests.
        N)	rŒ   r¤   r&   r,   ÚfeedÚcloseÚAssertionErrorr   r4   )r6   rš   rª   r-   r/   ÚparserÚer7   r7   r8   r¬   è  s   

€ü
zHTMLParserTreeBuilder.feed)NN)rŒ   rŽ   r�   r�   r/   r   )NNN)
rš   r#   r›   rœ   r�   rœ   rž   rŸ   r:   r    )rš   r#   rª   r«   r:   r;   )r�   r‚   rƒ   Ú__doc__rˆ   r„   r‰   Ú
HTMLPARSERrŠ   r   r   r‹   r�   r3   r©   r$   r¬   Ú__classcell__r7   r7   r˜   r8   r   q  s    
 ý!ûH)1r±   Ú
__future__r   Ú__license__Ú__all__Úhtml.parserr   r…   Útypingr   r   r   r	   r
   r   r   r   r   r   r   Úbs4.elementr   r   r   r   r   r   Ú
bs4.dammitr   r   Úbs4.builderr   r   r   r   Úbs4.exceptionsr   Úbs4r   r    Úbs4._typingr!   r"   r#   r²   r&   rJ   r$   r   r7   r7   r7   r8   Ú<module>   s,   ÿ4   5