Viewing File: /usr/lib64/python3.6/html/__pycache__/parser.cpython-36.opt-1.pyc

3

sjI@sdZddlZddlZddlZddlmZdgZejdZejdZ	ejdZ
ejdZejd	Zejd
Z
ejdZejdZejd
ZejdejZejd
ZejdZGdddejZdS)zA parser for HTML and XHTML.N)unescape
HTMLParserz[&<]z
&[a-zA-Z#]z%&([a-zA-Z][-.a-zA-Z0-9]*)[^a-zA-Z0-9]z)&#(?:[0-9]+|[xX][0-9a-fA-F]+)[^0-9a-fA-F]z	<[a-zA-Z]>z--\s*>z+([a-zA-Z][^\t\n\r\f />\x00]*)(?:\s|/(?!>))*z]((?<=[\'"\s/])[^\s/>][^\s/=>]*)(\s*=+\s*(\'[^\']*\'|"[^"]*"|(?![\'"])[^>\s]*))?(?:\s|/(?!>))*aF
  <[a-zA-Z][^\t\n\r\f />\x00]*       # tag name
  (?:[\s/]*                          # optional whitespace before attribute name
    (?:(?<=['"\s/])[^\s/>][^\s/=>]*  # attribute name
      (?:\s*=+\s*                    # value indicator
        (?:'[^']*'                   # LITA-enclosed value
          |"[^"]*"                   # LIT-enclosed value
          |(?!['"])[^>\s]*           # bare value
         )
         (?:\s*,)*                   # possibly followed by a comma
       )?(?:\s|/(?!>))*
     )*
   )?
  \s*                                # trailing whitespace
z#</\s*([a-zA-Z][-.a-zA-Z0-9:_]*)\s*>c@seZdZdZd:ZddddZdd	Zd
dZdd
ZdZ	ddZ
ddZddZddZ
ddZd;ddZddZddZd d!Zd"d#Zd$d%Zd&d'Zd(d)Zd*d+Zd,d-Zd.d/Zd0d1Zd2d3Zd4d5Zd6d7Zd8d9ZdS)<raEFind tags and other markup and call handler functions.

    Usage:
        p = HTMLParser()
        p.feed(data)
        ...
        p.close()

    Start tags are handled by calling self.handle_starttag() or
    self.handle_startendtag(); end tags by self.handle_endtag().  The
    data between tags is passed from the parser to the derived class
    by calling self.handle_data() with the data as argument (the data
    may be split up in arbitrary chunks).  If convert_charrefs is
    True the character references are converted automatically to the
    corresponding Unicode character (and self.handle_data() is no
    longer split in chunks), otherwise they are passed by calling
    self.handle_entityref() or self.handle_charref() with the string
    containing respectively the named or numeric reference as the
    argument.
    scriptstyleT)convert_charrefscCs||_|jdS)zInitialize and reset this instance.

        If convert_charrefs is True (the default), all character references
        are automatically converted to the corresponding Unicode characters.
        N)rreset)selfrr
/usr/lib64/python3.6/parser.py__init__WszHTMLParser.__init__cCs:d|_d|_t|_d|_g|_d|_d|_tj	j
|dS)z1Reset this instance.  Loses all unprocessed data.z???Nr)rawdatalasttaginteresting_normalinteresting
cdata_elem_pending_pending_len_parse_threshold_markupbase
ParserBaser)r	r
r
rr`szHTMLParser.resetcCs|jt|7_|j|jkr,|jj|n~|jsB|j|7_n,|jj||jdj|j7_|jjd|_t|j}|jdt|j|krd|_nt|j|_dS)zFeed data to the parser.

        Call this as often as you want, with as little or as much text
        as you want (may include '\n').
        r
rrN)	rlenrrappendrjoincleargoahead)r	datanr
r
rfeedks



zHTMLParser.feedcCs:|jr,|jdj|j7_|jjd|_|jddS)zHandle any buffered data.r
rrN)rrrrrr)r	r
r
rcloses

zHTMLParser.closeNcCs|jS)z)Return full source of start tag: '<...>'.)_HTMLParser__starttag_text)r	r
r
rget_starttag_textszHTMLParser.get_starttag_textcCs$|j|_tjd|jtj|_dS)Nz</\s*%s\s*>)lowerrrecompileIr)r	elemr
r
rset_cdata_modes
zHTMLParser.set_cdata_modecCst|_d|_dS)N)rrr)r	r
r
rclear_cdata_modeszHTMLParser.clear_cdata_modecCsL|j}d}t|}x||kr|jr||jr||jd|}|dkr|jdt||d}|dkrvtjdj	||rvP|}n(|j
j	||}|r|j}n|jrP|}||kr|jr|jr|jt
|||n|j||||j||}||krP|j}|d|rLtj||r&|j|}	n|d|r>|j|}	nl|d|rV|j|}	nT|d|rn|j|}	n<|d	|r|j|}	n$|d
|kr|jd|d
}	nP|	dkr>|sP|jd|d
}	|	dkr|jd|d
}	|	dkr|d
}	n|	d
7}	|jr,|jr,|jt
|||	n|j|||	|j||	}q|d|rtj||}|r|jd
d}
|j|
|j}	|d|	d
s|	d
}	|j||	}qn:d||dkr|j|||d
|j||d
}Pq|d|rtj||}|rN|jd
}
|j|
|j}	|d|	d
s@|	d
}	|j||	}qtj||}|r|r|j||dkr|j}	|	|kr|}	|j||d
}Pn,|d
|kr|jd|j||d
}nPqqW|r:||kr:|jr:|jr|jr|jt
|||n|j||||j||}||d|_dS)Nr<&"z[\s;]z</z<!--z<?z<!rrz&#;)rrrrfindrfindmaxr%r&searchrstarthandle_datarZ	updatepos
startswithstarttagopenmatchparse_starttagparse_endtag
parse_commentparse_piparse_html_declarationcharrefgrouphandle_charrefend	entityrefhandle_entityref
incomplete)r	rBrirjZampposr9r7knamer
r
rrs












zHTMLParser.goaheadcCs|j}|||ddkr$|j|S|||ddkrB|j|S|||djdkr|jd|d}|d
krvdS|j||d	||dS|j|SdS)Nz<!--z<![	z	<!doctyperrr.r0r0)rr<Zparse_marked_sectionr$r1handle_declparse_bogus_comment)r	rFrgtposr
r
rr>s

z!HTMLParser.parse_html_declarationrcCsD|j}|jd|d}|dkr"dS|r<|j||d||dS)Nrr.rr0r0)rr1handle_comment)r	rFZreportrposr
r
rrN1szHTMLParser.parse_bogus_commentcCsH|j}tj||d}|sdS|j}|j||d||j}|S)Nr.rr0)rpicloser4r5	handle_pirB)r	rFrr9rGr
r
rr==szHTMLParser.parse_picCsd|_|j|}|dkr|S|j}||||_g}tj||d}|j}|jdj|_}x||kr$t	j||}|s~P|jddd\}	}
}|
sd}n^|dddko|d
dkns|dddko|ddknr|dd}|rt
|}|j|	j|f|j}qbW|||j}|d
kr|j
\}
}d	|jkr|
|jjd	}
t|j|jjd	}n|t|j}|j||||S|jdr|j||n"|j||||jkr|j||S)Nrrr.rK'"r/>
r0r0r0)rrV)r"check_for_whole_start_tagrtagfind_tolerantr9rBr@r$rattrfind_tolerantrrstripZgetposcountrr2r6endswithhandle_startendtaghandle_starttagCDATA_CONTENT_ELEMENTSr))r	rFendposrattrsr9rHtagmZattrnamerestZ	attrvaluerBlinenooffsetr
r
rr:IsP
(*

zHTMLParser.parse_starttagcCs|j}tj||}|r|j}|||d}|dkr>|dS|dkr~|jd|rZ|dS|jd|rjd	S||krv|S|dS|dkrd
S|dkrdS||kr|S|dStddS)Nrr/z/>r.r
z6abcdefghijklmnopqrstuvwxyz=/ABCDEFGHIJKLMNOPQRSTUVWXYZzwe should not get here!r0r0r0)rlocatestarttagend_tolerantr9rBr7AssertionError)r	rFrrdrGnextr
r
rrX|s.z$HTMLParser.check_for_whole_start_tagcCs|j}tj||d}|sdS|j}tj||}|s|jdk	rV|j||||Stj||d}|s|||ddkr|dS|j	|S|j
dj}|jd|j}|j
||dS|j
dj}|jdk	r||jkr|j||||S|j
|j|j|S)Nrr.rKz</>rr0)r	endendtagr4rB
endtagfindr9rr6rYrNr@r$r1
handle_endtagr*)r	rFrr9rOZ	namematchZtagnamer(r
r
rr;s6


zHTMLParser.parse_endtagcCs|j|||j|dS)N)r_rn)r	rcrbr
r
rr^szHTMLParser.handle_startendtagcCsdS)Nr
)r	rcrbr
r
rr_szHTMLParser.handle_starttagcCsdS)Nr
)r	rcr
r
rrnszHTMLParser.handle_endtagcCsdS)Nr
)r	rIr
r
rrAszHTMLParser.handle_charrefcCsdS)Nr
)r	rIr
r
rrDszHTMLParser.handle_entityrefcCsdS)Nr
)r	rr
r
rr6szHTMLParser.handle_datacCsdS)Nr
)r	rr
r
rrPszHTMLParser.handle_commentcCsdS)Nr
)r	Zdeclr
r
rrMszHTMLParser.handle_declcCsdS)Nr
)r	rr
r
rrSszHTMLParser.handle_picCsdS)Nr
)r	rr
r
runknown_declszHTMLParser.unknown_declcCstjdtddt|S)NzZThe unescape method is deprecated and will be removed in 3.5, use html.unescape() instead.r.)
stacklevel)warningswarnDeprecationWarningr)r	sr
r
rrs
zHTMLParser.unescape)rr)r)__name__
__module____qualname____doc__r`rrr r!r"r#r)r*rr>rNr=r:rXr;r^r_rnrArDr6rPrMrSrorr
r
r
rr?s8	z
3"()rxr%rqrZhtmlr__all__r&rrErCr?r8rRZcommentcloserYrZVERBOSErirlrmrrr
r
r
r<module>s(












Back to Directory File Manager