MpjddlZddlZddlmZddlmZmZmZmZm Z ddl m Z ddl m Z ddlmZddlmZddlmZdd lmZdd lmZmZddlZej0d Zed gd ZdedefdZde defdZdedeedffdZdedee dffdZ!dede"fdZ#dede fdZ$dedefdZ%de ddfdZ&de'de fdZ(de defdZ)de defd Z*de defd!Z+d"ee defd#Z, d:d"ee d$ed%edeeee e e ee ffee ffd&Z-de'defd'Z.de'defd(Z/de'defd)Z0d*e'd+e'defd,Z1d;d-e'd.e'd/e de'fd0Z2 dd?d=d d}td@dAurtdB||j|d }||jdCd }|Sy )Da8Return the codepage to use with a specific fcharsetN. Args: fcharsetN (int): The numeric argument N for a charsetN control word. Returns: (int OR None) Returns the int for a codepage if known. Returns None for unknown charsets or charsets with no corresponding codepage (such as OEM or DEFAULT.) r ANSI_CHARSET0x00)namehexdecimalidrDEFAULT_CHARSET0x01NSYMBOL_CHARSET0x02SHIFTJIS_CHARSET0x80HANGUL_CHARSET0x81GB2312_CHARSET0x86CHINESEBIG5_CHARSET0x88 GREEK_CHARSET0xA1TURKISH_CHARSET0xA2HEBREW_CHARSET0xB1ARABIC_CHARSET0xB2BALTIC_CHARSET0xBARUSSIAN_CHARSET0xCC THAI_CHARSET0xDEj EE_CHARSET0xEE OEM_CHARSET0xFFRTFDE.text_extractionTzGetting charset for r6)rrget)r.charsetscharset charset_ids r'get_codepage_num_from_fcharsetrtQs) .vt D) #&1$ G) "! F) &Vcs K ) $6CS I ) $6CS I ) )# N) O&3D I) %FSd K) $6CT J) $6CT J) $6CT J) %FSd K) N#3 G) Lv F) M$ G!)H$+,429+>?ll9d+G[[t,  c|jd}t|} |d}|jdd}d|zS#t$rYywxYw)aExtract the font number controlword default font if it exists. If an RTF file uses a default font, the default font number is specified with the \deffN control word, which must precede the font-table group. Args: tree (Tree): A lark Tree object. Should be the DeEncapsulator.full_tree object. Returns: The default font control number if it exists from the first `\deffN`. None if not found. ct|dS)Ns\deffr )vs r'z"get_default_font..s.q)<rurNr+) scan_valueslistr r)rdeff_gen deff_optionsdeffdeff_nums r'get_default_fontrxs\<H>LA::ab>  s8 AA font_treecni}|jD]}t|tsd}d}d}|jD]p}t|dr |j}t|dr$t |jdd}t |}Lt|dsYt |jdd}r|d}| t|}|| t|}| t|} nd} djtt|} t||| | ||<|S#t$rd}YfwxYw#t$rd}YgwxYw)zCreate a font tree dictionary with appropriate codeces to decode text. Args: font_tree (Tree): The .rtf font table object decoded as a tree. Returns: A dictionary which maps font numbers to appropriate python codeces needed to decode text. Nr+s \fcharset s\cpgru)rrr rr intrtcheck_codepage_numr"get_python_codecjoinr|rr) rparsed_font_treerrfcharsetcpg_numtok fchar_num codepage_numrtree_strs r'parse_font_treers_""#V dD !DHG}} 1/V<99D1#}E #CIIabM 2I=iHH1#x@!#))AB-0G 1# ','9('C ")0C,'9''B  +,\:E EHHT*Ft*L%MN)0|UH)U &G#VH !&,'+ , &,'+ ,s$1 D D& D#"D#& D43D4rcztj|}tjdj |||S)zReturns the python codec needed to decode bytes to unicode. Args: codepage_num (int): A codepage number. Returns: The name of the codec in the Python codec registry. Used as the name for enacoding/decoding. z6Found python codec corresponding to code page {0}: {1})r codepage2codeclogdebugformat)r text_codecs r'rrs5)),7JIIFMMl\fgh rucFtgd}||vr|Std|d)a Provide the codepage number back to you if it is valid. Args: codepage_num (int): A possible codepage number. Returns: The codepage number IF it is a valid codepage number Raises: ValueError: The codepage_num provided isn't a valid codepage number. )%iiiiiiiiiRiTiWiYiZi\i]i^i_i`iaibieifrgikr?rGrCrKiiitiuiviwixiyizi{i|i}iirkrcr2rOrSrWr[r_iiQi'i'i'i'i'i'i'i'i'i'i!'i%'i-'i_'ia'ib'i.i.i Ni!Ni"Ni#Ni$Ni%NiNiNiNiNiNi%Oi-Oi1Oi5Oi6Oi8OiM    ) 0 0 22ruc dd}d}t|D]\}}t|trt|s$t |j r|}|}>|At |j r|} t|j |j }td||j|j|j|j|j|j} | ||<tdd|jdz|jdz|j|j|j|j} | ||<d}d}-t"j%dj'||dur1dj!||Dcgc]}|j c}||<t)d|S#t$r|} |durkdj!||Dcgc]}|j ncc}wc}||<dj!||Dcgc]}|j ncc}wc}|<n| Yd} ~ d} ~ wwxYwcc}w) z Raises: ValueError: A Standalone high-surrogate was found. High surrogate followed by a illegal low-surrogate character. Nr start_posend_poslineend_linecolumn end_columnrurTrz^Standalone high-surrogate found. High surrogate followed by a illegal low-surrogate character.)rrr rrr rrr rrrrrrUnicodeDecodeErrorrrrrr") rr0use_ASCII_alternatives_on_unicode_decode_failuresurrogate_startsurrogate_highic surrogate_lowsurrogate_value surrogate_tok blank_tokr%s r'merge_surrogate_charsrsFON"-K! a   a %agg."#!" ,(1$%M%*?@T@T@M@S@S+U).h.=8F8P8P6C6K6K3A3F3F7D7M7M5C5J5J9F9Q9Q)S 5B1$)(*-4B4L4LQ4N2?2G2G2I/=/B/B3@3I3I1?1F1F5B5M5M%O '0 *.)-HH]ddesuBCDG4O47HHyYgOh=i!agg=i4j1(*JKK[-K\ O.%KtS8;S\]kSlAma!''AmAm8nH_5*-((Y}E]3^AGG3^3^*_HQK"$H(%>js7)CF%8H- % H*.H%G "H%9H H%%H*czt|tr+|jdk(r|jj dryy)N CONTROLWORDs\ucTF)rr rr rrs r'rr=s0$ 99 %zz$$W- rucV|jj}t|dd}|S)N)r rr)r#cur_ucs r'rrDs( ::   D ab]F Mruc,|D]}t|syy)zsChecks if an tree's children includes a hexarray tree. children (array): the children object from a tree. TF) is_hexarray)rr#s r' has_hexarrayrKs#  t  rucXt|tr|jjdk(ryy)zpChecks if an item is a hexarray tree. item (Tree or Token): an item to check to see if its a hex array rTF)rr rr rs r'rrUs$ $ 99??j ( rucp|jdd}tj|j}|S)z^Convert hex encoded string to bytes. item (str): a hex encoded string in format \'XX s\'ru)replacebytesfromhexr)r# hexstring hex_bytess r'get_bytes_from_hex_encodedr_s1  VS)I i..01I ructddurtdj|||d}|j|}|j }tddurtdj||||S)zDecode a bytes object using a specified codec. item (bytes): A bytes object. codec (str): The name of the codec to use to decode the bytes roTzdecoding char {0} with font {1}CP1252z)char {0} decoded into {1} using codec {2})rrrrr)r#rdecodeds r'decode_hex_charrhsy +,4=DDT5QR }kk% GnnG+,4GNNtU\^cde NrucHeZdZ d dZdefdZdefdZdeefdZ dZ y) TextDecoderNcX||_||_||_d|_g|_i|_y)a  keep_fontdef: (bool) If False (default), will remove fontdef's from object tree once they are processed. initial_byte_count: (int) The initial Unicode Character Byte Count. Does not need to be set unless you are only providing a RTF snippet which does not contain the RTF header which sets the information. use_ASCII_alternatives_on_unicode_decode_failure: (bool) If we encounter errors when decoding unicode chars we will use the ASCII alternative since that's what they are included for. N) keep_fontdefucbcr default_font font_stack font_table)selfr initial_byte_countrs r'__init__zTextDecoder.__init__|s3)& @p=!ruobjct||_|jg|_t|jd}t ||_tddurtd|yy)Z obj (Tree): A lark Tree object. Should be the DeEncapsulator.full_tree. rroTzFONT TABLE FOUND: N) rr"r#r(rrr$rr)r%r( raw_fonttbls r' set_font_infozTextDecoder.set_font_infosb -S1,,-$S\\!_5 )+6 / 0D 8 "4[M B C 9ruc|j||j}|j|Dcgc]}|c}|_ycc}w)r*N)r,riterate_on_children)r%r(rrs r'update_childrenzTextDecoder.update_childrens; 3<<#'#;#;H#EFaF Fs Arc~t|r1t||j\}}t|||j}|S)N)r)rrr!rr)r%rrs r' prep_unicodezTextDecoder.prep_unicodesG !( +#>hIM#T Hi -X-6-1-b-bdH ruc # Kg}tddur"tdtdt|z|j|}tddurtdt|z|D]m}t |rb|j j |jj|j |j|jdusl|qt|rt|jj}td||j|j|j |j"|j$|j&}tddurtd|d ||t)|rd}d}d}d} d} d} d} |j*D]} | ^| j}| j}| j }| j"} | j$} | j&} t-| j} c| j}| j"} | j&} | t-| jz } |j.|j d }|j0}t3| |}td||||| | | }|&t5|t6r4|j9|j*Dcgc]}|c}|_|j|p|D]}|j j;ycc}ww) NroTz2Starting to iterate on text extraction children...z PREP-BEFORE: z PREP-AFTER: rrzUNICODE TOKEN z: )rrrr1r-r#rr r!r rrrr rrrrrrrrrr$rrrr r.r)r%r set_fontsr#r decoded_tok_hex_start_pos _hex_end_pos_hex_start_line _hex_end_line_hex_start_column_hex_end_column base_byteshexchildcurrent_fontdef current_codec decoded_hexdecoded_hex_tokrs r'r.zTextDecoder.iterate_on_childrens / 0D 8  T U X > ?$$X. / 0D 8 tH~ = >> Dd#&&tzz'7'7'9:  ,$$,J#D)/ ;BBD#H$+.2nn,0LL)--1]]+/;;/3@  78D@'.b (NO!!T"!%# "& $ $(!"&! $ QH!))1););'/'7'7 *2--(0(9(9 ,4OO)*2*=*=%?%O '/'7'7 (0(9(9 *2*=*="&@&PP Q#'//$//"2E"F / 5 5 -j-H "'(32@0<-<1>/@3B#D&%D$',0,D,DT]],S Tq T   }> ~ "A OO   ! " !UsCLG'L> K>rSs "00%)449g  Y M N44$4> % D $c$eCHo$N4E#d(O..t..b 3 3 bSbSb0 jc jd j"#8  4 Ed0 U t  T%[ T  :>23P$u+P26P,/P8=$)$u+tE$u+