
    Dj              
           S r SSKJrJr  \" S5      r\" S5      r1 Skr1 Skr SS\	S	\
S
\\
   S\\	   4S jjrS\	4S jrg)a  Text processing utilities for TTS inference.

Provides:
- ``chunk_text_punctuation()``: Splits long text into model-friendly chunks at
  sentence boundaries, with abbreviation-aware punctuation splitting.
- ``add_punctuation()``: Appends missing end punctuation (Chinese or English).
    )ListOptionalu   .,;:!?。，；：！？u   "''）]》》>」】>   !"'),.:;?]}   …   、   。   】   ！   ）   ，   ：   ；   ？   ……>5   Co.Dr.Fr.Ft.Jr.Lt.Mr.Ms.Mt.No.Rd.Sr.St.Vs.vs.Apr.Aug.Ave.Col.Cpl.Dec.Est.Etc.Feb.Gen.Gov.Hon.Inc.Jan.Ltd.Maj.Mar.Mrs.Nov.Oct.Rep.Rev.Sen.Sep.Sgt.def.e.g.fig.i.e.Blvd.Capt.Cmdr.Corp.Dept.Pres.Prof.Sept.approx.Ntext	chunk_lenmin_chunk_lenreturnc                 <   / n/ n[        U 5      nU H  n[        U5      S:X  a9  [        U5      S:w  a*  U[        ;   d
  U[        ;   a  US   R	                  U5        MK  UR	                  U5        U[        ;   d  Mh  SnUS:X  aE  SR                  U5      R                  5       nU(       a  UR                  5       S   n	U	[        ;   a  SnU(       a  M  UR	                  U5        / nM     [        U5      S:w  a  UR	                  U5        / n
/ nU HS  n[        U5      [        U5      -   U::  a  UR                  U5        M1  [        U5      S:  a  U
R	                  U5        UnMU     [        U5      S:  a  U
R	                  U5        Ub  [        U
5      S:  =(       a    [        U
S   5      U:  n/ n[        U
5       H  u  nnUS:X  a  U(       a  US   R                  U5        M)  [        U5      U:  a  UR	                  U5        MK  [        U5      S:X  a  UR	                  U5        Mm  US   R                  U5        M     OU
nU Vs/ s HH  nSR                  U5      R                  5       (       d  M)  SR                  U5      R                  5       PMJ     nnU$ s  snf )z
Splits the input tokens list into chunks according to punctuations,
avoiding splits on common abbreviations (e.g., Mr., No.).
r   Fr
    T   )listlenSPLIT_PUNCTUATIONCLOSING_MARKSappendjoinstripsplitABBREVIATIONSextend	enumerate)rP   rQ   rR   	sentencescurrent_sentencetokens_listtokenis_abbreviationtemp_str	last_wordmerged_chunkscurrent_chunksentencefirst_chunk_short_flagfinal_chunksichunkchunk_stringss                     4/mnt/workspace/git/OmniVoice/omnivoice/utils/text.pychunk_text_punctuationrs   w   sv    It*K  !Q&I!#++u/EbM  ' ##E* ))"'C<!ww'78>>@H$,NN$4R$8	$5.2O&$$%56')$5 8 !)* MM}H-:  *=!A%$$]3$M  =A]+  "Ls=+;'<}'L 	 !-0HAuAv0R ''.u:. ''.<(A-$++E2$R(//6 1 % -9,85BGGEN<P<P<RL   s   (J2#Jc                     U R                  5       n U (       d  U $ U S   [        ;  a  [        S U  5       5      nX(       a  SOS-  n U $ )z2Add punctuation if there is not in the end of textrU   c              3   L   #    U  H  nS Us=:*  =(       a    S:*  Os  v   M     g7f)u   一u   鿿N ).0chars     rr   	<genexpr>"add_punctuation.<locals>.<genexpr>   s      G$$T55X55$s   "$r   r
   )r^   END_PUNCTUATIONany)rP   
is_chineses     rr   add_punctuationr~      sD    ::<DBx&G$GG
,K    )N)__doc__typingr   r   setrZ   r[   r{   r`   strintrs   r~   rv   r   rr   <module>r      s   $ " 23 -.<6x $(U
UU C=U 
#Y	Up# r   