+
     ÌÙhŽ  c                   sX   € R t ^ RIt^ RIt ! R R]P                  P                  4      tR# )z§

enchant.tokenize.en:    Tokenizer for the English language

This module implements a PyEnchant text tokenizer for the English
language, based on very simple rules.

Nc                   sZ   a € ] tR t^,t o RtRR.tRR ltR tR tR t	R t
R	 tR
 tRtV tR# )Útokenizea>  Iterator splitting text into words, reporting position.

This iterator takes a text string as input, and yields tuples
representing each distinct word found in the text.  The tuples
take the form:

    (<word>,<pos>)

Where <word> is the word string found and <pos> is the position
of the start of the word within the text.

The optional argument <valid_chars> may be used to specify a
list of additional characters that can form part of a word.
By default, this list contains only the apostrophe ('). Note that
these characters cannot appear at the start or end of a word.
ZposNc                sô   € W n         Wn        ^ V n         V^ ,          p\        V\        4      '       d   V P                  4        R# V P                  4        R#   \         d    T P                  4         R# i ; i)i    N)Ú_valid_charsÚ_textÚ_offsetZ
isinstanceZstrÚ_initialize_for_unicodeÚ_initialize_for_binaryZ
IndexError)ÚselfÚtextZvalid_charsZchar1ó   &&& Ú8/usr/lib/python3.14/site-packages/enchant/tokenize/en.pyÚ__init__Ztokenize.__init__@   sf   € Ø'ÔØŒ
ØˆŒð	.Ø˜•GˆEô ˜%¤×%Ò%Ø×,Ñ,Ö.à×+Ñ+Ö-øô ô 	*Ø×'Ñ'×)ð	*ús   •	A ÁA7Á6A7c                óV   € V P                   V n        V P                  f
   RV n        R # R # ©N)Z')Ú_consume_alpha_bÚ_consume_alphar   ©r   ó   &r	   r   Ztokenize._initialize_for_binaryT   s)   € Ø"×3Ñ3ˆÔØ×ÑÒ$Ø &ˆDÖñ %ó    c                r   r   )Ú_consume_alpha_ur   r   r   r   r	   r   Z tokenize._initialize_for_unicodeY   s+   € Ø"×3Ñ3ˆÔØ×ÑÒ$ð
 !'ˆDÖñ %r   c                s¢   € V\        V4      8  g   Q hW,          P                  4       '       d   ^# W,          R8¼  d   V P                  W4      # ^ # )a-  Consume an alphabetic character from the given bytestring.

Given a bytestring and the current offset, this method returns
the number of characters occupied by the next alphabetic character
in the string.  Non-ASCII bytes are interpreted as utf-8 and can
result in multiple characters being consumed.
t   Â€)ÚlenÚisalphaÚ_consume_alpha_utf8)r   r   Úoffsets   &&&r	   r   Ztokenize._consume_alpha_bb   sH   € ð œ˜D›	Ô!Ð!Ð!Ø<×Ñ×!Ò!ÙØ\˜VÔ#Ø×+Ñ+¨DÓ9Ð9Ùr   c                s"  € ^pRpV'       g%   V^8:  d     WW#,            P                  R4      pK,  V'       g   ^ # VP                  4       '       d   V# \        P                  ! V4      ^ ,          R8X  d   V# ^ #   \         do     YY#,            P                  4       pM?  \         d2    RP                  YY#,             Uu. uF  qfNK  	  Mu upi up4      p Mi ; iTP                  R4      p Kì  i ; i  \         d    T^,          p EK  i ; i)zAConsume a sequence of utf8 bytes forming an alphabetic character.Z Zutf8ÚM)ZdecodeZAttributeErrorZtostringZjoinZUnicodeDecodeErrorr   ÚunicodedataÚcategory)r   r   r   ÚincrZuZsZcs   &&&    r	   r   Ztokenize._consume_alpha_utf8q   sø   € àˆØˆß˜ œ	ðð	)à f¥mÐ4×;Ñ;¸FÓC’A÷ ÙØ9‰9;Š;ØˆKÜ×Ò Ó" 1Õ%¨Ô,ØˆKÙøô &ô )ðOØ ¨&­-Ð8×AÑAÓC™øÜ)ô OØŸG™G°¸f½mÑ0LÓ$MÑ0L¨1¢QÒ0LùÔ$MÓNšðOúàŸ™ Ó(“Að)ûô &ô Ø˜•	”ðúsY   –A9 Á9C2ÂBÂC2Â"CÃ 
CÃ
CÃC2ÃCÃC2Ã.C5 Ã1C2Ã2C5 Ã5DÄDc                s  € V\        V4      8  g   Q h^ pW,          P                  4       '       dV   ^pW#,           \        V4      8  d>   \        P                  ! WV,           ,          4      ^ ,          R8w  d    V# V^,          pKS  V# )a  Consume an alphabetic character from the given unicode string.

Given a unicode string and the current offset, this method returns
the number of characters occupied by the next alphabetic character
in the string.  Trailing combining characters are consumed as a
single letter.
r   )r   r   r   r   )r   r   r   r   r   r	   r   Ztokenize._consume_alpha_u‹   sv   € ð œ˜D›	Ô!Ð!Ð!ØˆØ<×Ñ×!Ò!ØˆDØ•-¤# d£)Ô+Ü×'Ò'¨°d­]Õ(;Ó<¸QÕ?À3ÔFØàˆð ˜•	’Øˆr   c                sö  € V P                   pV P                  pV\        V4      8  dÃ   V\        V4      8  d&   V P                  W4      pV'       d   MV^,          pK5  TpV\        V4      8  d?   V P                  W4      pV'       g   W,          V P                  9   d   ^pMM
W#,          pKN  WB8w  g   K  W^,
          ,          V P                  9   d   V^,
          pK)  W n        WV V3# W n        \        4       h)i   )r   r   r   r   r   ZStopIteration)r   r   r   r   Zcur_poss   &    r	   ÚnextZtokenize.next   sÔ   € Øz‰zˆØ—‘ˆØ”s˜4“yÔ àœ3˜t›9Ô$Ø×*Ñ*¨4Ó8ßØØ˜!•’ØˆGàœ3˜t›9Ô$Ø×*Ñ*¨4Ó8ßØ•| t×'8Ñ'8Ô8Ø ™àØ•’àÖ à A:Õ&¨$×*;Ñ*;Ô;Ø# aZ’FØ%”Ø VÐ,¨gÐ6Ð6ØŒÜ‹oÐr   )r   r   r   r   )N)Z__name__Z
__module__Z__qualname__Z__firstlineno__Ú__doc__Z_DOC_ERRORSr
   r   r   r   r   r   r   Z__static_attributes__Z__classdictcell__)Z__classdict__s   @r	   r    r    ,   s=   ø‡ € ñð" ˜%.€Kô.ò('ò
'òòò4÷$ð r   r    )r   r   Zenchant.tokenizeZenchantr    ) r   r	   Ú<module>r      s,   ðñ<ó ã ôMˆw×Ñ×(Ñ(ö Mr   