Ë
    ¡[;jð,  ã                  ó    — d dl mZ d dlmZmZ d dlmZmZ d dlm	Z	m
Z
 dddœd„Z	 dd„Zdddœd	„Zdddœd
„Zdddœd„Zd„ Zddœd„Zddœd„Zy)é    )Úannotations)Úcommon_affixÚconv_sequences)Úis_noneÚsetupPandas)ÚEditopÚEditopsN)Ú	processorÚscore_cutoffc               óf  — |� || «      }  ||«      }| syt        | |«      \  } }dt        | «      z  dz
  }i }|j                  }d}| D ]  } ||d«      |z  ||<   |dz  }Œ |D ]  }	 ||	d«      }
||
z  }||z   ||z
  z  }Œ t        |«      t        | «       d j	                  d«      }|�||k\  r|S dS )aâ  
    Calculates the length of the longest common subsequence

    Parameters
    ----------
    s1 : Sequence[Hashable]
        First string to compare.
    s2 : Sequence[Hashable]
        Second string to compare.
    processor: callable, optional
        Optional callable that is used to preprocess the strings before
        comparing them. Default is None, which deactivates this behaviour.
    score_cutoff : int, optional
        Maximum distance between s1 and s2, that is
        considered as a result. If the similarity is smaller than score_cutoff,
        0 is returned instead. Default is None, which deactivates
        this behaviour.

    Returns
    -------
    similarity : int
        similarity between s1 and s2
    Nr   é   Ú0)r   ÚlenÚgetÚbinÚcount)Ús1Ús2r
   r   ÚSÚblockÚ	block_getÚxÚch1Úch2ÚMatchesÚuÚress                úeG:\00. PROJECTS\API\Inventory\templateJSON\kerjaOCR\Lib\site-packages\rapidfuzz/distance/LCSseq_py.pyÚ
similarityr   
   sì   € ð< ÐÙ�r‹]ˆÙ�r‹]ˆáØä˜B Ó#�F€BˆØ	
Œc�"‹g‰˜Ñ€AØ€EØ—	‘	€IØ	€AÛˆÙ˜s AÓ&¨Ñ*ˆˆc‰
Ø	ˆa‰‰ð ó ˆÙ˜C Ó#ˆØ�‰KˆØ�‰U�q˜1‘uÑ‰ð ô ˆa‹&”#�b“'��Ð
×
"Ñ
" 3Ó
'€CØÐ'¨3°,Ò+>ˆ3ÐFÀQÐFó    c                óæ   — |sydt        |«      z  dz
  }| j                  }|D ]  } ||d«      }||z  }||z   ||z
  z  }Œ t        |«      t        |«       d  j                  d«      }	|�|	|k\  r|	S dS ©Nr   r   r   )r   r   r   r   )
r   r   r   r   r   r   r   r   r   r   s
             r   Ú_block_similarityr#   B   s�   € ñ Øà	
Œc�"‹g‰˜Ñ€AØ—	‘	€IãˆÙ˜C Ó#ˆØ�‰KˆØ�‰U�q˜1‘uÑ‰ð ô ˆa‹&”#�b“'��Ð
×
"Ñ
" 3Ó
'€CØÐ'¨3°,Ò+>ˆ3ÐFÀQÐFr    c               ó¾   — |� || «      }  ||«      }t        | |«      \  } }t        t        | «      t        |«      «      }t        | |«      }||z
  }|�||k  r|S |dz   S )aŒ  
    Calculates the LCS distance in the range [0, max].

    This is calculated as ``max(len1, len2) - similarity``.

    Parameters
    ----------
    s1 : Sequence[Hashable]
        First string to compare.
    s2 : Sequence[Hashable]
        Second string to compare.
    processor: callable, optional
        Optional callable that is used to preprocess the strings before
        comparing them. Default is None, which deactivates this behaviour.
    score_cutoff : int, optional
        Maximum distance between s1 and s2, that is
        considered as a result. If the distance is bigger than score_cutoff,
        score_cutoff + 1 is returned instead. Default is None, which deactivates
        this behaviour.

    Returns
    -------
    distance : int
        distance between s1 and s2

    Examples
    --------
    Find the LCS distance between two strings:

    >>> from rapidfuzz.distance import LCSseq
    >>> LCSseq.distance("lewenstein", "levenshtein")
    2

    Setting a maximum distance allows the implementation to select
    a more efficient implementation:

    >>> LCSseq.distance("lewenstein", "levenshtein", score_cutoff=1)
    2

    r   )r   Úmaxr   r   )r   r   r
   r   ÚmaximumÚsimÚdists          r   Údistancer)   X   ss   € ð^ ÐÙ�r‹]ˆÙ�r‹]ˆä˜B Ó#�F€BˆÜ”#�b“'œ3˜r›7Ó#€GÜ
�R˜Ó
€CØ�S‰=€DØ Ð(¨D°LÒ,@ˆ4ÐWÀ|ÐVWÑGWÐWr    c               ó   — t        «        t        | «      st        |«      ry|� || «      }  ||«      }| r|syt        | |«      \  } }t        t	        | «      t	        |«      «      }t        | |«      |z  }|�||k  r|S dS )a2  
    Calculates a normalized LCS similarity in the range [1, 0].

    This is calculated as ``distance / max(len1, len2)``.

    Parameters
    ----------
    s1 : Sequence[Hashable]
        First string to compare.
    s2 : Sequence[Hashable]
        Second string to compare.
    processor: callable, optional
        Optional callable that is used to preprocess the strings before
        comparing them. Default is None, which deactivates this behaviour.
    score_cutoff : float, optional
        Optional argument for a score threshold as a float between 0 and 1.0.
        For norm_dist > score_cutoff 1.0 is returned instead. Default is 1.0,
        which deactivates this behaviour.

    Returns
    -------
    norm_dist : float
        normalized distance between s1 and s2 as a float between 0 and 1.0
    ç      ð?r   r   )r   r   r   r%   r   r)   )r   r   r
   r   r&   Únorm_sims         r   Únormalized_distancer-   ’   s…   € ô> „MÜˆr„{”g˜b”kØàÐÙ�r‹]ˆÙ�r‹]ˆá‘RØä˜B Ó#�F€BˆÜ”#�b“'œ3˜r›7Ó#€GÜ˜˜BÓ 'Ñ)€HØ$Ð,°¸LÒ0Hˆ8ÐPÈqÐPr    c               óœ   — t        «        t        | «      st        |«      ry|� || «      }  ||«      }dt        | |«      z
  }|�||k\  r|S dS )a�  
    Calculates a normalized LCS similarity in the range [0, 1].

    This is calculated as ``1 - normalized_distance``

    Parameters
    ----------
    s1 : Sequence[Hashable]
        First string to compare.
    s2 : Sequence[Hashable]
        Second string to compare.
    processor: callable, optional
        Optional callable that is used to preprocess the strings before
        comparing them. Default is None, which deactivates this behaviour.
    score_cutoff : float, optional
        Optional argument for a score threshold as a float between 0 and 1.0.
        For norm_sim < score_cutoff 0 is returned instead. Default is 0,
        which deactivates this behaviour.

    Returns
    -------
    norm_sim : float
        normalized similarity between s1 and s2 as a float between 0 and 1.0

    Examples
    --------
    Find the normalized LCS similarity between two strings:

    >>> from rapidfuzz.distance import LCSseq
    >>> LCSseq.normalized_similarity("lewenstein", "levenshtein")
    0.8181818181818181

    Setting a score_cutoff allows the implementation to select
    a more efficient implementation:

    >>> LCSseq.normalized_similarity("lewenstein", "levenshtein", score_cutoff=0.9)
    0.0

    When a different processor is used s1 and s2 do not have to be strings

    >>> LCSseq.normalized_similarity(["lewenstein"], ["levenshtein"], processor=lambda s: s[0])
    0.81818181818181
    g        r+   r   )r   r   r-   )r   r   r
   r   r,   s        r   Únormalized_similarityr/   Â   s[   € ôd „MÜˆr„{”g˜b”kØàÐÙ�r‹]ˆÙ�r‹]ˆàÔ(¨¨RÓ0Ñ0€HØ$Ð,°¸LÒ0Hˆ8ÐPÈqÐPr    c                óB  — | sdg fS dt        | «      z  dz
  }i }|j                  }d}| D ]  } ||d«      |z  ||<   |dz  }Œ g }|D ],  } ||d«      }	||	z  }
||
z   ||
z
  z  }|j                  |«       Œ. t        |«      t        | «       d  j	                  d«      }||fS r"   )r   r   Úappendr   r   )r   r   r   r   r   r   r   Úmatrixr   r   r   r'   s               r   Ú_matrixr3      sÌ   € ÙØ�2ˆwˆà	
Œc�"‹g‰˜Ñ€AØ€EØ—	‘	€IØ	€AÛˆÙ˜s AÓ&¨Ñ*ˆˆc‰
Ø	ˆa‰‰ð ð €FÛˆÙ˜C Ó#ˆØ�‰KˆØ�‰U�q˜1‘uÑˆØ�‰�aÕð	 ô ˆa‹&”#�b“'��Ð
×
"Ñ
" 3Ó
'€CØ�ˆ=Ðr    ©r
   c               ót  — |� || «      }  ||«      }t        | |«      \  } }t        | |«      \  }}| |t        | «      |z
   } ||t        |«      |z
   }t        | |«      \  }}t	        g dd«      }t        | «      |z   |z   |_        t        |«      |z   |z   |_        t        | «      t        |«      z   d|z  z
  }|dk(  r|S dg|z  }	t        | «      }
t        |«      }|dk7  r{|
dk7  rv||dz
     d|
dz
  z  z  r!|dz  }|
dz  }
t        d|
|z   ||z   «      |	|<   n9|dz  }|r-||dz
     d|
dz
  z  z  s|dz  }t        d|
|z   ||z   «      |	|<   n|
dz  }
|dk7  r|
dk7  rŒv|
dk7  r&|dz  }|
dz  }
t        d|
|z   ||z   «      |	|<   |
dk7  rŒ&|dk7  r&|dz  }|dz  }t        d|
|z   ||z   «      |	|<   |dk7  rŒ&|	|_        |S )uf  
    Return Editops describing how to turn s1 into s2.

    Parameters
    ----------
    s1 : Sequence[Hashable]
        First string to compare.
    s2 : Sequence[Hashable]
        Second string to compare.
    processor: callable, optional
        Optional callable that is used to preprocess the strings before
        comparing them. Default is None, which deactivates this behaviour.

    Returns
    -------
    editops : Editops
        edit operations required to turn s1 into s2

    Notes
    -----
    The alignment is calculated using an algorithm of Heikki HyyrÃ¶, which is
    described in [6]_. It has a time complexity and memory usage of ``O([N/64] * M)``.

    References
    ----------
    .. [6] HyyrÃ¶, Heikki. "A Note on Bit-Parallel Alignment Computation."
           Stringology (2004).

    Examples
    --------
    >>> from rapidfuzz.distance import LCSseq
    >>> for tag, src_pos, dest_pos in LCSseq.editops("qabxcd", "abycdf"):
    ...    print(("%7s s1[%d] s2[%d]" % (tag, src_pos, dest_pos)))
     delete s1[0] s2[0]
     delete s1[3] s2[2]
     insert s1[4] s2[2]
     insert s1[6] s2[5]
    Nr   é   r   ÚdeleteÚinsert)	r   r   r   r3   r	   Ú_src_lenÚ	_dest_lenr   Ú_editops)r   r   r
   Ú
prefix_lenÚ
suffix_lenr'   r2   Úeditopsr(   Úeditop_listÚcolÚrows               r   r>   r>     s?  € ðX ÐÙ�r‹]ˆÙ�r‹]ˆä˜B Ó#�F€BˆÜ)¨"¨bÓ1Ñ€J�
Ø	ˆJœ˜R› :Ñ-Ð	.€BØ	ˆJœ˜R› :Ñ-Ð	.€BÜ˜"˜b“/�K€Cˆä�b˜!˜QÓ€GÜ˜2“w Ñ+¨jÑ8€GÔÜ˜B› *Ñ,¨zÑ9€GÔäˆr‹7”S˜“WÑ˜q 3™wÑ&€DØˆq‚yØˆà�&˜4‘-€KÜ
ˆb‹'€CÜ
ˆb‹'€CØ
�Š(�s˜a’xà�#˜‘'‰?˜a C¨!¡G™nÒ-Ø�A‰IˆDØ�1‰HˆCÜ & x°°zÑ1AÀ3ÈÑCSÓ TˆK˜Òà�1‰HˆCñ ˜F 3¨¡7™O¨q°S¸1±W©~Ò>Ø˜‘	�Ü$*¨8°S¸:Ñ5EÀsÈZÑGWÓ$X�˜DÒ!ð �q‘�ð �Š(�s˜a“xð" �Š(Ø�‰	ˆØˆq‰ˆÜ" 8¨S°:Ñ-=¸sÀZÑ?OÓPˆ�DÑð �‹(ð
 �Š(Ø�‰	ˆØˆq‰ˆÜ" 8¨S°:Ñ-=¸sÀZÑ?OÓPˆ�DÑð �‹(ð
 #€GÔØ€Nr    c               ó:   — t        | ||¬«      j                  «       S )u  
    Return Opcodes describing how to turn s1 into s2.

    Parameters
    ----------
    s1 : Sequence[Hashable]
        First string to compare.
    s2 : Sequence[Hashable]
        Second string to compare.
    processor: callable, optional
        Optional callable that is used to preprocess the strings before
        comparing them. Default is None, which deactivates this behaviour.

    Returns
    -------
    opcodes : Opcodes
        edit operations required to turn s1 into s2

    Notes
    -----
    The alignment is calculated using an algorithm of Heikki HyyrÃ¶, which is
    described in [7]_. It has a time complexity and memory usage of ``O([N/64] * M)``.

    References
    ----------
    .. [7] HyyrÃ¶, Heikki. "A Note on Bit-Parallel Alignment Computation."
           Stringology (2004).

    Examples
    --------
    >>> from rapidfuzz.distance import LCSseq

    >>> a = "qabxcd"
    >>> b = "abycdf"
    >>> for tag, i1, i2, j1, j2 in LCSseq.opcodes(a, b):
    ...    print(("%7s a[%d:%d] (%s) b[%d:%d] (%s)" %
    ...           (tag, i1, i2, a[i1:i2], j1, j2, b[j1:j2])))
     delete a[0:1] (q) b[0:0] ()
      equal a[1:3] (ab) b[0:2] (ab)
     delete a[3:4] (x) b[2:2] ()
     insert a[4:4] () b[2:3] (y)
      equal a[4:6] (cd) b[3:5] (cd)
     insert a[6:6] () b[5:6] (f)
    r4   )r>   Ú
as_opcodes)r   r   r
   s      r   ÚopcodesrD   x  s   € ôd �2�r YÔ/×:Ñ:Ó<Ð<r    )N)Ú
__future__r   Úrapidfuzz._common_pyr   r   Úrapidfuzz._utilsr   r   Ú!rapidfuzz.distance._initialize_pyr   r	   r   r#   r)   r-   r/   r3   r>   rD   © r    r   Ú<module>rJ      su   ðõ #ç =ß 1ß =ð Øô5Gðx ó	Gð4 Øô7Xð| Øô-Qðh Øô;Qò|ð8 ô	]ðH õ	2=r    