23 lines
671 B
Python
23 lines
671 B
Python
#!/usr/bin/env python3
|
|
# -*- coding: utf-8 -*-
|
|
|
|
import textdistance
|
|
|
|
|
|
#
|
|
# Name matching via textual similarity search
|
|
# + Returns two (normalised) distance measures: the Jaro-Winkler Distance and the regular Levenshtein Distance
|
|
#
|
|
def match_name_textualsim(name1, name2):
|
|
jaro_winkler = textdistance.jaro_winkler.normalized_similarity(name1, name2)
|
|
levenshtein = textdistance.levenshtein.normalized_similarity(name1, name2)
|
|
|
|
return jaro_winkler, levenshtein
|
|
|
|
|
|
#
|
|
# Name matching via phonetic matching algorithm (using the normalised Match Rating Approach)
|
|
#
|
|
def match_name_mra(name1, name2):
|
|
return textdistance.mra.normalized_similarity(name1, name2)
|