Logo Lanfrica

Abe2G/Amharic-Simple-Text-Preprocessing-Usin-Python

Domain:

natural language processing
Creator:
Abe
Host:
Amharic text preprocessing. Hope this can help you. # Amharic-Simple-Text-Preprocessing-Usin-Python # Short Form Expansion and Character Level Normalization import re class normalize(object): expansion_file_dir='' # assume you have file with list of short forms with their expansion as gazeter short_form_dict={} # Constructor def __init__(self): self.short_form_dict=self.get_short_forms() def get_short_forms(self): text=open(self.expansion_file_dir,encoding='utf8') exp={} for line in iter(text): line=line.strip() if not line: # line is blank continue else: expanded=line.split("-") exp[expanded[0].strip()]=expanded[1].replace(" ",'_').strip() return exp # method that expand short form file def expand_short_form(self,input_short_word): if input_short_word in self.short_form_dict: return self.short_form_dict[input_short_word] else: return input_short_word #method to normalize character level missmatch such as ጸሀይ and ፀሐይ def normalize_char_level_missmatch(self,input_token): rep1=re.sub('[ሃኅኃሐሓኻ]','ሀ',input_token) rep2=re.sub('[ሑኁዅ]','ሁ',rep1) rep3=re.sub('[ኂሒኺ]','ሂ',rep2) rep4=re.sub('[ኌሔዄ]','ሄ',rep3) rep5=re.sub('[ሕኅ]','ህ',rep4) rep6=re.sub('[ኆሖኾ]','ሆ',rep5) rep7=re.sub('[ሠ]','ሰ',rep6) rep8=re.sub('[ሡ]','ሱ',rep7) rep9=re.sub('[ሢ]','ሲ',rep8) rep10=re.sub('[ሣ]','ሳ',rep9) rep11=re.sub('[ሤ]','ሴ',rep10) rep12=re.sub('[ሥ]','ስ',rep11) rep13=re.sub('[ሦ]','ሶ',rep12) rep14=re.sub('[ዓኣዐ]','አ',rep13) rep15=re.sub('[ዑ]','ኡ',rep14) rep16=re.sub('[ዒ]','ኢ',rep15) rep17=re.sub('[ዔ]','ኤ',rep16) rep18=re.sub('[ዕ]','እ',rep17) rep19=re.sub('[ዖ]','ኦ',rep18) rep20=re.sub('[ጸ]','ፀ',rep19) rep21=re.sub('[ጹ]','ፁ',rep20) rep22=re.sub('[ጺ]','ፂ',rep21) rep23=re.sub('[ጻ]','ፃ',rep22) rep24=re.sub('[ጼ]','ፄ',rep23) rep25=re.sub('[ጽ]','ፅ',rep24) rep26=re.sub('[ጾ]','ፆ',rep25) #Normalizing words with Labialized Amharic characters such as በልቱዋል or በልቱአል to በልቷል rep27=re.sub('(ሉ[ዋአ])','ሏ',rep26) rep28=re.sub('(ሙ[ዋአ])','ሟ',rep27) rep29=re.sub('(ቱ[ዋአ])','ቷ',rep28) rep30=re.sub('(ሩ[ዋአ])','ሯ',rep29) rep31=re.sub('(ሱ[ዋአ])','ሷ',rep30) rep32=re.sub('(ሹ[ዋአ …