1 changed files with 88 additions and 0 deletions
@ -0,0 +1,88 @@ |
|||
import re |
|||
import os |
|||
|
|||
""" |
|||
Replace every space with a special character and then re-replace them? You lose the space between `:«` with current method. |
|||
""" |
|||
|
|||
path = __file__ |
|||
|
|||
directory = os.path.abspath(os.getcwd()) |
|||
|
|||
def add_tags(filename): |
|||
# Only grab 0XX-canticle from `*.txt` file names |
|||
xmlfile = filename.split('.')[0] + '.xml' |
|||
tmpxmlfile = '.' + filename.split('.')[0] + '.tmp.xml' |
|||
with open(xmlfile,'w') as Cxml: |
|||
with open(tmpxmlfile,'w') as Cxmltmp: |
|||
with open(filename,'r') as Ctxt: |
|||
for spaced_line in Ctxt: |
|||
# Use `~` as a placeholder for spaces later |
|||
line = re.sub(' ','~',spaced_line) |
|||
words = line.split('~') |
|||
for i in range(len(words)): |
|||
word = words[i] |
|||
# Iterate from ending to beginning of word |
|||
L = len(word) - 1 |
|||
# Catch single chars like `-` |
|||
if L == 0: |
|||
before_end_word_chars = word |
|||
ending_punct_chars = '' |
|||
# Catch single chars at end of line |
|||
elif L == 1 and '\n' in word: |
|||
before_end_word_chars = word.rstrip() |
|||
ending_punct_chars = '\n' |
|||
else: |
|||
print('Got to <- with: ' + word + ' and this L: ' + str(L) + ' in ' + filename) |
|||
# We need `word[L] != '-'` because there are some words like `-:` |
|||
while word[L] != '’' and word[L] != '-' and not word[L].isalpha(): |
|||
L-=1 |
|||
else: |
|||
ending_punct_begin_index = L + 1 |
|||
ending_punct_chars = word[ending_punct_begin_index:] |
|||
before_end_word_chars = word[:ending_punct_begin_index] |
|||
# Iterate from start to remainder of word |
|||
L = len(word) - 1 |
|||
if L == 0: |
|||
middle_word_chars = word |
|||
starting_punct_chars = '' |
|||
elif L == 1 and '\n' in word: |
|||
middle_word_chars = word.rstrip() |
|||
starting_punct_chars = '' |
|||
else: |
|||
#print('Got -> with this word: ' + word + ' and this L: ' + str(L) + ' in ' + filename) |
|||
while word[-L - 1] != '‘' and word[-L - 1] != '’' and word[L] != '-' and not word[-L - 1].isalpha(): |
|||
L-=1 |
|||
else: |
|||
starting_punct_end_index = -L - 1 |
|||
starting_punct_chars = word[:starting_punct_end_index] |
|||
middle_word_chars = before_end_word_chars[-L-1:] |
|||
# Pad with spaces to ensure we get spaces between words |
|||
# and importantly _around_ punctuation. We want: |
|||
# `<w>Rispuosemi</w>: «<w>Non</w>` and not `<w>Rispuosemi</w>:«<w>Non</w>`. |
|||
new_word = ' ' + starting_punct_chars + '<w>' + middle_word_chars + '</w>' + ending_punct_chars + ' ' |
|||
Cxmltmp.write(new_word) |
|||
Ctxt.close() |
|||
Cxmltmp.close() |
|||
with open(tmpxmlfile,'r') as Cxmltmp: |
|||
# Add space before opening word tag |
|||
def add_space(match): |
|||
match = match.group() |
|||
return re.sub('<w>', ' <w>', match) |
|||
for line in Cxmltmp: |
|||
# Swap double spaces from new_word with single spaces |
|||
line_spaced = re.sub(' ',' ',line) |
|||
# If there's a single space at the beginning of the line, delete it |
|||
line_spaced = line_spaced.lstrip(' ') |
|||
# Add three tabs at the beginning |
|||
line_spaced = re.sub('^','\t\t\t',line_spaced) |
|||
Cxml.write(line_spaced) |
|||
Cxmltmp.close() |
|||
os.remove(tmpxmlfile) |
|||
Cxml.close() |
|||
|
|||
for f in os.listdir(directory): |
|||
if f != 'word-tags.py' and f != '.word-tags.py.swp': |
|||
add_tags(f) |
|||
if f.endswith('.txt'): |
|||
os.remove(os.path.join(directory, f)) |
|||
Loading…
Reference in new issue