diff --git a/scripts.d/py-scripts.d/parser-txt-to-xml.py b/scripts.d/py-scripts.d/parser-txt-to-xml.py
new file mode 100644
index 0000000..63b561c
--- /dev/null
+++ b/scripts.d/py-scripts.d/parser-txt-to-xml.py
@@ -0,0 +1,247 @@
+import os
+import re # find and replace
+import shutil # move tmp file for overwrite
+
+cantoDir = os.path.abspath(os.getcwd())
+
+def getXMLFileName(xmlCantoFile):
+ xmlCanto = xmlCantoFile.split('.')[0] + '.xml'
+ return xmlCanto
+
+def addWordTags(cantoFile):
+ """
+ - [x] Parse filename to create xml file
+ - [x] Open file as one huge string (keep track of newlines)
+ - [x] Replace every space with a dummy char not used in the text, like `~`
+ - [ ] Traverse the big file-string from beginning to end
+ - [x] Define a "word-chars" group: letters, apostrofes
+ - [x] If you encounter a char that's a word-char
+ put a `` in front of it and move to the next.
+ - [x] If the next char is also a word-char, move on
+ to the next because we're still in a word.
+ - [x] If the next char is not a word-char,
+ put a `` in front of it and move to the next.
+ - [x] Replace all `~` with spaces and get back the newlines
+ - [x] Write to XML file
+ """
+ xmlCanto = getXMLFileName(cantoFile)
+ with open(xmlCanto, 'w') as xmlCanto:
+ with open(cantoFile, 'r') as txtCanto:
+ originalText = txtCanto.read()
+ # Prepare canto for traversing
+ plainText = traversePreparation(originalText, '\n', '+', False)
+ plainText = traversePreparation(plainText, ' ', '~', False)
+
+ # Define variables
+ wordChars = '[a-zA-ZÀ-ÿ0-9‘’]'
+ openWord = ''
+ closeWord = ''
+ insideWord = False
+
+ # Traverse canto
+ for char in plainText:
+ if re.match(wordChars, char):
+ if insideWord == False:
+ xmlCanto.write(openWord)
+ xmlCanto.write(char)
+ insideWord = True
+ else:
+ xmlCanto.write(char)
+ else:
+ if char == '~':
+ if insideWord == True:
+ xmlCanto.write(closeWord + ' ')
+ insideWord = False
+ else:
+ xmlCanto.write(' ')
+ elif char == '+':
+ if insideWord == True:
+ xmlCanto.write(closeWord + '\n') # line ends with word
+ insideWord = False
+ else:
+ xmlCanto.write('\n') # line ends with punctuation
+ elif insideWord == True:
+ xmlCanto.write(closeWord + char) # punctuation right after to word
+ insideWord = False
+ else:
+ xmlCanto.write(char) # punctuation after punctuation
+ txtCanto.close()
+ xmlCanto.close()
+
+def traversePreparation(cantoText, find, replace, negate):
+ """
+ Substitute find with replace.
+ If negate is True, substitute replace with find.
+ """
+ if negate == False:
+ replacedText = cantoText.replace(find, replace)
+ else:
+ replacedText = cantoText.replace(replace, find)
+ return replacedText
+
+def addLineTags(xmlCantoFile):
+ """
+ add to the beginning of each line number n
+ add to the end of each line
+ """
+ xmlCanto = getXMLFileName(xmlCantoFile)
+ tmpFile = '.' + xmlCantoFile + '.tmp'
+ with open(xmlCanto, 'r') as xmlCanto:
+ with open(tmpFile, 'w') as tmpCanto:
+ i = 1
+ for line in xmlCanto:
+ # replace '\n' in line with ' \n'
+ newLine = re.sub('\n', ' \n', line)
+ newLine = ' ' + newLine
+ i += 1
+ tmpCanto.write(newLine)
+ closeCopyRemoveTmp(tmpFile, tmpCanto, xmlCantoFile, xmlCanto)
+
+def addTercetTags(xmlCantoFile):
+ """
+ add ... around each group of three lines until the end
+ """
+ xmlCanto = getXMLFileName(xmlCantoFile)
+ tmpFile = '.' + xmlCantoFile + '.tmp'
+ with open(xmlCanto, 'r') as xmlCanto:
+ lastLine = xmlCanto.readlines()[-1]
+ lastLineNum = int(lastLine.split('"')[1])
+ xmlCanto.seek(0)
+ with open(tmpFile, 'w') as tmpCanto:
+ i = 0
+ j = 2
+ for line in xmlCanto:
+ #print(line)
+ i += 1
+ #print('new i: ' + str(i))
+ if i == 1:
+ newLine = '\n' + line
+ tmpCanto.write(newLine)
+ #print('first line newline: ' + newLine)
+ elif i % 3 == 0:
+ tercetNum = str(j)
+ newLineSub = '\n\n\n'
+ newLine = re.sub('\n', newLineSub, line)
+ tmpCanto.write(newLine)
+ j += 1
+ #print('every 3 lines newline: ' + newLine)
+ elif i == lastLineNum:
+ newLine = re.sub('\n', '\n', line)
+ tmpCanto.write(newLine)
+ #print('last line newline: ' + newLine)
+ else:
+ tmpCanto.write(line)
+ closeCopyRemoveTmp(tmpFile, tmpCanto, xmlCantoFile, xmlCanto)
+
+def wrapTags(tag, xmlCantoFile):
+ """
+ used to add , , and tags around body
+ """
+ xmlCanto = getXMLFileName(xmlCantoFile)
+ if tag == 'TEI':
+ openTag = '\n'
+ else:
+ openTag = '<' + tag + '>\n'
+ closeTag = '\n' + tag + '>'
+ with open(xmlCanto, 'r') as xmlCantoContents:
+ contents = xmlCantoContents.read()
+ with open(xmlCanto, 'w') as xmlCantoContents:
+ xmlCantoContents.write(openTag)
+ xmlCantoContents.write(contents)
+ xmlCantoContents.write(closeTag)
+ xmlCantoContents.close()
+
+def addTEIHeaderTags(xmlCantoFile):
+ """
+ The following is the minimum required TEI header
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+ """
+ xmlCanto = getXMLFileName(xmlCantoFile)
+
+def addSpaces(xmlCantoFile):
+ """
+ Add a given number of spaces before each opening and closing tag.
+ User specifies the number of spaces only once, in a series of lists
+ which hold strings of each tag that needs spaces before it. This
+ function parses the names of the list to determine the number of
+ spaces to add before each tag (e.g. list2 => 2 spaces). It replaces
+ the tags in the document with the spaces affixed.
+ """
+ # XML is hierarchical, so each "level" needs its own number of spaces
+ levelDict = {
+ '0' : ['TEI'],
+ '1' : ['teiHeader', 'text' ],
+ '2' : ['fileDesc' , 'canto' ],
+ '3' : ['titleStmt', 'tercets'],
+ '4' : [ 'tercet' ],
+ '5' : [ 'l' ]
+ }
+
+ #levelDictLen = int(sorted(levelDict.keys())[-1]) + 1 # number of levels. add 1 to make range() work below
+ tags = ['TEI', 'teiHeader', 'text', 'fileDesc', 'canto', 'titleStmt', 'tercets', 'tercet', 'l' ]
+
+ xmlCanto = getXMLFileName(xmlCantoFile)
+ tmpFile = '.' + xmlCantoFile + '.tmp'
+
+ with open(xmlCantoFile, 'r') as xmlCanto:
+ with open(tmpFile, 'w') as tmpCanto:
+ for line in xmlCanto:
+ for tag in tags:
+ # need open tag regex b/c we don't know if tag has attributes,
+ # and if we did, we might not know their value
+ openTagRegex = '(<' + tag + ')(>|\s.*>)' # match and but not
+ closeTag = '' + tag + '>'
+ for level in levelDict.keys():
+ if tag in levelDict[level]:
+ spaces = ' ' * int(level)
+ if re.search(openTagRegex, line):
+ newLine = re.sub(openTagRegex, spaces + r"\1\2", line)
+ tmpCanto.write(newLine)
+ elif closeTag in line:
+ if tag != 'l': # don't add spaces to closing line tags
+ newLine = re.sub(closeTag, spaces + closeTag, line)
+ tmpCanto.write(newLine)
+ closeCopyRemoveTmp(tmpFile, tmpCanto, xmlCantoFile, xmlCanto)
+
+def closeCopyRemoveTmp(tmpFile, tmpCanto, xmlCantoFile, xmlCanto):
+ tmpCanto.close()
+ xmlCanto.close()
+ shutil.copy(tmpFile, xmlCantoFile)
+ os.remove(tmpFile)
+
+def main():
+ for cantoFile in os.listdir(cantoDir):
+ xmlCantoName = getXMLFileName(cantoFile)
+ if re.search('\.txt$', cantoFile):
+ addWordTags(cantoFile)
+ addLineTags(xmlCantoName)
+ addTercetTags(xmlCantoName)
+ wrapTags('tercets', xmlCantoName)
+ wrapTags('canto', xmlCantoName)
+ wrapTags('text', xmlCantoName)
+ #addTEIHeaderTags()
+ wrapTags('TEI', xmlCantoName)
+ addSpaces(xmlCantoName)
+ os.remove(cantoFile) # don't delete txt files while debugging so you don't have to keep copying them
+
+
+main()
diff --git a/scripts.d/py-scripts.d/remove-txt-blank-lines.py b/scripts.d/py-scripts.d/remove-txt-blank-lines.py
new file mode 100644
index 0000000..c2667b0
--- /dev/null
+++ b/scripts.d/py-scripts.d/remove-txt-blank-lines.py
@@ -0,0 +1,24 @@
+import shutil
+import os
+import re
+
+cantoDir = os.path.abspath(os.getcwd())
+
+def main():
+ """
+ Take txt file from `dante-visualized` project and remove blank lines
+ """
+ for txtFile in os.listdir(cantoDir):
+ if re.search('\.txt$', txtFile):
+ tmpFile = '.' + txtFile+ '.tmp'
+ with open(txtFile, 'r') as txt:
+ with open(tmpFile, 'w') as tmp:
+ for line in txt:
+ if line.rstrip():
+ tmp.write(line)
+ tmp.close()
+ txt.close()
+ shutil.copy(tmpFile, txtFile)
+ os.remove(tmpFile)
+
+main()
diff --git a/scripts.d/py-scripts.d/rename-txt.py b/scripts.d/py-scripts.d/rename-txt.py
new file mode 100644
index 0000000..f0a198d
--- /dev/null
+++ b/scripts.d/py-scripts.d/rename-txt.py
@@ -0,0 +1,22 @@
+import re
+import os
+
+canto_dir = os.path.abspath(os.getcwd())
+canto_files = os.listdir(canto_dir)
+
+def rename_canto(canto):
+ """
+ Take txt canto files from `dante-visualized` and change the name styling
+ """
+ # parse canto name
+ # put it back together
+ print('canto name: ' + canto)
+ new_canto = re.sub('\_','-',canto)
+ new_canto = re.sub('Canto[IVX]*-','',new_canto)
+ new_canto = new_canto.lower()
+ print('--> new canto name: ' + new_canto)
+ os.rename(canto,new_canto)
+
+for canto in canto_files:
+ if canto != 'rename-txt.py' and canto != '.rename-txt.py.swp':
+ rename_canto(canto)
diff --git a/scripts.d/py-scripts.d/word-count-generator.py b/scripts.d/py-scripts.d/word-count-generator.py
new file mode 100644
index 0000000..e3f0ec1
--- /dev/null
+++ b/scripts.d/py-scripts.d/word-count-generator.py
@@ -0,0 +1,124 @@
+import re
+import os
+import xml.etree.ElementTree as ET
+
+"""
+This script takes xml canto files and counts the number of unique words as
+defined by the value of `` nodes.
+
+8 files are generated: csv and html of
+ - sorted numerically and alphabetically
+ - sorted numerically and alphabetically with blacklist applied
+"""
+
+cantoDir = os.path.abspath(os.getcwd())
+
+blacklist = ('e', 'che', 'la', 'di', 'a', 'per', '’l', 'mi', 'tu', 'con',
+ 'lo', 'sì', 'sua', 'li', 'del', 'si', 'nel', 'era', 'ch’i’', 'è',
+ 'là', 'le', 'de', 'quella', 'poi', 'ma', 'il', 'già', '’n', 'una',
+ 'tanto', 'se’', 'quel', 'qual', 'perché', 'in', 'come', 'al', 'da',
+ 'un', 'i', 'ti', 'a', 'per', 'con', 'lo', 'al', 'poi', 'E', 'il')
+
+def createWordCountDict(cantoFile):
+ """
+ for node in xpathQueryAllWordNodes:
+ if xpathValue(node) in wordCountDict(keys):
+ wordCountDict[node] += 1
+ else:
+ wordCountDict[xpathValue(node)] = 1
+ """
+ wordCountDict = dict()
+ tree = ET.parse(cantoFile)
+ root = tree.getroot()
+ # TODO
+ words = root.findall(".//{http://www.tei-c.org/ns/1.0}w")
+ for word in words:
+ if word.text in wordCountDict:
+ wordCountDict[word.text] += 1
+ else:
+ wordCountDict[word.text] = 1
+ return wordCountDict
+
+def numSort(wordCountDict):
+ # Need reverse = True to sort greatest to least
+ wordCountDictNumSorted = dict(sorted(wordCountDict.items(), key=lambda item: item[1], reverse = True))
+ return wordCountDictNumSorted
+
+def alphaSort(wordCountDict):
+ # Capital letters come first with this implementation
+ wordCountDictAlphaSorted = dict(sorted(wordCountDict.items()))
+ return wordCountDictAlphaSorted
+
+def blackSort(wordCountDictSorted):
+ sortedCopy = dict(wordCountDictSorted)
+ for word in wordCountDictSorted:
+ if word in blacklist:
+ del sortedCopy[word]
+ return sortedCopy
+
+def dictToCsv(sortedDict, sortType, cantoFile):
+ """
+ write a sorted dict to a CSV file for that canto
+ name it based on sorting type (numerical or alphabetical)
+ abbreviate sortType as `num` or `alpha`
+ """
+ csvCanto = cantoFile.split('.')[0] + '-' + sortType + '.csv'
+ cantoNum = str(int(cantoFile.split('-')[0]))
+ canticle = cantoFile.split('-')[1]
+
+ with open(csvCanto, 'w') as csvCanto:
+ cantoInformation = canticle + ',' + cantoNum + '\n'
+ csvCanto.write(cantoInformation)
+ for word, count in sortedDict.items():
+ csvInput = word + ',' + str(count) + '\n'
+ csvCanto.write(csvInput)
+ csvCanto.close()
+
+def dictToHtml(sortedDict, sortType, cantoFile):
+ htmlCanto = cantoFile.split('.')[0] + '-' + sortType + '.html'
+ with open(htmlCanto, 'w') as htmlCanto:
+ cantoNum = str(int(cantoFile.split('-')[0]))
+ canticle = cantoFile.split('-')[1].title()
+ if sortType == 'num':
+ sortInformation = 'Words sorted by most to least occurrences.'
+ elif sortType == 'num-black':
+ sortInformation = 'Words sorted by most to least occurrences, with blacklist applied.'
+ elif sortType == 'alpha':
+ sortInformation = 'Words sorted alphabetically.'
+ elif sortType == 'alpha-black':
+ sortInformation = 'Words sorted alphabetically, with blacklist applied.'
+ cantoHeaders = '' + canticle + ' ' + cantoNum + '
\n' + sortInformation + '
\n'
+ boilerplate = '\n' + cantoHeaders + '\n\n| Word | Count | \n
\n'
+ boilerclose = '
\n'
+ htmlCanto.write(boilerplate)
+ for word, count in sortedDict.items():
+ htmlInput = '\n| ' + word + ' | ' + str(count) + ' | \n
\n'
+ htmlCanto.write(htmlInput)
+ htmlCanto.write(boilerclose)
+ htmlCanto.close()
+
+def main():
+ for canto in os.listdir(cantoDir):
+ if re.search('\.xml$', canto):
+ cantoFile = canto
+ wordCountDict = createWordCountDict(cantoFile)
+
+ # create sorted dicts
+ wordCountDictNumSorted = numSort(wordCountDict)
+ wordCountDictNumSortedBlack = blackSort(wordCountDictNumSorted)
+ wordCountDictAlphaSorted = alphaSort(wordCountDict)
+ wordCountDictAlphaSortedBlack = blackSort(wordCountDictAlphaSorted)
+
+ # create output files
+ dictToCsv(wordCountDictNumSorted, 'num', cantoFile) # `num-full`: sort numerically with no blacklist
+ dictToCsv(wordCountDictNumSortedBlack, 'num-black', cantoFile) # `num`: sort numerically with blacklist
+ dictToCsv(wordCountDictAlphaSorted, 'alpha', cantoFile)
+ dictToCsv(wordCountDictAlphaSortedBlack, 'alpha-black', cantoFile)
+
+ dictToHtml(wordCountDictNumSorted, 'num', cantoFile)
+ dictToHtml(wordCountDictNumSortedBlack, 'num-black', cantoFile)
+ dictToHtml(wordCountDictAlphaSorted, 'alpha', cantoFile)
+ dictToHtml(wordCountDictAlphaSortedBlack, 'alpha-black', cantoFile)
+
+
+main()