Browse Source

deletions

main
lhess2021 4 years ago
parent
commit
ddb3670c46
  1. 8061
      comedy-xml.d/inferno-complete.xml
  2. 8119
      comedy-xml.d/paradiso-complete.xml
  3. 8113
      comedy-xml.d/purgatorio-complete.xml
  4. 114
      py-scripts.d/word-tags.py
  5. 9
      shell-scripts.d/canticle-base.xml
  6. 55
      shell-scripts.d/canticle-to-xml-automated.sh
  7. 25
      shell-scripts.d/full-canticles.sh
  8. 13
      shell-scripts.d/rename-txt.sh
  9. 36
      shell-scripts.d/tercet-tags.sh
  10. 49
      shell-scripts.d/top-level-tags.sh
  11. 71
      shell-scripts.d/xpath.sh

8061
comedy-xml.d/inferno-complete.xml

File diff suppressed because it is too large

8119
comedy-xml.d/paradiso-complete.xml

File diff suppressed because it is too large

8113
comedy-xml.d/purgatorio-complete.xml

File diff suppressed because it is too large

114
py-scripts.d/word-tags.py

@ -1,114 +0,0 @@
import re
import os
canto_dir = os.path.abspath(os.getcwd())
print('\n\n====================' + canto_dir + '====================')
"""
Define XML file name
Load txt file
Iterate through lines in txt file
Iterate through words in each line
Iterate through each letter from end to beginning
catch single chars like `-`
catch single chars at end of line like `-\n`
if the char is not `` or `-` or alphabetical, go to the next letter in word (backwards)
else separate the ending punct from beginning chars
Iterate through each letter from beginning to remainder of word
if no punctuation, then all remainder is middle chars
catch single chars like `-\n`
if the char is not `` or `` or `-` or alphabetical, go to the next letter in word (forwards)
else separate middle chars from beginning punctuation
Build the tag word from <beginning punct> + <middle chars> + <ending punct>
Write the tag word to the temp XML file
Open the temp XML file
Iterate over lines
Catch double spaces
Catch leading spaces
Add leading tabs and leading line tag
Add closing line tag
"""
def add_tags(canto_txt_file):
# only grab 0xx-canticle from `*.txt` file names
xmlfile = canto_txt_file.split('.')[0] + '.xml'
tmpxmlfile = '.' + canto_txt_file.split('.')[0] + '.tmp.xml'
with open(xmlfile,'w') as cxml:
with open(tmpxmlfile,'w') as cxmltmp:
with open(canto_txt_file,'r') as ctxt:
for spaced_line in ctxt:
# use `~` as a placeholder for spaces later
line = re.sub(' ','~',spaced_line)
words = line.split('~')
for i in range(len(words)):
word = words[i]
# Iterate from ending to beginning of word
L = len(word) - 1
# Catch single chars like `-`
if L == 0:
before_end_word_chars = word
ending_punct_chars = ''
# Catch single chars at end of line
elif L == 1 and '\n' in word:
before_end_word_chars = word.rstrip()
ending_punct_chars = '\n'
else:
print('Got to <- with: ' + word + ' and this L: ' + str(L) + ' in ' + canto_txt_file)
# We need `word[L] != '-'` because there are some words like `-:`
while word[L] != '' and word[L] != '-' and not word[L].isalpha():
L-=1
else:
ending_punct_begin_index = L + 1
ending_punct_chars = word[ending_punct_begin_index:]
before_end_word_chars = word[:ending_punct_begin_index]
# Iterate from start to remainder of word
L = len(word) - 1
if L == 0:
middle_word_chars = word
starting_punct_chars = ''
elif L == 1 and '\n' in word:
middle_word_chars = word.rstrip()
starting_punct_chars = ''
else:
print('Got -> with this word: ' + word + ' and this L: ' + str(L) + ' in ' + canto_txt_file)
while word[-L - 1] != '' and word[-L - 1] != '' and word[L] != '-' and not word[-L - 1].isalpha():
L-=1
else:
starting_punct_end_index = -L - 1
starting_punct_chars = word[:starting_punct_end_index]
middle_word_chars = before_end_word_chars[-L-1:]
# Pad with spaces to ensure we get spaces between words
# and importantly _around_ punctuation. We want:
# `<w>Rispuosemi</w>: «<w>Non</w>` and not `<w>Rispuosemi</w>:«<w>Non</w>`.
new_word = ' ' + starting_punct_chars + '<w>' + middle_word_chars + '</w>' + ending_punct_chars + ' '
Cxmltmp.write(new_word)
Ctxt.close()
Cxmltmp.close()
with open(tmpxmlfile,'r') as Cxmltmp:
# Add space before opening word tag
def add_space(match):
match = match.group()
return re.sub('<w>', ' <w>', match)
j = 1
for line in Cxmltmp:
# Swap double spaces from new_word with single spaces
line_spaced = re.sub(' ',' ',line)
# If there's a single space at the beginning of the line, delete it
line_spaced = line_spaced.lstrip(' ')
# Add three tabs and opening line tags at the beginning
line_tag = '\t\t\t\t\t<l l_num="' + str(j) + '"> '
line_spaced = re.sub('^',line_tag,line_spaced)
# Add closing line tags
line_spaced = re.sub('\n',' </l>\n',line_spaced)
Cxml.write(line_spaced)
j += 1
Cxmltmp.close()
os.remove(tmpxmlfile)
Cxml.close()
for canto in os.listdir(canto_dir):
if canto != 'word-tags.py' and canto != '.word-tags.py.swp':
add_tags(canto)
if canto.endswith('.txt'):
os.remove(os.path.join(canto_dir, canto))

9
shell-scripts.d/canticle-base.xml

@ -1,9 +0,0 @@
<?xml version="1.0" encoding="UTF-8"?>
<TEI xmlns="http://www.tei-c.org/ns/1.0">
<teiHeader>
<fileDesc>
<titleStmt>CANTICLE</titleStmt>
</fileDesc>
</teiHeader>
<text>

55
shell-scripts.d/canticle-to-xml-automated.sh

@ -1,55 +0,0 @@
#!/bin/bash
# A script which converts plaintext cantos from one canticle to XML formatting.
canticles=( 'inferno' 'purgatorio' 'paradiso' )
comedy_xml_dir=../comedy-xml.d
comedy_xml_dir_bak=../comedy-xml.d.bak
comedy_plaintext_dir=../comedy-plaintext.d
shell_scripts_dir="$(pwd)"
if [[ "$(pwd)" =~ shell-scripts.d ]]; then
clear
if [[ ! -d "$comedy_xml_dir" ]]; then
echo 'Making `comedy-xml.d`'
mkdir $comedy_xml_dir
else
echo 'Moving your `../comedy-xml.d/` to `../comedy-xml.d.bak`'
mv $comedy_xml_dir $comedy_xml_dir_bak && mkdir $comedy_xml_dir
fi
for canticle in "${canticles[@]}";
do
echo -e "\nRenaming dirs..."
dir="$canticle-txt.d"
dir_xml="$canticle-xml.d"
if [[ -d $comedy_plaintext_dir ]]; then
echo -e "\tMoving $dir to $dir_xml"
cp -r "$comedy_plaintext_dir/$dir" "$comedy_xml_dir/$dir_xml"
else
echo "Could not find dir: $comedy_plaintext_dir"
fi
done
for canticle in "${canticles[@]}";
do
dir_xml="$comedy_xml_dir/$canticle-xml.d"
echo "Copying word-tags.py" && cp ../py-scripts.d/word-tags.py $dir_xml/word-tags.py &&\
echo "Changing dir to $dir_xml" && cd $dir_xml &&\
echo "Running word-tags.py" && python3 word-tags.py &&\
echo "Removing word-tags.py" && rm word-tags.py &&\
echo "Changing dir to $shell_scripts_dir" && cd $shell_scripts_dir &&\
echo "Copying tercet-tags.sh" && cp ../shell-scripts.d/tercet-tags.sh $dir_xml/tercet-tags.sh &&\
echo "Changing dir to $dir_xml" && cd $dir_xml &&\
echo "Running tercet-tags.sh" && ./tercet-tags.sh &&\
echo "Removing tercet-tags.sh" && rm tercet-tags.sh &&\
echo "Changing dir to $shell_scripts_dir" && cd $shell_scripts_dir &&\
echo "Copying top-level-tags.sh" && cp ../shell-scripts.d/top-level-tags.sh $dir_xml/top-level-tags.sh &&\
echo "Changing dir to $dir_xml" && cd $dir_xml &&\
echo "Running top-level-tags.sh" && ./top-level-tags.sh &&\
echo "Removing top-level-tags.sh" && rm top-level-tags.sh &&\
echo "Adding .gitignore" && echo 'xpath.sh' > .gitignore &&\
echo "Changing dir to $shell_scripts_dir" && cd $shell_scripts_dir
done
else
echo "Wrong directory?"
fi

25
shell-scripts.d/full-canticles.sh

@ -1,25 +0,0 @@
#!/bin/bash
# Make full canticle files from comedy-xml.d
xml_dir=../comedy-xml.d
canticles=( 'inferno' 'purgatorio' 'paradiso' )
for canticle in "${canticles[@]}";
do
canticle_dir="$xml_dir/$canticle-xml.d/"
canticle_complete=$xml_dir"/$canticle-complete.xml"
cp ./canticle-base.xml $canticle_complete
sed -i "s/CANTICLE/The\ Complete\ ${canticle^}/" $canticle_complete
echo -e "\t\t<$canticle>" >> $canticle_complete
for canto in $canticle_dir*;
do
canto_text="$(xmlstarlet sel -t -c "_:TEI/_:text/_:canto" $canto | sed "s/^/\t/g")"
echo -e "$canto_text" >> $canticle_complete
done
echo -e "\t\t</$canticle>" >> $canticle_complete
echo -e "\t</text>" >> $canticle_complete
echo "</TEI>" >> $canticle_complete
sed -i 's/<canto\ xmlns=\"http:\/\/www.tei-c.org\/ns\/1.0\"/\t\t<canto/g' $canticle_complete
done

13
shell-scripts.d/rename-txt.sh

@ -1,13 +0,0 @@
#!/bin/bash
# Rename canto files
for file in *;
do
if [[ $file != r.sh ]]; then
num="$(echo $file | tr -d -c 0-9)"
pad="$(printf "%03d\n" $num)"
echo "$file: $pad"
mv $file "$pad-purgatorio.txt"
fi
done

36
shell-scripts.d/tercet-tags.sh

@ -1,36 +0,0 @@
#!/bin/bash
# Add terzina tags to each canto.xml file in a directory
# Also add canto number attribute.
echo -e "\nAdding terzina tags and numbers."
for file in *;
do
if [[ "$file" != tercet-tags.sh ]] && [[ "$file" != canticle-to-xml.sh ]]; then # exclude this script
if [[ "$file" != sed* ]]; then # exclude sed file made later
echo -e "Modifying $file"
# Every third line (mod 3) add a closing tag, newline, and opening tag.
# TODO
# How many tabs to add?
awk -i inplace '{print}; NR % 3 == 0 {print "\t\t\t\t</tercet>\n\t\t\t\t<tercet t_num=~NUM~>"}' "$file"
# Add an opening tag at the beginning and a closing tag at the end because the above awk doesn't.
sed -i '1s/^/\t\t\t\t<tercet t_num="1">\n/' "$file"
# Also remove the newline at the end of the file.
sed -i '$ d' "$file"
echo -e "\t\t\t\t</tercet>" >> "$file"
# Add the canto number attribute (at most 160/3 ~ 54 terzine)
for i in $(seq 1 54);
do
# Substitute in the actual number with sed. I couldn't figure out
# how to include the $i variable with the surrounding attribute
# quotes in the awk statement, so I just used sed here separately.
# We start at i=2 because the awk command above
# doesn't add a terzina tag to the first terzina.
TERZ_NUM=$((i + 1))
# Replace only first occurence of ~NUM~.
sed -i -e "0,/~NUM~/ s/~NUM~/\"$TERZ_NUM\"/" "$file"
done
fi
fi
done

49
shell-scripts.d/top-level-tags.sh

@ -1,49 +0,0 @@
#!/bin/bash
# This script adds top-level tags to your XML files.
# - XML header
# - TEI header
echo -e "\nAdding top-level tags."
for file in *;
do
if [[ "$file" != 'top-level-tags.sh' ]] && [[ "$file" != 'canticle-to-xml.sh' ]]; then
ZEROS_CANTO_NUM="$(echo "'$file'" | tr -d -c 0-9)" # get canto number
#CANTO_NUM="$(echo $ZEROS_CANTO_NUM | tr -d 0)"
CANTO_NUM=$((10#$ZEROS_CANTO_NUM))
if [[ "$file" == *inferno* ]]; then
CANTICLE="Inferno"
COMM_NUM="$CANTO_NUM" # same as inferno canto number
elif [[ "$file" == *purgatorio* ]]; then
CANTICLE="Purgatorio"
COMM_NUM="$(($CANTO_NUM + 34))" # add 34 if in purgatory
elif [[ "$file" == *paradiso* ]]; then
CANTICLE="Paradiso"
COMM_NUM="$(($CANTO_NUM + 67))" # add 68 if in paradise
else
echo -e "\nWhat file is this? $file"
fi
# Add tags to first line of file
sed -i -e '1s/^/<?xml version="1.0" encoding="UTF-8"?>\
\n<TEI xmlns="http:\/\/www.tei-c.org\/ns\/1.0">\
\t<teiHeader>\
\t\t<fileDesc>\
\t\t\t<titleStmt>Canto ~TITLE_CANTO_NUM~ of ~CANTICLE~<\/titleStmt>\
\t\t<\/fileDesc>\
\t<\/teiHeader>\
\t<text>\
\t\t<canto canto_num=~CANTO_NUM~ comm_num=~COMM_NUM~>\
\t\t\t<tercets>\n/' "$file"
# Replace canto number, comedy number, and canticle
sed -i -e "s/~TITLE_CANTO_NUM~/$CANTO_NUM/" "$file"
sed -i -e "s/~CANTO_NUM~/\"$CANTO_NUM\"/" "$file"
sed -i -e "s/~COMM_NUM~/\"$COMM_NUM\"/" "$file"
sed -i -e "s/~CANTICLE~/$CANTICLE/" "$file"
# Add tags to end of file
echo -e "\t\t\t</tercets>" >> "$file"
echo -e "\t\t</canto>" >> "$file"
echo -e "\t</text>" >> "$file"
echo -e "</TEI>" >> "$file"
fi
done

71
shell-scripts.d/xpath.sh

@ -1,71 +0,0 @@
#!/bin/bash
# An interactive xpath query commandline tool.
# Stores your previous query to reduce typing.
# File for previous query
touch .prev-xpath.tmp
if [[ ! -e ".gitignore" ]]; then
echo 'xpath.sh' > .gitignore
fi
# Colored output
GREEN='\033[0;32m'
NC='\033[0m' # no color
# Ask for canto file
clear
read -p "Choose a canto: [i,r,p][1-34] " CANTO
canticle="$(echo $CANTO | tr -d -c a-z)"
canto_num="$(echo $CANTO | tr -d -c 0-9)"
pad_num="$(printf "%03d\n" $canto_num)"
# Create complete filename
if [[ "$canticle" == 'i' ]]; then
canticle='Inferno'
file="$pad_num-inferno-petrocchi.xml"
echo $file
elif [[ "$canticle" == 'r' ]]; then
canticle='Purgatorio'
file="$pad_num-purgatorio-petrocchi.xml"
elif [[ "$canticle" == 'p' ]]; then
canticle='Paradiso'
file="$pad_num-paradiso-petrocchi.xml"
else
echo "Not a valid canto file: $file"
fi
# Start interactive query
if [[ -e "$file" ]]; then
clear && echo -e "You are searching ${canticle^} ${CANTO_NUM}.\nAdd a trailing slash ${GREEN}\`/\`${NC} to your search before submitting it. The default first two search choices are ${GREEN}\`canto\`${NC} and then ${GREEN}\`terzine\`${NC}. Press ${GREEN}<enter>${NC} for those defaults and type something else to search that.\n"
read -p "Load previous: (Y)/n " PREV
case "$PREV" in
N|n)
search=''
;;
Y|y|*)
search="$(cat .prev-xpath.tmp)"
;;
esac
i='0'
# Start while loop for the query. Break the loop if the query ends with a slash `/`.
while true; do
read -e -i "$search" -p "$(echo -e "\nSearch: \n "${GREEN}"===> "${NC})" nsearch
if [[ "${nsearch: -1}" == / ]]; then # if last char trailing slash then search
path="${nsearch::-1}" # remove trailing slash
echo "$path" > .prev-xpath.tmp # store query for next time
# perform the search (requires `xmlstarlet` and `bat` pager)
xmlstarlet sel -t -v "$path" $file | batcat -l xml
exit
elif [[ "$i" == 0 ]] && [[ "$nsearch" == '' ]]; then # load '/TEI/text' query on first loop
search="_:TEI/_:text/_:"
elif [[ "$nsearch" == '_:TEI/_:text/_:' ]]; then # load '/TEI/text/canto/tercets' on second loop
search="_:TEI/_:text/_:canto/_:tercets/_:"
else
search="${nsearch}" # else keep search the same
fi
i=$[$i+1]
done
else
echo "Canto file not found. What directory are you in?"
fi
Loading…
Cancel
Save