Commit 1ab52399 authored by Morgane Pica's avatar Morgane Pica
Browse files

Addition of construction scripts before documentation.

parent c07c3813
Loading
Loading
Loading
Loading
+119 −0
Changes for corpus-construction/add-structure-to-transkribus-tei/add_xmlid_to_divs.ipynb: 119 added lines, 0 removed lines.
Original line number Diff line number Diff line
%% Cell type:code id: tags:

``` python
def add_ids(xml_in, xml_out, fileID):

    """

    Function constructing identifiers for TEI-XML divs,
    according to the ConDÉ project schema, all values separated by -,
    all body div numbers formatted with three digits and
    all front/back div numbers formatted with two digits.

    For //text/front and //text/back divs, the construction is as follows:
    - witness id,
    - type of edition (base/txm/simplified),
    - current version (alpha/beta)
    - frontMatter or backMatter
    - number of current front/div or back/div
    - subtype of current front/div if any,
    - number of current front/div/div if subject div is inside a div itself
        (max 2 levels of div in front and back).

    For //text/body divs, construction is as follows:
    - witness id,
    - type of edition (base/txm/simplified),
    - current version (alpha/beta)
    - current part div number,
    - current chapter div number
        (if subject div is a chapter or section)
    - current section div number
        (if subject div is a section).

    All body divs need to be typed (part/chapter/section)for this script to function.


    """

    import xml.etree.ElementTree as ET

    section_counter = 0
    chapter_counter = 0
    part_counter = 0
    front_counter = 0

    ET.register_namespace("", "http://www.tei-c.org/ns/1.0")
    ET.register_namespace('xml','http://www.w3.org/XML/1998/namespace')

    tree = ET.parse(xml_in)
    root = tree.getroot()

    textElement = root.find('.//{http://www.tei-c.org/ns/1.0}text')
    textElement.set('{http://www.w3.org/XML/1998/namespace}id', fileID)

    """for item in root.findall(".//{http://www.tei-c.org/ns/1.0}front/*"):
    #for item in root.findall(".//front/*"):

        front_counter += 1

        if item.tag == "{http://www.tei-c.org/ns/1.0}titlePage":
        #if item.tag == "titlePage":

            frontID = fileID + "-frontMatter-" + str("{:02}".format(front_counter)) + "-titlepage"
            item.set("{http://www.w3.org/XML/1998/namespace}id", frontID)

        elif item.get("type"):

            frontID = fileID + "-frontMatter-" + str("{:02}".format(front_counter)) + "-" + item.get("type")
            item.set("{http://www.w3.org/XML/1998/namespace}id", frontID)

        else:

            frontID = fileID + "-frontMatter-" + str("{:02}".format(front_counter))
            item.set("{http://www.w3.org/XML/1998/namespace}id", frontID)"""

    #for part in root.findall(".//{http://www.tei-c.org/ns/1.0}div[@type='part']"):
    for part in root.findall(".//{http://www.tei-c.org/ns/1.0}body/{http://www.tei-c.org/ns/1.0}div[@type='part']"):

        chapter_counter = 0
        part_counter += 1
        part_identifier = fileID + "-" + str("{:03}".format(part_counter))

        if part.get("{http://www.w3.org/XML/1998/namespace}id"):
            del part.attrib['{http://www.w3.org/XML/1998/namespace}id']
        part.set("{http://www.w3.org/XML/1998/namespace}id", part_identifier)

        #for chapter in part.findall(".//{http://www.tei-c.org/ns/1.0}div[@type='chapter']"):
        for chapter in part.findall(".//{http://www.tei-c.org/ns/1.0}div[@type='chapter']"):

            section_counter = 0
            chapter_counter += 1
            chapter_identifier = fileID + "-" + str("{:03}".format(part_counter)) + "-" + str("{:03}".format(chapter_counter))

            if chapter.get("{http://www.w3.org/XML/1998/namespace}id"):
                del chapter.attrib["{http://www.w3.org/XML/1998/namespace}id"]
            chapter.set("{http://www.w3.org/XML/1998/namespace}id", chapter_identifier)

            #for section in chapter.findall(".//{http://www.tei-c.org/ns/1.0}div[@type='section']"):
            for section in chapter.findall(".//{http://www.tei-c.org/ns/1.0}div[@type='section']"):

                section_counter += 1
                section_identifier = fileID + "-" + str("{:03}".format(part_counter)) + "-" + str("{:03}".format(chapter_counter)) + "-" + str("{:03}".format(section_counter))

                if section.get("{http://www.w3.org/XML/1998/namespace}id"):
                    del section.attrib["{http://www.w3.org/XML/1998/namespace}id"]
                section.set("{http://www.w3.org/XML/1998/namespace}id", section_identifier)

    # On écrit le TEI obtenu dans le fichier spécifié en second paramètre.
    tree.write(xml_out, encoding="unicode")
```

%% Cell type:code id: tags:

``` python
add_ids(
    "/home/erminea/Documents/CONDE/nov-21_renum/terrien_base.xml",
    "/home/erminea/Documents/CONDE/nov-21_divID/terrien_base.xml",
    "terrien-base-beta"
)
```
+273 −0
Changes for corpus-construction/add-structure-to-transkribus-tei/extract-ambiguous-tokens-from-xml.ipynb: 273 added lines, 0 removed lines.
Original line number Diff line number Diff line
%% Cell type:code id: tags:

``` python
import xml.etree.ElementTree as ET
import csv
import datetime
```

%% Cell type:code id: tags:

``` python
def get_text(token):

    texte = ""
    choice = ["corr", "expan", "reg"]
    w_token = ET.fromstring(token)

    if w_token.text:
        texte += w_token.text

    """for item in w_token.findall("./*"):
        # if item.tag == '{http://tei-c.org/ns/1.0}height' or item.tag == '{http://tei-c.org/ns/1.0}supplied':
        if item.tag == 'c' or item.tag == 'supplied':
            texte += str(item.text)
                        # S'il y a du texte après la balise fermante et avant
                        # le prochain enfant ou la balise fermante du <w>,
                        # on l'ajoute.
            if item.tail:
                texte += str(item.tail)

                    # elif item.tag == '{http://tei-c.org/ns/1.0}lb':
        elif item.tag == 'lb':
            if item.tail:
                texte += str(item.tail)

                    # Si l'enfant est un <choice>, on récupère le texte de son
                    # second enfant et on vérifie s'il y a du texte après le <choice>.
                    # elif item.tag == '{http://tei-c.org/ns/1.0}choice':
        elif item.tag == 'choice':
            for subitem in item:
                if subitem.tag in choice:
                    texte += str(subitem.text)
            if item.tail:
                texte += str(item.tail)

        elif item.tag == 'add':
                        # On refait tous les tests.
            if item.find('.') == None :
                texte = str(item.text)

            else:

                if item.text:
                    texte += str(item.text)

                for subitem in item:
                    if subitem.tag == 'lb':
                        if subitem.tail:
                            texte += str(subitem.tail)
                    elif subitem.tag == 'choice':
                        texte += str(subitem[1].text)
                        if subitem.tail:
                            texte += str(subitem.tail)
                            """

    for item in w_token:

        # Si l'enfant est un <height>, on récupère son texte.
        if item.tag == 'height' or item.tag == 'supplied':
            texte += str(item.text)
            # S'il y a du texte après la balise fermante et avant
            # le prochain enfant ou la balise fermante du <w>,
            # on l'ajoute.
            if item.tail:
                texte += str(item.tail)

        elif item.tag == 'lb':
            if item.tail:
                texte += str(item.tail)

                    # Si l'enfant est un <choice>, on récupère le texte de son
                    # second enfant et on vérifie s'il y a du texte après le <choice>.
        elif item.tag == 'choice':
            texte += str(item[1].text)
            if item.tail:
                texte += str(item.tail)

        elif item.tag == 'c':
            texte += item.text
            if item.tail:
                texte += str(item.tail)

        elif item.tag == 'add':
            # On refait tous les tests.
            if item.find('.') == None :
                texte = str(item.text)

            else:

                if item.text:
                    texte += str(item.text)

                for subitem in item:
                    if subitem.tag == 'lb':
                        if subitem.tail:
                            texte += str(subitem.tail)
                    elif subitem.tag == 'choice':
                        texte += str(subitem[1].text)
                        if subitem.tail:
                            texte += str(subitem.tail)

    return texte
```

%% Cell type:code id: tags:

``` python
def extraction(xml_entree, csv_simple, csv_concordancier, txt_stats):

    dico_tokens={}

    # colonnes des CSV:
    simple_cols = ["ID", "TOKEN", "LEMMES", "POS"]
    concord_cols = ["ID", "POS", "GAUCHE", "TOKEN", "DROIT"]

    # compteurs
    nb_total_tokens = 0
    pos_ambigus = 0
    pos_uniques = 0
    pos_inc = 0
    lemmes_ambigus = 0
    lemmes_uniques = 0
    lemmes_inc = 0

    # Pour que Python comprenne les éléments dont on parlera,
    # il faut lui donner la déclaration TEI, mais comme c'est
    # la seule qu'on utilisera, pas besoin de lui donner un préfixe.
    # ET.register_namespace('', "http://tei-c.org/ns/1.0")

    # On va chercher le fichier XML-TEI et on le lit.
    tree = ET.parse(xml_entree)
    root = tree.getroot()

    for word in root.findall('.//w'):
        dico_tokens[int(word.get('n'))] = get_text(ET.tostring(word))

    # On ouvre le CSV de sortie en mode "écriture", on y écrit le nom des colonnes.
    with open(csv_simple, 'w') as csv_file:
        csv_contenu = csv.DictWriter(csv_file, fieldnames = simple_cols, delimiter=";")
        csv_contenu.writeheader()

        # On boucle sur les éléments <w> du XML, dans l'ordre du fichier.
        # for word in root.findall('.//{http://tei-c.org/ns/1.0}w'):
        for word in root.findall('.//w'):

            nb_total_tokens += 1

            # On récupère les @n, @lemma et @pos dans les variables
            # "numero", "lemmes" et "pos"
            # et on crée la chaîne "texte", pour l'instant vide.
            numero = str(word.get('n'))
            lemmes = str(word.get('lemma'))
            pos = str(word.get('pos'))
            texte = get_text(ET.tostring(word))

            if '|' in lemmes:
                lemmes_ambigus += 1

                if '|' in pos:
                    pos_ambigus += 1
                elif pos=="Inconnu":
                    pos_inc += 1

                csv_contenu.writerow(
                    {
                        "ID":numero,
                        "TOKEN":texte,
                        "LEMMES":lemmes,
                        "POS":pos
                    }
                )

            elif lemmes=="INC":
                lemmes_inc += 1

                if '|' in pos:
                    pos_ambigus += 1
                elif pos=="Inconnu":
                    pos_inc += 1

                csv_contenu.writerow(
                    {
                        "ID":numero,
                        "TOKEN":texte,
                        "LEMMES":lemmes,
                        "POS":pos
                    }
                )

            else:
                lemmes_uniques += 1


    """with open(csv_concordancier, 'w') as csv_file:
        csv_contenu = csv.DictWriter(csv_file, fieldnames = concord_cols, delimiter=";")
        csv_contenu.writeheader()

        # On boucle sur les éléments <w> du XML, dans l'ordre du fichier.
        # for word in root.findall('.//{http://tei-c.org/ns/1.0}w'):
        for word in root.findall('.//w'):

            # On récupère les @n, @lemma et @pos dans les variables
            # "numero", "lemmes" et "pos"
            # et on crée la chaîne "texte", pour l'instant vide.
            numero = str(word.get('n'))
            lemmes = str(word.get('lemma'))
            pos = str(word.get('pos'))
            texte = get_text(ET.tostring(word))
            ["ID", "POSG", "GAUCHE", "TOKEN", "DROIT", "POSD"]
            if '|' in lemmes or '|' in pos or lemmes=="INC" or pos=="Inconnu":

                gauche = [dico_tokens[int(numero)-3], dico_tokens[int(numero)-2], dico_tokens[int(numero)-1]]
                droit = [dico_tokens[int(numero)+1], dico_tokens[int(numero)+2], dico_tokens[int(numero)+3]]

                csv_contenu.writerow(
                    {
                        "ID":numero,
                        "POS":pos,
                        "GAUCHE": " ".join(gauche),
                        "TOKEN": texte,
                        "DROITE":" ".join(droit)

                    }
                )"""


    pourcentage_lemmes = lemmes_uniques * 100 / nb_total_tokens
    pourcentage_pos = pos_uniques * 100 / nb_total_tokens
    pourcentage_lemmes_inc = lemmes_inc * 100 / nb_total_tokens
    pourcentage_extraits = pos_ambigus * 100 / nb_total_tokens

    with open(txt_stats, "w") as file:
        file.write(str(datetime.datetime.now()))
        file.write(round(pourcentage_lemmes,2), "% de lemmes uniques.")
        file.write(round(pourcentage_pos,2), "% de POS uniques.")
        file.write(lemmes_inc, "lemmes inconnus, soit", round(pourcentage_lemmes_inc,2), "%.")
```

%% Cell type:code id: tags:

``` python
extraction('/home/erminea/Documents/CONDE/Rouille-TS/Rouille_19-lemmatise_div-ided.xml',
          '/home/erminea/Documents/CONDE/Rouille-TS/rouille_ambig_tableau.csv',
          '/home/erminea/Documents/CONDE/Rouille-TS/rouille_ambig_concord.csv',
          '/home/erminea/Documents/CONDE/Rouille-TS/stats.txt')
```

%% Output

    ---------------------------------------------------------------------------
    TypeError                                 Traceback (most recent call last)
    <ipython-input-27-2eec5b523fc6> in <module>
    ----> 1 extraction('/home/erminea/Documents/CONDE/Rouille-TS/Rouille_19-lemmatise_div-ided.xml',
          2           '/home/erminea/Documents/CONDE/Rouille-TS/rouille_ambig_tableau.csv',
          3           '/home/erminea/Documents/CONDE/Rouille-TS/rouille_ambig_concord.csv',
          4           '/home/erminea/Documents/CONDE/Rouille-TS/stats.txt')
    <ipython-input-26-92e1d0861357> in extraction(xml_entree, csv_simple, csv_concordancier, txt_stats)
        125     with open(txt_stats, "w") as file:
        126         file.write(str(datetime.datetime.now()))
    --> 127         file.write(round(pourcentage_lemmes,2), "% de lemmes uniques.")
        128         file.write(round(pourcentage_pos,2), "% de POS uniques.")
        129         file.write(lemmes_inc, "lemmes inconnus, soit", round(pourcentage_lemmes_inc,2), "%.")
    TypeError: write() takes exactly one argument (2 given)
+35 −0
Changes for corpus-construction/add-structure-to-transkribus-tei/move_notes_pesnelle.ipynb: 35 added lines, 0 removed lines.
Original line number Diff line number Diff line
%% Cell type:code id: tags:

``` python
import xml.etree.ElementTree as ET
import re


entree_tree = ET.parse('data/2020-07-10_Pesnelle_T1.xml')
entree_root = entree_tree.getroot()

# On boucle sur les div[@type='section'].
for section in entree_root.findall('.//div[@type="section"]'):

    # On y cherche les appels de note, on s'arrête sur chaque.
    for appel_note in section.findall('.//note[@type="appel"]'):
        # On garde en mémoire le numéro de l'appel de note.
        n_appel = appel_note.get('n')

        # On boucle sur les notes de bas-de-page de la section courante.
        for note_pleine in section.findall('./note[p]'):
            # Pour chaque note on regarde si elle porte le même numéro que l'appel courant.
            if note_pleine.get('n') == n_appel:
                # Si c'est le cas, on place la note dans l'appel et on la supprime.
                appel_note.append(note_pleine)
                section.remove(note_pleine)

a_ecrire = ET.tostring(entree_root, encoding="unicode", method="xml")

ecriture = open('data/2020-07-10_Pesnelle_T1_notes.xml', "w")
ecriture.write(a_ecrire)
```

%% Output

    5893039
+106 −0

File added.

Preview size limit exceeded, changes collapsed.

+154 −0

File added.

Preview size limit exceeded, changes collapsed.

Loading