Commit 038d5698 authored by MPica's avatar MPica
Browse files

Documentation finished on pos-per-lemma-table.ipynb.

parent 80b7e902
Loading
Loading
Loading
Loading
+44740 −0

File added.

Preview size limit exceeded, changes collapsed.

+152 −47
Changes for corpus-construction/which-lemmas-does-the-corpus-have/pos-per-lemmas-table.ipynb: 152 added lines, 47 removed lines.
Original line number Diff line number Diff line
{
 "cells": [
  {
   "cell_type": "markdown",
   "metadata": {},
   "source": [
    "# Which lemmas does the corpus have?\n",
    "\n",
    "This script was written for correction purposes: it is meant to display the lemmas present inside the corpus and their associated forms and POS, so as to assess the coherence of the linguistic encoding.\n",
    "\n",
    "Note: This script may be used on another TEI-XML corpus, if lemma/POS/regularization is structured the same way.\n",
    "\n",
    "### IMPORTS and declarations"
   ]
  },
  {
   "cell_type": "code",
   "execution_count": null,
   "metadata": {},
   "outputs": [],
   "source": [
    "import xml.etree.ElementTree as ET\n",
    "import csv\n",
    "\n",
    "ET.register_namespace('', 'http://www.tei-c.org/ns/1.0')"
   ]
  },
  {
   "cell_type": "markdown",
   "metadata": {},
   "source": [
    "\n",
    "### FUNCTION: extract the modernized text from one token\n",
    "\n",
    "Note: As it is widely used throughout my scripts, this function may be made into a separate Python file in the future, to be imported in other scripts."
   ]
  },
  {
   "cell_type": "code",
   "execution_count": 1,
@@ -7,55 +42,61 @@
   "outputs": [],
   "source": [
    "def extraire_forme(word):\n",
    "    # et on crée la chaîne \"texte\", pour l'instant vide.\n",
    "    texte = \"\"\n",
    "        \n",
    "    # Si <w> n'a pas d'enfant, on récupère le texte tel-quel.\n",
    "    if word.find('.') == None :\n",
    "        texte = str(word.text)\n",
    "    \"\"\"\n",
    "    Function taking a <tei:w> element and\n",
    "    returning its compiled textual content.\n",
    "    \n",
    "    # Sinon, on compile.\n",
    "    else:\n",
    "    :param word: ET.Element('{http://www.tei-c.org/ns/1.0}w')\n",
    "    \n",
    "    \"\"\"\n",
    "    \n",
    "    # Preparing the return string as an empty string.\n",
    "    texte = \"\"\n",
    "        \n",
    "        # S'il y a du texte avant le premier enfant, on l'ajoute.\n",
    "    # If there is text directly inside <w> element and\n",
    "    # before the first child, add it.\n",
    "    if word.text:\n",
    "        texte += str(word.text)\n",
    "                \n",
    "        # On boucle sur les enfants du <w> actuel.\n",
    "    # Loop on all current <w> children.for item in word:\n",
    "    for item in word:\n",
    "        \n",
    "            # Si l'enfant est un <height>, on récupère son texte.\n",
    "        # If current child is <tei:height> or <tei:supplied>\n",
    "        if item.tag == 'height' or item.tag == 'supplied':\n",
    "            # Add text.\n",
    "            texte += str(item.text)\n",
    "                # S'il y a du texte après la balise fermante et avant\n",
    "                # le prochain enfant ou la balise fermante du <w>,\n",
    "                # on l'ajoute.\n",
    "            # If any, add the text following current child.\n",
    "            if item.tail:\n",
    "                texte += str(item.tail)\n",
    "                 \n",
    "        # If current child is <tei:lb>, add the following text.\n",
    "        elif item.tag == 'lb':\n",
    "            if item.tail:\n",
    "                texte += str(item.tail)\n",
    "                        \n",
    "            # Si l'enfant est un <choice>, on récupère le texte de son\n",
    "            # second enfant et on vérifie s'il y a du texte après le <choice>.\n",
    "        # If current child is <tei:choice>, add the second child of <choice>\n",
    "        # (<tei:reg> or <tei:expan>), then add the text following current child if any.\n",
    "        elif item.tag == 'choice':\n",
    "            texte += str(item[1].text)\n",
    "            if item.tail:\n",
    "                texte += str(item.tail)\n",
    "            \n",
    "        # If current child is <tei:c>, add its text, then the following text if any.\n",
    "        elif item.tag == 'c':\n",
    "            texte += item.text\n",
    "            if item.tail:\n",
    "                texte += str(item.tail)\n",
    "                \n",
    "        # If current child is <tei:hi>, add its text, then the following text if any.\n",
    "        elif item.tag == 'hi':\n",
    "            texte += item.text\n",
    "            if item.tail:\n",
    "                texte += item.tail\n",
    "            \n",
    "        # If current child is <tei:add>, loop on its children and do the same checks.\n",
    "        elif item.tag == 'add':\n",
    "                # On refait tous les tests.\n",
    "            \n",
    "            if item.find('.') == None :\n",
    "                texte = str(item.text)\n",
    "                            \n",
@@ -75,6 +116,13 @@
    "    return texte"
   ]
  },
  {
   "cell_type": "markdown",
   "metadata": {},
   "source": [
    "### FUNCTION: make a dictionary of all found lemmas"
   ]
  },
  {
   "cell_type": "code",
   "execution_count": 2,
@@ -83,52 +131,110 @@
   "source": [
    "def export_lemmas_to_dict(chemin_entree, witname, dict_name):\n",
    "    \n",
    "    print(witname)\n",
    "    \"\"\"\n",
    "    Function taking a TEI-XML file with <tei:w> elements and recording their linguistic information\n",
    "    (//self::tei:w/@lemma, //self::tei:w/@pos and //self::tei:w/text)\n",
    "    in order to complete a general dictionary containing one entry by possible lemma in the corpus.\n",
    "    Entries in this dictionary are structured thus:\n",
    "    dict_name[lemma] = {'pos':[pos], 'témoins':[witname], 'forms':[texte]}\n",
    "    \n",
    "    import xml.etree.ElementTree as ET\n",
    "    import csv\n",
    "    :param chemin_entree: The path to the TEI-XML file which needs to be analysed, as a string.\n",
    "    :param witname: A string which will be used to identify the current file in the dictionary.\n",
    "    :param dict_name: The dictionary to which new information will be added.\n",
    "    \n",
    "    \"\"\"\n",
    "    \n",
    "    # Show which source is currently treated.\n",
    "    print(witname)\n",
    "    \n",
    "    ET.register_namespace('', 'http://www.tei-c.org/ns/1.0')\n",
    "    # Open and parse the current TEI-XML file.\n",
    "    with open(chemin_entree) as xmlfile:\n",
    "        tree = ET.parse(chemin_entree)\n",
    "        root = tree.getroot()\n",
    "    \n",
    "    # On boucle sur les éléments <w> du XML, dans l'ordre du fichier.\n",
    "        # Loop on all <tei:w> elements in the document, in reading order.\n",
    "        for word in root.findall('.//{http://www.tei-c.org/ns/1.0}w'):\n",
    "            \n",
    "            \n",
    "            # FIRST GET THE TEXT OF THE CURRENT TOKEN.\n",
    "            \n",
    "            # If <tei:w> element has no child, just take its text.\n",
    "            if word.find('.') == None :\n",
    "                texte = str(word.text)\n",
    "            \n",
    "            # Otherwise, compile the text from children elements.\n",
    "            else:\n",
    "                texte = extraire_forme(word)\n",
    "            \n",
    "            \n",
    "            # THEN WE GET OTHER USEFUL INFORMATION.\n",
    "            \n",
    "            # If the token is correctly enriched, get its lemma\n",
    "            # (replacing any < with text which can't be confused\n",
    "            # with an actual \"crochet\" token).\n",
    "            if word.get('lemma'):\n",
    "                lemma = word.get('lemma').replace('<', 'CCRROOCCHHEETT')\n",
    "            \n",
    "            # If there is no lemma, use text instead.\n",
    "            elif word.text:\n",
    "                lemma = word.text\n",
    "            \n",
    "            # If there is still no text, use an easily identifiable string instead.\n",
    "            else:\n",
    "                lemma = 'No lemma'\n",
    "            \n",
    "            # If there is an @pos, use it. If not, cf previous comment.\n",
    "            if word.get('pos'):\n",
    "                pos = word.get('pos')\n",
    "            else:\n",
    "                pos = \"No pos\"\n",
    "            \n",
    "            \n",
    "            # THEN WE CAN START RECORDING INFORMATION\n",
    "            \n",
    "            # If the lemma is already present inside the dictionary,\n",
    "            # complete the entry with new information if needed.\n",
    "            if lemma in dict_name.keys():\n",
    "                \n",
    "                if pos in dict_name[lemma]['pos']:\n",
    "                    \n",
    "                    # If the current lemma has already been recorded\n",
    "                    # with the current POS but not the current file,\n",
    "                    # just add the file name to the list.\n",
    "                    if witname not in dict_name[lemma]['témoins']:\n",
    "                        dict_name[lemma]['témoins'].append(witname)\n",
    "                \n",
    "                else:\n",
    "                    # If the current lemma not yet been recorded\n",
    "                    # with the current POS, add the POS to the list.\n",
    "                    dict_name[lemma]['pos'].append(pos)\n",
    "                    \n",
    "                    # In addition, if the current file name was not\n",
    "                    # yet recorded for the current lemma, add it.\n",
    "                    if witname not in general_dict[lemma]['témoins']:\n",
    "                        dict_name[lemma]['témoins'].append(witname)\n",
    "                \n",
    "                # Regardless of whether POS was already recorded,\n",
    "                # if the current form was not, add it.\n",
    "                if texte not in dict_name[lemma]['forms']:\n",
    "                    dict_name[lemma]['forms'].append(texte)\n",
    "            \n",
    "            \n",
    "            # If the lemma is not already present inside the dictionary,\n",
    "            # initiate the entry with current information.\n",
    "            else:\n",
    "                dict_name[lemma] = {'pos':[pos], 'témoins':[witname], 'forms':[texte]}"
   ]
  },
  {
   "cell_type": "markdown",
   "metadata": {},
   "source": [
    "### Apply previous functions and export results to CSV\n",
    "\n",
    "Note: To use this script in another corpus, this is where most of names and paths needs to be changed."
   ]
  },
  {
   "cell_type": "code",
   "execution_count": 3,
@@ -44492,10 +44598,10 @@
    }
   ],
   "source": [
    "import csv\n",
    "\n",
    "# Start the general lemma dictionary.\n",
    "general_dict = {}\n",
    "\n",
    "# Make a list of columns for output CSV file.\n",
    "colonnes = [\n",
    "    'Lemme',\n",
    "    'Catégories possibles',\n",
@@ -44511,36 +44617,33 @@
    "    'TAC',\n",
    "    'Terrien'\n",
    "]\n",
    "\"\"\"\n",
    "export_lemmas_to_dict(\"/home/erminea/Documents/CONDE/editions-reserve/base-version/instructions_base.xml\", 'Instructions', general_dict)\n",
    "export_lemmas_to_dict(\"/home/erminea/Documents/CONDE/editions-reserve/base-version/rouille_base.xml\", 'Rouillé', general_dict)\n",
    "export_lemmas_to_dict(\"/home/erminea/Documents/CONDE/editions-reserve/base-version/tac_base.xml\", 'TAC', general_dict)\n",
    "export_lemmas_to_dict(\"/home/erminea/Documents/CONDE/editions-reserve/base-version/terrien_base.xml\", 'Terrien', general_dict)\n",
    "export_lemmas_to_dict(\"/home/erminea/Documents/CONDE/editions-reserve/base-version/instructions_base.xml\", 'Instructions', general_dict)\n",
    "#export_lemmas_to_dict(\"/home/erminea/Documents/CONDE/editions-reserve/base-version/basnage_base.xml\", 'Basnage', general_dict)\n",
    "#export_lemmas_to_dict(\"/home/erminea/Documents/CONDE/editions-reserve/base-version/berault_base.xml\", 'Bérault', general_dict)\n",
    "#export_lemmas_to_dict(\"/home/erminea/Documents/CONDE/editions-reserve/base-version/gc_base_renum.xml\", 'GC', general_dict)\n",
    "#export_lemmas_to_dict(\"/home/erminea/Documents/CONDE/editions-reserve/base-version/merville_base.xml\", 'Merville', general_dict)\n",
    "#export_lemmas_to_dict(\"/home/erminea/Documents/CONDE/editions-reserve/base-version/pesnelle_base.xml\", 'Pesnelle', general_dict)\n",
    "#export_lemmas_to_dict(\"/home/erminea/Documents/CONDE/editions-reserve/base-version/ruines_base.xml\", 'Ruines', general_dict)\n",
    "\"\"\"\n",
    "export_lemmas_to_dict('/home/erminea/Documents/CONDE/zinaida-lemmes/editions-zinaida-15092021/base-version/gc_base_renum.xml', \"GC\", general_dict)\n",
    "export_lemmas_to_dict('/home/erminea/Documents/CONDE/zinaida-lemmes/editions-zinaida-15092021/base-version/tac_base.xml', \"TAC\", general_dict)\n",
    "export_lemmas_to_dict('/home/erminea/Documents/CONDE/zinaida-lemmes/editions-zinaida-15092021/base-version/instructions_base.xml', \"Instructions\", general_dict)\n",
    "export_lemmas_to_dict('/home/erminea/Documents/CONDE/zinaida-lemmes/editions-zinaida-15092021/base-version/rouille_base.xml', \"Rouillé\", general_dict)\n",
    "export_lemmas_to_dict('/home/erminea/Documents/CONDE/zinaida-lemmes/editions-zinaida-15092021/base-version/terrien_base.xml', \"Terrien\", general_dict)\n",
    "export_lemmas_to_dict('/home/erminea/Documents/CONDE/zinaida-lemmes/prune-depuis-cloud/merville_base.xml', 'Merville', general_dict)\n",
    "export_lemmas_to_dict('/home/erminea/Documents/CONDE/zinaida-lemmes/editions-zinaida-15092021/base-version/basnage_base.xml', \"Basnage\", general_dict)\n",
    "export_lemmas_to_dict('/home/erminea/Documents/CONDE/zinaida-lemmes/editions-zinaida-15092021/base-version/berault_base.xml', \"Bérault\", general_dict)\n",
    "export_lemmas_to_dict('/home/erminea/Documents/CONDE/zinaida-lemmes/editions-zinaida-15092021/base-version/pesnelle_base.xml', \"Pesnelle\", general_dict)\n",
    "\n",
    "# Complete the general lemma dictionary with the information from one source at a time.\n",
    "# This may be optimized later by using variables and a for loop.\n",
    "export_lemmas_to_dict('/local/path/to/base-version/gc_base_renum.xml', \"GC\", general_dict)\n",
    "export_lemmas_to_dict('/local/path/to/base-version/tac_base.xml', \"TAC\", general_dict)\n",
    "export_lemmas_to_dict('/local/path/to/base-version/instructions_base.xml', \"Instructions\", general_dict)\n",
    "export_lemmas_to_dict('/local/path/to/base-version/rouille_base.xml', \"Rouillé\", general_dict)\n",
    "export_lemmas_to_dict('/local/path/to/base-version/terrien_base.xml', \"Terrien\", general_dict)\n",
    "export_lemmas_to_dict('/local/path/to/base-version/merville_base.xml', 'Merville', general_dict)\n",
    "export_lemmas_to_dict('/local/path/to/base-version/basnage_base.xml', \"Basnage\", general_dict)\n",
    "export_lemmas_to_dict('/local/path/to/base-version/berault_base.xml', \"Bérault\", general_dict)\n",
    "export_lemmas_to_dict('/local/path/to/base-version/pesnelle_base.xml', \"Pesnelle\", general_dict)\n",
    "\n",
    "\n",
    "# Open and prepare output CSV file.\n",
    "with open('tableau_lemmes_tout.csv', 'w') as csvtobe:\n",
    "    csvwriting = csv.DictWriter(csvtobe, fieldnames=colonnes)\n",
    "    csvwriting.writeheader()\n",
    "    \n",
    "    # Loop on all lemma entries in the general lemma dictionary.\n",
    "    for lemma in sorted(general_dict.keys()):\n",
    "        # Let me check where you are and if everything is alright.\n",
    "        print(lemma, ' -> ', general_dict[lemma])\n",
    "        \n",
    "        # FOR EACH SOURCE, CHECK WHETHER THE CURRENT LEMMA WAS SEEN INSIDE IT.\n",
    "        # Make a str variable recording this fact, to be used directly in CSV output.\n",
    "        \n",
    "        if 'Basnage' in general_dict[lemma]['témoins']:\n",
    "            basnage = 'X'\n",
    "        else:\n",
@@ -44591,6 +44694,8 @@
    "        else:\n",
    "            terrien=''\n",
    "        \n",
    "        # Make a new line in the CSV file containing the information\n",
    "        # on the current lemma.\n",
    "        csvwriting.writerow(\n",
    "            {\n",
    "                'Lemme': lemma,\n",
@@ -44613,7 +44718,7 @@
 ],
 "metadata": {
  "kernelspec": {
   "display_name": "Python 3",
   "display_name": "Python 3 (ipykernel)",
   "language": "python",
   "name": "python3"
  },
@@ -44627,7 +44732,7 @@
   "name": "python",
   "nbconvert_exporter": "python",
   "pygments_lexer": "ipython3",
   "version": "3.8.3"
   "version": "3.9.7"
  }
 },
 "nbformat": 4,