Commit fe024f3a authored by Kirill Milintsevich's avatar Kirill Milintsevich
Browse files

Removed unused code

parent a10f2905
Loading
Loading
Loading
Loading

TranscriptsToCONLLU.ipynb

deleted100644 → 0
+0 −205
Changes for TranscriptsToCONLLU.ipynb: 0 added lines, 205 removed lines.
Original line number Diff line number Diff line
%% Cell type:code id: tags:

``` python
import stanza
from pathlib import Path
from collections import namedtuple, Counter
import csv
import conllu
from conllu.models import TokenList, Token
import logging
import re
```

%% Cell type:code id: tags:

``` python
logger_format = f'%(asctime)s %(levelname)s: %(message)s'
logging.basicConfig(level=logging.DEBUG, format=logger_format)
```

%% Cell type:code id: tags:

``` python
transcript_dir = Path('data/transcripts')
save_dir = Path('data/transcripts/conllu')

ellie_line = re.compile('(.+)\s\((.+)\)')
abbreviation = re.compile('[a-z](?:_[a-z])+')

def transcript_name(idx):
    return f'{idx}_TRANSCRIPT.csv'

sent = {0: 'negative', 1: 'neutral', 2: 'positive'}
```

%% Cell type:code id: tags:

``` python
def normalize_abbreviations(line):
    abbreviation = re.compile('[a-z](?:_[a-z])+')
    abbs_text = abbreviation.findall(line)
    abbs_new = [a.replace('_', '').upper() for a in abbs_text]
    for a, b in zip(abbs_text, abbs_new):
        line = line.replace(a, b)
    return line
```

%% Cell type:code id: tags:

``` python
TranscriptLine = namedtuple('TranscriptLine', ['actor', 'text'])
```

%% Cell type:code id: tags:

``` python
def read_transcript(path):
    transcript = []
    reader = csv.reader(open(path, encoding='utf-8'), delimiter='\t')
    next(reader)
    for line in reader:
        if len(line) == 4:
            ellie_match = ellie_line.match(line[3])
            if line[2] == 'Ellie' and ellie_match:
                transcript.append(TranscriptLine(line[2], (ellie_match[1], normalize_abbreviations(ellie_match[2]))))
            else:
                transcript.append(TranscriptLine(line[2], normalize_abbreviations(line[3])))
    return transcript
```

%% Cell type:code id: tags:

``` python
transcripts = {int(path.name.split('_')[0]): read_transcript(path) for path in transcript_dir.iterdir() if path.suffix == '.csv'}
```

%% Cell type:code id: tags:

``` python
ellie_lines = Counter()
abbreviations = Counter()
for idx, transcript in transcripts.items():
    for line in transcript:
        if isinstance(line.text, tuple):
            ellie_lines[line.text] += 1
            abbs = abbreviation.findall(line.text[1])
            if abbs:
                abbreviations.update(abbs)
        elif isinstance(line.text, str):
            abbs = abbreviation.findall(line.text)
            if abbs:
                abbreviations.update(abbs)
```

%% Cell type:code id: tags:

``` python
nlp = stanza.Pipeline('en', tokenize_no_ssplit=True, processors='tokenize', verbose=True)
```

%% Output

    2022-11-04 17:08:35 INFO: Checking for updates to resources.json in case models have been updated.  Note: this behavior can be turned off with download_method=None or download_method=DownloadMethod.REUSE_RESOURCES
    2022-11-04 17:08:35,869 INFO: Checking for updates to resources.json in case models have been updated.  Note: this behavior can be turned off with download_method=None or download_method=DownloadMethod.REUSE_RESOURCES
    2022-11-04 17:08:35,873 DEBUG: Starting new HTTPS connection (1): raw.githubusercontent.com:443
    2022-11-04 17:08:35,939 DEBUG: https://raw.githubusercontent.com:443 "GET /stanfordnlp/stanza-resources/main/resources_1.4.1.json HTTP/1.1" 200 28918


    2022-11-04 17:08:35 INFO: Loading these models for language: en (English):
    ========================
    | Processor | Package  |
    ------------------------
    | tokenize  | combined |
    ========================
    
    2022-11-04 17:08:35,955 INFO: Loading these models for language: en (English):
    ========================
    | Processor | Package  |
    ------------------------
    | tokenize  | combined |
    ========================
    
    2022-11-04 17:08:35 INFO: Use device: cpu
    2022-11-04 17:08:35,955 INFO: Use device: cpu
    2022-11-04 17:08:35 INFO: Loading: tokenize
    2022-11-04 17:08:35,956 INFO: Loading: tokenize
    2022-11-04 17:08:35 INFO: Done loading processors!
    2022-11-04 17:08:35,959 INFO: Done loading processors!

%% Cell type:code id: tags:

``` python
def normalize_sentence_dict(sentence_dict):
    DEFAULT_FIELDS = ('id', 'form', 'lemma', 'upos', 'xpos', 'feats', 'head', 'deprel', 'deps', 'misc')
    token_dict = {field: '_' for field in DEFAULT_FIELDS}
    new_sentence = []
    for token in sentence_dict:
        for key, value in token.items():
            if key == 'text':
                key = 'form'
            token_dict[key] = value
        new_sentence.append(token_dict)
        token_dict = {field: '_' for field in DEFAULT_FIELDS}
    return new_sentence
```

%% Cell type:code id: tags:

``` python
if not save_dir.exists():
    save_dir.mkdir()

docs = []
actors_list = []
ellie_codes_list = []
ids = []
logging.info(f"[Reading transcripts...]")
for idx, transcript in transcripts.items():
    doc = '\n\n'.join([t.text if not isinstance(t.text, tuple) else t.text[1] for t in transcript])
    actor = [t.actor for t in transcript]
    ellie_code = [t.text[0] if isinstance(t.text, tuple) else None for t in transcript]

    docs.append(doc)
    actors_list.append(actor)
    ellie_codes_list.append(ellie_code)
    ids.append(idx)
logging.info("[Done reading!]")

logging.info("[Processing the transcripts with Stanza...]")
in_docs = [stanza.Document([], text=d) for d in docs]
stanza_docs = nlp(in_docs)
logging.info("[Done processing!]")

logging.info("[Writting CoNLL-U files...]")
for doc, actors, ellie_codes, idx in zip(stanza_docs, actors_list, ellie_codes_list, ids):
    c = []
    for actor, sentence, ellie_code in zip(actors, doc.sentences, ellie_codes):
        sentence_dict = sentence.to_dict()
        meta = {
            'text': sentence.text,
            'actor': actor,
        }
        if ellie_code:
            meta['ellie_code'] = ellie_code
        c.append(TokenList(normalize_sentence_dict(sentence_dict), metadata=meta))

    save_name = Path(f'{idx}_TRANSCRIPT.conllu')
    with open(save_dir / save_name, 'w', encoding='utf-8') as f:
        f.write(''.join([tokens.serialize() for tokens in c]))
logging.info("[Done!]")
```

%% Output

    2022-11-04 17:19:48,858 INFO: [Reading transcripts...]
    2022-11-04 17:19:48,870 INFO: [Done reading!]
    2022-11-04 17:19:48,871 INFO: [Processing the transcripts with Stanza...]
    2022-11-04 17:20:20,360 INFO: [Done processing!]
    2022-11-04 17:20:20,361 INFO: [Writting CoNLL-U files...]
    2022-11-04 17:20:22,703 INFO: [Done!]

%% Cell type:code id: tags:

``` python
```

config_bert.ini

deleted100644 → 0
+0 −12
Changes for config_bert.ini: 0 added lines, 12 removed lines.
Original line number Diff line number Diff line
[Defaults]
bert_model = sentence-transformers/all-distilroberta-v1
transcript_dir = data/transcripts/conllu
train_labels = data/train_split_Depression_AVEC2017.csv
dev_labels = data/dev_split_Depression_AVEC2017.csv
pretrained_path = external_resources/pretrained.pt
line2type_path = data/ellie_lines_labeled.tsv
seed = 0
save_dir = saved_models
encoder_hidden_dim = 300
encoder_num_layers = 1
batch_size = 8
 No newline at end of file