Commit 0627873b authored by Laurette Chardon's avatar Laurette Chardon
Browse files

Modifications qui seront validées :

	nouveau fichier : Clusterisation2D_RevueSyntSem21.py
parent 90cece7c
Loading
Loading
Loading
Loading
+170 −0
Changes for Clusterisation2D_RevueSyntSem21.py: 170 added lines, 0 removed lines.
Original line number Diff line number Diff line
# Programme Clusterisation2D_RevueSynSem21.py juin 2020
# Auteur : Laurette Chardon
# But : créer les graphes de clusterisation avec igraph en prenant comme exemple le mot SEC
# -> programme en accès public sous https://git.unicaen.fr/crisco-des-public/etude-sec




import igraph 
import numpy as np
from LectureSyno_paires_CreationMatAdj import lecture, lecture_listemots

listemots=[]
G=igraph.Graph()

print('version igraph :',igraph.__version__)

# ----------------------------------------------------------------------
def lecture(mot,fichier):
    
    listemots=[]
    

    listemots.append(mot)
    for ligne in fichier:
        i1=ligne.index("\t")
        mot1=ligne[0:i1]
        ligne=ligne[i1+1:]
        i2=ligne.index("\t")
        
        mot2=ligne[0:i2]
        if (mot == mot1): # si le 1er mot de la ligne est celui qu'on cherche
            listemots.append(mot2)
    
    return(listemots)
# -----------------------------------------------------------------------
def lecture_listemots(MatAdj,lm,fichier):
    
    for ligne in fichier:
        i1=ligne.index("\t")
        mot1=ligne[0:i1]
        ligne=ligne[i1+1:]
        i2=ligne.index("\t")
        mot2=ligne[0:i2]
        i1=0
        for m1 in lm: # on compare le mot lu à la vedette et à tous ses synonymes 
            if (m1 == mot1):
                MatAdj[i1][i1]=1 # chaque synonyme de la vedette est synonyme de lui-même (diagonale)
                i2=0
                for m2 in lm:
                    if (m2 == mot2):
                        print(i1,'-',m1,'-',i2,'-',m2,'- à garder')
                        MatAdj[i1][i2]=1
                    i2=i2+1
            i1=i1+1
    return(MatAdj)
#--------------------- PROGRAMME PRINCIPAL ------------

fichier = open ("syno_paires_sec.txt","r",encoding="utf-8")
mot='sec'

listemots=lecture(mot,fichier)
fichier.close()
print(listemots)
print('0:',listemots[0])
print('1:',listemots[1])
dim=len(listemots)
print(dim)

G.add_vertices(dim) # on crée les sommets avec la liste de mots trouvée = vedette + ses syno dans listemots

MatAdj=np.zeros((dim,dim),dtype=int)
print (MatAdj)
fichier = open ("syno_paires_sec.txt","r",encoding="utf-8")
MatAdj=lecture_listemots(MatAdj,listemots,fichier) # dans lecture_listemots on remplit la structure Graph avec les edges
fichier.close()
print (MatAdj)

# Création du graphe à partir de la matrice d'adjacence
np.fill_diagonal(MatAdj,0) # on remplit la diagonale de 0 -> éviter les arêtes sur les sommets
G=igraph.Graph.Adjacency(MatAdj.tolist(),"MAX")
# voir les différents modes possibles : https://igraph.org/python/doc/igraph.GraphBase-class.html#Adjacency

G.vs["name"]=listemots # On ajoute à chaque sommet un intitulé

#labels=listemots
labels=[]
for i in listemots:
  ind=listemots.index(i)
  labels.append(str(i+" "+str(G.vs[ind].degree())))



SizeMarkers=G.vs.degree()

# G.vs['name'].index(mot) -> index de la vedette dans la liste G.vs.['name']
# Si la boule vedette est trop grosse, décommenter la ligne ci-dessous
SizeMarkers[G.vs['name'].index(mot)]=max(G.vs[1:].degree()) # le mot vedette est celui qui a beaucoup de edge -> disproportionné pour l'affichage
# on lui affecte comme  nb de edge le max du nb de edges de ses syno


SizeLine=[]
for e in G.get_edgelist():
    SizeLine.append(G.edge_connectivity(e[0],e[1]))


# ---------------
# création des clusters avec les différents algorithmes

comms1 = G.community_multilevel()

comms2 = G.community_infomap()
comms3 = G.community_fastgreedy()
comms4 = G.community_leading_eigenvector(clusters=6)

comms5 = G.community_walktrap()


comms6 = G.community_optimal_modularity() # très long ... 
comms7=G.community_spinglass()
comms8=G.community_edge_betweenness()
comms9=G.community_label_propagation()

# ---------------
# initialisation des paramètres d'affichage


visual_style3 = {}
visual_style3["vertex_label"] = labels

visual_style3["margin"] = 40
#visual_style3["vertex_color"]="#33FFF3"
#visual_style3["vertex_color"]='red'
visual_style3["vertex_label_size"]=12
visual_style3["vertex_size"]= [i*2 for i in SizeMarkers]
visual_style3["vertex_label_dist"]=1
#visual_style3["vertex_label_color"]="red"
visual_style3["bbox"] = (1000, 1000)
visual_style3["edge_color"]="gray" # couleurs prédéfinies : brown, gray ... cf http://www.proftnj.com/RGB3.htm
visual_style3["edge_width"] = [i/8 for i in SizeLine]
visual_style3["title"] = "Mot : "+ mot.upper()
visual_style3["legend"] = "Mot : "+ mot.upper()
#visual_style3["mark_groups"]= True



# ----------------------------------
# Création des images résultat

igraph.plot(comms1,  mot+"-multilevelNoMarkGroups.svg", **visual_style3)

igraph.plot(comms2,  mot+"-infomapNoMarkGroups.svg", mark_groups = False, **visual_style3)

igraph.plot(comms3,  mot+"-fastgreedy.svg", mark_groups = True, **visual_style3)

igraph.plot(comms4,  mot+"-leading-eigenvector.svg", mark_groups = True, **visual_style3)

igraph.plot(comms5,  mot+"-walktrap.svg", **visual_style3)
igraph.plot(comms6,  mot+"-optimal_modularity.svg", mark_groups = True, **visual_style3)
igraph.plot(comms7,  mot+"-spinglassNoMarkGroups.svg", mark_groups = False, **visual_style3)
igraph.plot(comms8,  mot+"-edge_betweenness.svg", mark_groups = True, **visual_style3)
igraph.plot(comms9,  mot+"-label_propagation.svg", mark_groups = True, **visual_style3)
# ---------------