initial release

Browse files

Files changed (10) hide show

README.md +29 -0
config.json +0 -0
maker.py +62 -0
merges.txt +0 -0
pytorch_model.bin +3 -0
special_tokens_map.json +51 -0
tokenizer.json +0 -0
tokenizer_config.json +58 -0
ud.py +81 -0
vocab.json +0 -0

README.md ADDED Viewed

	@@ -0,0 +1,29 @@

+---
+language:
+- "uk"
+tags:
+- "ukrainian"
+- "token-classification"
+- "pos"
+- "dependency-parsing"
+base_model: benjamin/roberta-large-wechsel-ukrainian
+datasets:
+- "universal_dependencies"
+license: "mit"
+pipeline_tag: "token-classification"
+---
+# roberta-large-wechsel-ukrainian-ud-goeswith
+## Model Description
+This is a RoBERTa model for POS-tagging and dependency-parsing (using `goeswith` for subwords), derived from [roberta-large-wechsel-ukrainian](https://huggingface.co/benjamin/roberta-large-wechsel-ukrainian).
+## How to Use
+```py
+from transformers import pipeline
+nlp=pipeline("universal-dependencies","KoichiYasuoka/roberta-large-wechsel-ukrainian-ud-goeswith",trust_remote_code=True,aggregation_strategy="simple")
+print(nlp("Біжать алеї звуків, саджених у гами."))
+```

config.json ADDED Viewed

The diff for this file is too large to render. See raw diff

maker.py ADDED Viewed

	@@ -0,0 +1,62 @@

+#! /usr/bin/python3
+src="benjamin/roberta-large-wechsel-ukrainian"
+tgt="KoichiYasuoka/roberta-large-wechsel-ukrainian-ud-goeswith"
+url="https://github.com/UniversalDependencies/UD_Ukrainian-"
+import os
+for e in ["IU","ParlaMint"]:
+  u=url+e
+  d=os.path.basename(u)
+  os.system("test -d "+d+" || git clone --depth=1 "+u)
+os.system("for F in train dev test ; do cat UD_Ukrainian-*/*-$F.conllu > $F.conllu ; done")
+class UDgoeswithDataset(object):
+  def __init__(self,conllu,tokenizer):
+    self.ids,self.tags,label=[],[],set()
+    with open(conllu,"r",encoding="utf-8") as r:
+      cls,sep,msk=tokenizer.cls_token_id,tokenizer.sep_token_id,tokenizer.mask_token_id
+      dep,c,m="-|_|dep",[],False
+      for s in r:
+        t=s.split("\t")
+        if len(t)==10:
+          if t[0].isdecimal():
+            i=int(t[0])
+            if m:
+              t[1]=" "+t[1]
+            c.append(t)
+            m=t[9].find("SpaceAfter=No")<0
+        elif c!=[]:
+          v=tokenizer([t[1] for t in c],add_special_tokens=False)["input_ids"]
+          for i in range(len(v)-1,-1,-1):
+            for j in range(1,len(v[i])):
+              c.insert(i+1,[c[i][0],"_","_","X","_","_",c[i][0],"goeswith","_","_"])
+          y=["0"]+[t[0] for t in c]
+          h=[i if t[6]=="0" else y.index(t[6]) for i,t in enumerate(c,1)]
+          p,v=[t[3]+"|"+t[5]+"|"+t[7] for t in c],sum(v,[])
+          if len(v)<tokenizer.model_max_length-3:
+            self.ids.append([cls]+v+[sep])
+            self.tags.append([dep]+p+[dep])
+            label=set(sum([self.tags[-1],list(label)],[]))
+            for i,k in enumerate(v):
+              self.ids.append([cls]+v[0:i]+[msk]+v[i+1:]+[sep,k])
+              self.tags.append([dep]+[t if h[j]==i+1 else dep for j,t in enumerate(p)]+[dep,dep])
+          c,m=[],False
+    self.label2id={l:i for i,l in enumerate(sorted(label))}
+  def __call__(*args):
+    label=set(sum([list(t.label2id) for t in args],[]))
+    lid={l:i for i,l in enumerate(sorted(label))}
+    for t in args:
+      t.label2id=lid
+    return lid
+  __len__=lambda self:len(self.ids)
+  __getitem__=lambda self,i:{"input_ids":self.ids[i],"labels":[self.label2id[t] for t in self.tags[i]]}
+from transformers import AutoTokenizer,AutoConfig,AutoModelForTokenClassification,DataCollatorForTokenClassification,TrainingArguments,Trainer
+tkz=AutoTokenizer.from_pretrained(src)
+trainDS=UDgoeswithDataset("train.conllu",tkz)
+devDS=UDgoeswithDataset("dev.conllu",tkz)
+testDS=UDgoeswithDataset("test.conllu",tkz)
+lid=trainDS(devDS,testDS)
+cfg=AutoConfig.from_pretrained(src,num_labels=len(lid),label2id=lid,id2label={i:l for l,i in lid.items()})
+arg=TrainingArguments(num_train_epochs=3,per_device_train_batch_size=8,output_dir=tgt,overwrite_output_dir=True,save_total_limit=2,eval_strategy="epoch",learning_rate=5e-05,warmup_ratio=0.1,save_safetensors=False)
+trn=Trainer(args=arg,data_collator=DataCollatorForTokenClassification(tkz),model=AutoModelForTokenClassification.from_pretrained(src,config=cfg),train_dataset=trainDS,eval_dataset=devDS)
+trn.train()
+trn.save_model(tgt)
+tkz.save_pretrained(tgt)

merges.txt ADDED Viewed

The diff for this file is too large to render. See raw diff

pytorch_model.bin ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:4e39cc1d902591b19678e5b60142883e49d6d74c4bc229eb7d144bb725cd81e1
+size 1440696742

special_tokens_map.json ADDED Viewed

	@@ -0,0 +1,51 @@

+{
+  "bos_token": {
+    "content": "<s>",
+    "lstrip": false,
+    "normalized": false,
+    "rstrip": false,
+    "single_word": false
+  },
+  "cls_token": {
+    "content": "<s>",
+    "lstrip": false,
+    "normalized": false,
+    "rstrip": false,
+    "single_word": false
+  },
+  "eos_token": {
+    "content": "</s>",
+    "lstrip": false,
+    "normalized": false,
+    "rstrip": false,
+    "single_word": false
+  },
+  "mask_token": {
+    "content": "<mask>",
+    "lstrip": true,
+    "normalized": false,
+    "rstrip": false,
+    "single_word": false
+  },
+  "pad_token": {
+    "content": "<pad>",
+    "lstrip": false,
+    "normalized": false,
+    "rstrip": false,
+    "single_word": false
+  },
+  "sep_token": {
+    "content": "</s>",
+    "lstrip": false,
+    "normalized": false,
+    "rstrip": false,
+    "single_word": false
+  },
+  "unk_token": {
+    "content": "<unk>",
+    "lstrip": false,
+    "normalized": false,
+    "rstrip": false,
+    "single_word": false
+  }
+}

tokenizer.json ADDED Viewed

The diff for this file is too large to render. See raw diff

tokenizer_config.json ADDED Viewed

	@@ -0,0 +1,58 @@

+{
+  "add_prefix_space": false,
+  "added_tokens_decoder": {
+    "0": {
+      "content": "<s>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "1": {
+      "content": "<pad>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "2": {
+      "content": "</s>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "3": {
+      "content": "<unk>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "4": {
+      "content": "<mask>",
+      "lstrip": true,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    }
+  },
+  "bos_token": "<s>",
+  "clean_up_tokenization_spaces": false,
+  "cls_token": "<s>",
+  "eos_token": "</s>",
+  "errors": "replace",
+  "extra_special_tokens": {},
+  "mask_token": "<mask>",
+  "model_max_length": 512,
+  "pad_token": "<pad>",
+  "sep_token": "</s>",
+  "tokenizer_class": "RobertaTokenizer",
+  "trim_offsets": true,
+  "unk_token": "<unk>"
+}

ud.py ADDED Viewed

	@@ -0,0 +1,81 @@

+import numpy
+from transformers import TokenClassificationPipeline
+class UniversalDependenciesPipeline(TokenClassificationPipeline):
+  def _forward(self,model_inputs):
+    import torch
+    v=model_inputs["input_ids"][0].tolist()
+    with torch.no_grad():
+      e=self.model(input_ids=torch.tensor([v[0:i]+[self.tokenizer.mask_token_id]+v[i+1:]+[j] for i,j in enumerate(v[1:-1],1)],device=self.device))
+    return {"logits":e.logits[:,1:-2,:],**model_inputs}
+  def check_model_type(self,supported_models):
+    pass
+  def postprocess(self,model_outputs,**kwargs):
+    if "logits" not in model_outputs:
+      return "".join(self.postprocess(x,**kwargs) for x in model_outputs)
+    e=model_outputs["logits"].numpy()
+    r=[1 if i==0 else -1 if j.endswith("|root") else 0 for i,j in sorted(self.model.config.id2label.items())]
+    e+=numpy.where(numpy.add.outer(numpy.identity(e.shape[0]),r)==0,0,-numpy.inf)
+    g=self.model.config.label2id["X|_|goeswith"]
+    r=numpy.tri(e.shape[0])
+    for i in range(e.shape[0]):
+      for j in range(i+2,e.shape[1]):
+        r[i,j]=r[i,j-1] if numpy.argmax(e[i,j-1])==g else 1
+    e[:,:,g]+=numpy.where(r==0,0,-numpy.inf)
+    m,p=numpy.max(e,axis=2),numpy.argmax(e,axis=2)
+    h=self.chu_liu_edmonds(m)
+    z=[i for i,j in enumerate(h) if i==j]
+    if len(z)>1:
+      k,h=z[numpy.argmax(m[z,z])],numpy.min(m)-numpy.max(m)
+      m[:,z]+=[[0 if j in z and (i!=j or i==k) else h for i in z] for j in range(m.shape[0])]
+      h=self.chu_liu_edmonds(m)
+    v=[(s,e) for s,e in model_outputs["offset_mapping"][0].tolist() if s<e]
+    q=[self.model.config.id2label[p[j,i]].split("|") for i,j in enumerate(h)]
+    if "aggregation_strategy" in kwargs and kwargs["aggregation_strategy"]!="none":
+      for i,j in reversed(list(enumerate(q[1:],1))):
+        if j[-1]=="goeswith" and set([t[-1] for t in q[h[i]+1:i+1]])=={"goeswith"}:
+          h=[b if i>b else b-1 for a,b in enumerate(h) if i!=a]
+          v[i-1]=(v[i-1][0],v.pop(i)[1])
+          q.pop(i)
+        elif v[i-1][1]>v[i][0]:
+          h=[b if i>b else b-1 for a,b in enumerate(h) if i!=a]
+          v[i-1]=(v[i-1][0],v.pop(i)[1])
+          q.pop(i)
+    t=model_outputs["sentence"].replace("\n"," ")
+    for i,(s,e) in reversed(list(enumerate(v))):
+      w=t[s:e]
+      if w.startswith(" "):
+        j=len(w)-len(w.lstrip())
+        w=w.lstrip()
+        v[i]=(v[i][0]+j,v[i][1])
+      if w.endswith(" "):
+        j=len(w)-len(w.rstrip())
+        w=w.rstrip()
+        v[i]=(v[i][0],v[i][1]-j)
+      if w.strip()=="":
+        h=[b if i>b else b-1 for a,b in enumerate(h) if i!=a]
+        v.pop(i)
+        q.pop(i)
+    u="# text = "+t+"\n"
+    for i,(s,e) in enumerate(v):
+      u+="\t".join([str(i+1),t[s:e],"_",q[i][0],"_","|".join(q[i][1:-1]),str(0 if h[i]==i else h[i]+1),q[i][-1],"_","_" if i+1<len(v) and e<v[i+1][0] else "SpaceAfter=No"])+"\n"
+    return u+"\n"
+  def chu_liu_edmonds(self,matrix):
+    h=numpy.argmax(matrix,axis=0)
+    x=[-1 if i==j else j for i,j in enumerate(h)]
+    for b in [lambda x,i,j:-1 if i not in x else x[i],lambda x,i,j:-1 if j<0 else x[j]]:
+      y=[]
+      while x!=y:
+        y=list(x)
+        for i,j in enumerate(x):
+          x[i]=b(x,i,j)
+      if max(x)<0:
+        return h
+    y,x=[i for i,j in enumerate(x) if j==max(x)],[i for i,j in enumerate(x) if j<max(x)]
+    z=matrix-numpy.max(matrix,axis=0)
+    m=numpy.block([[z[x,:][:,x],numpy.max(z[x,:][:,y],axis=1).reshape(len(x),1)],[numpy.max(z[y,:][:,x],axis=0),numpy.max(z[y,y])]])
+    k=[j if i==len(x) else x[j] if j<len(x) else y[numpy.argmax(z[y,x[i]])] for i,j in enumerate(self.chu_liu_edmonds(m))]
+    h=[j if i in y else k[x.index(i)] for i,j in enumerate(h)]
+    i=y[numpy.argmax(z[x[k[-1]],y] if k[-1]<len(x) else z[y,y])]
+    h[i]=x[k[-1]] if k[-1]<len(x) else i
+    return h

vocab.json ADDED Viewed

The diff for this file is too large to render. See raw diff