Skip to content

Commit e8bad3f

Browse files
committed
Release Ancient Chinese tokenization model KYOTO_EVAHAN_TOK_LZH
1 parent c0d5d92 commit e8bad3f

2 files changed

Lines changed: 45 additions & 1 deletion

File tree

‎docs/references.bib‎

Lines changed: 39 additions & 1 deletion
Original file line numberDiff line numberDiff line change
@@ -1,13 +1,51 @@
11
%% This BibTeX bibliography file was created using BibDesk.
22
%% https://bibdesk.sourceforge.io/
33
4-
%% Created for hankcs at 2024-12-28 15:02:27 -0800
4+
%% Created for hankcs at 2025-01-11 17:43:12 -0800
55
66
77
%% Saved with string encoding Unicode (UTF-8)
88
99
1010
11+
@inproceedings{li-etal-2022-first,
12+
abstract = {This paper presents the results of the First Ancient Chinese Word Segmentation and POS Tagging Bakeoff (EvaHan), which was held at the Second Workshop on Language Technologies for Historical and Ancient Languages (LT4HALA) 2022, in the context of the 13th Edition of the Language Resources and Evaluation Conference (LREC 2022). We give the motivation for having an international shared contest, as well as the data and tracks. The contest is consisted of two modalities, closed and open. In the closed modality, the participants are only allowed to use the training data, obtained the highest F1 score of 96.03{\%} and 92.05{\%} in word segmentation and POS tagging. In the open modality, the participants can use whatever resource they have, with the highest F1 score of 96.34{\%} and 92.56{\%} in word segmentation and POS tagging. The scores on the blind test dataset decrease around 3 points, which shows that the out-of-vocabulary words still are the bottleneck for lexical analyzers.},
13+
address = {Marseille, France},
14+
author = {Li, Bin and Yuan, Yiguo and Lu, Jingya and Feng, Minxuan and Xu, Chao and Qu, Weiguang and Wang, Dongbo},
15+
booktitle = {Proceedings of the Second Workshop on Language Technologies for Historical and Ancient Languages},
16+
date-added = {2025-01-11 17:43:11 -0800},
17+
date-modified = {2025-01-11 17:43:11 -0800},
18+
editor = {Sprugnoli, Rachele and Passarotti, Marco},
19+
month = jun,
20+
pages = {135--140},
21+
publisher = {European Language Resources Association},
22+
title = {The First International {A}ncient {C}hinese Word Segmentation and {POS} Tagging Bakeoff: Overview of the {E}va{H}an 2022 Evaluation Campaign},
23+
url = {https://aclanthology.org/2022.lt4hala-1.19/},
24+
year = {2022},
25+
bdsk-url-1 = {https://aclanthology.org/2022.lt4hala-1.19/}}
26+
27+
@inproceedings{YASK:2019,
28+
abstract = {Classical Chinese is an isolating language without notational inflection, and its texts are continuous strings of Chinese characters without spaces or punctuations between words or sentences. In order to apply Universal Dependencies for classical Chinese, we need several ``not-universal'' treatments and enhancements. In this paper such treatments and enhancements are revealed.},
29+
author = {YASUOKA, Koichi},
30+
date-added = {2025-01-11 17:39:18 -0800},
31+
date-modified = {2025-01-11 17:39:18 -0800},
32+
journal = {DADH2019: 10th International Conference of Digital Archives and Digital Humanities},
33+
month = {12},
34+
publisher = {Digital Archives and Digital Humanities},
35+
title = {Universal Dependencies Treebank of the Four Books in Classical Chinese},
36+
url = {http://hdl.handle.net/2433/245217},
37+
year = {2019},
38+
bdsk-url-1 = {http://hdl.handle.net/2433/245217}}
39+
40+
@inproceedings{wang2022uncertainty,
41+
author = {Wang, Pengyu and Ren, Zhichen},
42+
booktitle = {Proceedings of the Second Workshop on Language Technologies for Historical and Ancient Languages},
43+
date-added = {2025-01-11 17:36:05 -0800},
44+
date-modified = {2025-01-11 17:36:05 -0800},
45+
pages = {164--168},
46+
title = {The Uncertainty-based Retrieval Framework for Ancient Chinese CWS and POS},
47+
year = {2022}}
48+
1149
@article{warner2024smarter,
1250
author = {Warner, Benjamin and Chaffin, Antoine and Clavi{\'e}, Benjamin and Weller, Orion and Hallstr{\"o}m, Oskar and Taghadouini, Said and Gallagher, Alexis and Biswas, Raja and Ladhak, Faisal and Aarsen, Tom and others},
1351
date-added = {2024-12-28 15:02:26 -0800},

‎hanlp/pretrained/tok.py‎

Lines changed: 6 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -34,6 +34,12 @@
3434
'Electra (:cite:`clark2020electra`) base model trained on MSR CWS dataset. Its performance is ``P: 98.71% R: 98.64% F1: 98.68%`` ' \
3535
'which is much higher than that of MTL model '
3636

37+
KYOTO_EVAHAN_TOK_LZH = 'http://download.hanlp.com/tok/extra/kyoto_evahan_tok_bert-ancient-chinese_tau_0.5_20250111_234146.zip'
38+
'Ancient Chinese tokenizer with bert-ancient-chinese (:cite:`wang2022uncertainty`) encoder trained on Classical Chinese ' \
39+
'Universal Dependencies Treebank (:cite:`YASK:2019`) and EvaHan corpus (:cite:`li-etal-2022-first`). ' \
40+
'Performance: {UD P: 98.85% R: 99.00% F1: 98.92%} on UD Kyoto, ' \
41+
'and {TestA P: 95.62% R: 96.56% F1: 96.09%} {TestB P: 94.93% R: 93.05% F1: 93.98%} on EvaHan.'
42+
3743
UD_TOK_MMINILMV2L6 = HANLP_URL + 'tok/ud_tok_mMiniLMv2L6_no_space_mul_20220619_091824.zip'
3844
'''
3945
mMiniLMv2 (:cite:`wang-etal-2021-minilmv2`) L6xH384 based tokenizer trained on UD 2.10.

0 commit comments

Comments
 (0)