UTF-8 dictionary

This commit is contained in:
Nishi 2026-06-19 18:21:53 +09:00
commit 4f76bb0590
12 changed files with 113 additions and 348 deletions

View file

@ -1,198 +0,0 @@
{ "カ五1", 91 },
{ "カ五2", 101 },
{ "カ五3", 131 },
{ "カ五4", 141 },
{ "カ五5", 111 },
{ "カ五6", 121 },
{ "カ五7", 151 },
{ "カ五8", 161 },
{ "カ五音便", 185 },
{ "カ変仮", 181 },
{ "カ変終体", 180 },
{ "カ変未", 178 },
{ "カ変命", 182 },
{ "カ変用", 179 },
{ "ガ五1", 92 },
{ "ガ五2", 102 },
{ "ガ五3", 132 },
{ "ガ五4", 142 },
{ "ガ五5", 112 },
{ "ガ五6", 122 },
{ "ガ五7", 152 },
{ "ガ五8", 162 },
{ "サ五1", 93 },
{ "サ五2", 103 },
{ "サ五3", 133 },
{ "サ五4", 143 },
{ "サ五5", 113 },
{ "サ五6", 123 },
{ "サ五7", 153 },
{ "サ五8", 163 },
{ "サ変", 80 },
{ "サ変仮", 175 },
{ "サ変終体", 174 },
{ "サ変未1", 171 },
{ "サ変未2", 172 },
{ "サ変未用", 173 },
{ "サ変命1", 176 },
{ "サ変命2", 177 },
{ "ザ変", 81 },
{ "タ五1", 94 },
{ "タ五2", 104 },
{ "タ五3", 134 },
{ "タ五4", 144 },
{ "タ五5", 114 },
{ "タ五6", 124 },
{ "タ五7", 154 },
{ "タ五8", 164 },
{ "ナ五", 95 },
{ "バ五1", 96 },
{ "バ五2", 105 },
{ "バ五3", 135 },
{ "バ五4", 145 },
{ "バ五5", 115 },
{ "バ五6", 125 },
{ "バ五7", 155 },
{ "バ五8", 165 },
{ "マ五1", 97 },
{ "マ五2", 106 },
{ "マ五3", 136 },
{ "マ五4", 146 },
{ "マ五5", 116 },
{ "マ五6", 126 },
{ "マ五7", 156 },
{ "マ五8", 166 },
{ "ラ五1", 98 },
{ "ラ五2", 107 },
{ "ラ五3", 137 },
{ "ラ五4", 147 },
{ "ラ五5", 117 },
{ "ラ五6", 127 },
{ "ラ五7", 157 },
{ "ラ五8", 167 },
{ "ワ五1", 99 },
{ "ワ五2", 108 },
{ "ワ五3", 138 },
{ "ワ五4", 148 },
{ "ワ五5", 118 },
{ "ワ五6", 128 },
{ "ワ五7", 158 },
{ "ワ五8", 168 },
{ "挨拶", 187 },
{ "衣服", -11 },
{ "一括", 200 },
{ "一段1", 90 },
{ "一段2", 100 },
{ "一段3", 130 },
{ "一段4", 140 },
{ "雨", -21 },
{ "音楽", -20 },
{ "海", -31 },
{ "開閉", -4 },
{ "活動", -5 },
{ "感動", 28 },
{ "企業", 23 },
{ "魚", -14 },
{ "形1", 60 },
{ "形10", 69 },
{ "形11", 70 },
{ "形2", 61 },
{ "形3", 62 },
{ "形4", 63 },
{ "形5", 64 },
{ "形6", 65 },
{ "形7", 66 },
{ "形8", 67 },
{ "形9", 68 },
{ "形動1", 71 },
{ "形動2", 72 },
{ "形動3", 73 },
{ "形動4", 74 },
{ "形動5", 75 },
{ "形動6", 76 },
{ "形動7", 77 },
{ "形動8", 78 },
{ "形動9", 79 },
{ "建築", -13 },
{ "県区", 25 },
{ "湖", -33 },
{ "国", -12 },
{ "試合", -10 },
{ "事件", -27 },
{ "時", -25 },
{ "酒", -23 },
{ "州", -29 },
{ "出版", -18 },
{ "所", -2 },
{ "助数", 29 },
{ "助数2", 54 },
{ "乗物", -17 },
{ "植物", -6 },
{ "食物", -16 },
{ "身体", -8 },
{ "人", -1 },
{ "数詞", 30 },
{ "数詞2", 55 },
{ "接続", 27 },
{ "接頭1", 31 },
{ "接頭2", 32 },
{ "接頭3", 33 },
{ "接頭4", 34 },
{ "接頭5", 35 },
{ "接尾1", 36 },
{ "接尾2", 37 },
{ "接尾3", 38 },
{ "接尾4", 39 },
{ "接尾5", 40 },
{ "接尾6", 41 },
{ "接尾7", 42 },
{ "接尾8", 43 },
{ "接尾9", 44 },
{ "川", -30 },
{ "組織", -3 },
{ "贈物", -26 },
{ "代1", 12 },
{ "代2", 13 },
{ "代3", 14 },
{ "代4", 15 },
{ "代5", 16 },
{ "代6", 17 },
{ "単漢", 189 },
{ "地位", -7 },
{ "地名", 24 },
{ "丁寧1", 183 },
{ "丁寧2", 184 },
{ "電車", -34 },
{ "都市", -24 },
{ "島", -32 },
{ "動物", -9 },
{ "特殊形容", 188 },
{ "特殊副", 186 },
{ "病気", -15 },
{ "苗字", 21 },
{ "副1", 45 },
{ "副2", 46 },
{ "副3", 47 },
{ "副4", 48 },
{ "副5", 49 },
{ "副6", 50 },
{ "副7", 51 },
{ "副8", 52 },
{ "副9", 53 },
{ "文字", -19 },
{ "法", -35 },
{ "名1", 1 },
{ "名10", 10 },
{ "名11", 11 },
{ "名2", 2 },
{ "名20", 18 },
{ "名3", 3 },
{ "名4", 4 },
{ "名5", 5 },
{ "名6", 6 },
{ "名7", 7 },
{ "名8", 8 },
{ "名9", 9 },
{ "名前", 22 },
{ "鳴物", -28 },
{ "連体", 26 }

View file

@ -39,19 +39,6 @@
#include "sj3mkdic.h"
int cnvyomi(int code) {
#ifdef UTF8
if(code == 0x30fc) return 1;
if(code == 0xff03) return 2;
if(code == 0xff20) return 3;
if('0' <= code && code <= '9') return code - '0' + 0x10;
if('A' <= code && code <= 'Z') return code - 'A' + 0x1a;
if(0x3041 <= code && code <= 0x3093) return code - 0x3041 + 0x4e;
if(code == 0x30f4) return 0xa1;
if(code == 0x30f5) return 0xa2;
if(code == 0x30f6) return 0xa3;
return 0;
#else
u_short hh;
u_char high;
u_char low;
@ -106,7 +93,6 @@ int cnvyomi(int code) {
}
return 0;
#endif
}
int h2kcode(int code) {
@ -207,14 +193,6 @@ void output_str(FILE* fp, char* p) {
void output_int(FILE* fp, int* p) {
while(*p) {
#ifdef UTF8
char buf[5];
utf8_print(buf, *p);
fputs(buf, fp);
p++;
#else
if(*p < 0x100) {
fputc(*p, fp);
p++;
@ -234,40 +212,10 @@ void output_int(FILE* fp, int* p) {
fputc(*p & 0xff, fp);
p++;
}
#endif
}
}
int yomi2zen(int code) {
#ifdef UTF8
switch(code) {
case 1:
return 0x30fc;
case 2:
return 0xff03;
case 3:
return 0xff20;
}
if(0x10 <= code && code <= 0x19) return code - 0x10 + '0';
if(0x1a <= code && code <= 0x4d) return code - 0x1a + 'A';
if(0x4e <= code && code <= 0xa0) return code - 0x4e + 0x3041;
if(code == 0xa1) return 0x30f4;
if(code == 0xa2) return 0x30f5;
if(code == 0xa3) return 0x30f6;
if(code == 0x30fc) return 1;
if(code == 0xff03) return 2;
if(code == 0xff20) return 3;
if('0' <= code && code <= '9') return code - '0' + 0x10;
if('A' <= code && code <= 'Z') return code - 'A' + 0x1a;
if(0x3041 <= code && code <= 0x3093) return code - 0x3041 + 0x4e;
if(code == 0x30f4) return 0xa1;
if(code == 0x30f5) return 0xa2;
if(code == 0x30f6) return 0xa3;
return code;
#else
static char num[] = {
'0', '1', '2', '3', '4', '5', '6', '7',
'8', '9', 'A', 'B', 'C', 'D', 'E', 'F'};
@ -307,7 +255,6 @@ int yomi2zen(int code) {
return (0xa400 + code - 0x4e + 0xa1);
return ((num[(code >> 4) & 0x0f] << 8) | num[code & 0x0f]);
#endif
}
void output_yomi(FILE* fp, u_char* p) {
@ -317,13 +264,7 @@ void output_yomi(FILE* fp, u_char* p) {
while(*p) {
i = yomi2zen(*p++);
#ifdef UTF8
utf8_print(buf, i);
fputs(buf, fp);
#else
fputc((i >> 8) & 0xff, fp);
fputc(i & 0xff, fp);
#endif
}
}

View file

@ -40,11 +40,7 @@ static struct gram_code {
int code;
} gramtbl[] = {
#ifdef UTF8
#include "GramTable.utf8"
#else
#include "GramTable"
#endif
};

View file

@ -173,8 +173,8 @@ makeyomi(int* yomi) {
j = cnvyomi(*y++);
if(j == 0) {
fprintf(stderr,
#ifdef UTF8
"不正な文字が読みに使われています\n"
#ifdef ENGLISH
"Invalid character used for pronunciation\n"
#else
"\311\324\300\265\244\312\312\270\273\372\244\254\306\311\244\337\244\313\273\310\244\357\244\354\244\306\244\244\244\336\244\271\n"
#endif

View file

@ -61,8 +61,8 @@ set_ofsask(u_char* src, u_char* dst) {
case LEADINGHANKAKU:
#ifndef USEHANKAKUINDICT
fprintf(stderr,
#ifdef UTF8
"\264\301\273\372\260\265\275\314\312\270\273\372\316\363\244\254\304\271\244\271\244\256\244\353\n"
#ifdef ENGLISH
"Attribute code error\n"
#else
"\302\260\300\255\245\263\241\274\245\311\241\246\245\250\245\351\241\274\n"
#endif
@ -76,8 +76,8 @@ set_ofsask(u_char* src, u_char* dst) {
#endif
case OFFSETASSYUKU:
fprintf(stderr,
#ifdef UTF8
"二重にオフセットが参照されている\n"
#ifdef ENGLISH
"Offset is referenced twice\n"
#else
"\306\363\275\305\244\313\245\252\245\325\245\273\245\303\245\310\244\254\273\262\276\310\244\265\244\354\244\306\244\244\244\353\n"
#endif
@ -116,8 +116,8 @@ int make_knjstr(u_char* src, int len, u_char* dst) {
case LEADINGHANKAKU:
#ifndef USEHANKAKUINDICT
fprintf(stderr,
#ifdef UTF8
"\264\301\273\372\260\265\275\314\312\270\273\372\316\363\244\254\304\271\244\271\244\256\244\353\n"
#ifdef ENGLISH
"Attribute code error\n"
#else
"\302\260\300\255\245\263\241\274\245\311\241\246\245\250\245\351\241\274\n"
#endif
@ -168,8 +168,8 @@ make_knjask() {
p = (u_char*)Malloc(len);
if(!p) {
fprintf(stderr,
#ifdef UTF8
"メモリが足りません\n"
#ifdef ENGLISH
"Not enough memory\n"
#else
"\245\341\245\342\245\352\244\254\302\255\244\352\244\336\244\273\244\363\n"
#endif
@ -217,8 +217,8 @@ void makeseg() {
i = make_knjask();
if(i >= 0x100) {
fprintf(stderr,
#ifdef UTF8
"漢字圧縮文字列が長すぎる\n"
#ifdef ENGLISH
"Kanji compression string is too long\n"
#else
"\264\301\273\372\260\265\275\314\312\270\273\372\316\363\244\254\304\271\244\271\244\256\244\353\n"
#endif
@ -301,8 +301,8 @@ void makeseg() {
mindex[idxpos++] = *p;
} else {
fprintf(stderr,
#ifdef UTF8
"インデックスがあふれました\n"
#ifdef ENGLISH
"Index overflow\n"
#else
"\245\244\245\363\245\307\245\303\245\257\245\271\244\254\244\242\244\325\244\354\244\336\244\267\244\277\n"
#endif
@ -315,8 +315,8 @@ void makeseg() {
mindex[idxpos] = 0;
} else {
fprintf(stderr,
#ifdef UTF8
"インデックスがあふれました\n"
#ifdef ENGLISH
"Index overflow\n"
#else
"\245\244\245\363\245\307\245\303\245\257\245\271\244\254\244\242\244\325\244\354\244\336\244\267\244\277\n"
#endif
@ -331,8 +331,8 @@ void makeseg() {
(long)(HEADERLENGTH + COMMENTLENGTH + MAININDEXLENGTH + idxnum * MAINSEGMENTLENGTH),
0) < 0) {
fprintf(stderr,
#ifdef UTF8
"出力ファイルでシークエラー\n"
#ifdef ENGLISH
"Output file seek error\n"
#else
"\275\320\316\317\245\325\245\241\245\244\245\353\244\307\245\267\241\274\245\257\245\250\245\351\241\274\n"
#endif
@ -341,8 +341,8 @@ void makeseg() {
}
if(Fwrite((char*)buf, sizeof(buf), 1, outfp) != 1) {
fprintf(stderr,
#ifdef UTF8
"出力ファイルでライトエラー\n"
#ifdef ENGLISH
"Output file write error\n"
#else
"\275\320\316\317\245\325\245\241\245\244\245\353\244\307\245\351\245\244\245\310\245\250\245\351\241\274\n"
#endif
@ -352,8 +352,8 @@ void makeseg() {
Fflush(outfp);
printf(
#ifdef UTF8
"セグメント番号:%d:"
#ifdef ENGLISH
"Segment number:%d:"
#else
"\245\273\245\260\245\341\245\363\245\310\310\326\271\346:%d:"
#endif
@ -362,10 +362,10 @@ void makeseg() {
output_yomi(stdout, douon_ptr->yptr);
#ifdef UTF8
printf("\tセグメント長:%d", i);
printf("\t同音語数:%d\n", douon_num);
printf("漢字圧縮文字列:");
#ifdef ENGLISH
printf("\tSegment length:%d", i);
printf("\tSame pronunciation words:%d\n", douon_num);
printf("Kanji compression:");
#else
printf("\t\245\273\245\260\245\341\245\363\245\310\304\271:%d", i);
printf("\t\306\261\262\273\270\354\277\364:%d\n", douon_num);
@ -373,7 +373,12 @@ void makeseg() {
#endif
for(i = j = 0; i < askknj_num; i++) {
putchar('\t');
if(i && j == 0) putchar('\t');
if(i && j == 0) {
putchar('\t');
#ifdef ENGLISH
putchar('\t');
#endif
}
output_knj(stdout, askknj[i]->kptr, askknj[i]->klen);
printf(":%d", askknj[i]->hindo + askknj[i]->exist);
if(++j >= 4) {
@ -389,9 +394,9 @@ void makeseg() {
err:
fprintf(stderr,
#ifdef UTF8
"セグメント構成中にバッファがあふれた\n"
"セグメント先頭の読み:"
#ifdef ENGLISH
"Buffer overflew during constructing segment\n"
"Pronunciation of segment beginning:"
#else
"\245\273\245\260\245\341\245\363\245\310\271\275\300\256\303\346\244\313\245\320\245\303\245\325\245\241\244\254\244\242\244\325\244\354\244\277\n"
"\245\273\245\260\245\341\245\363\245\310\300\350\306\254\244\316\306\311\244\337:"

View file

@ -87,7 +87,7 @@ readchar() {
}
}
return n;
return unicode_to_eucjp(n);
#else
int c1;
int c2;
@ -154,28 +154,6 @@ retry:
if(Isillegal(c))
error(ILLEGALFORMAT);
#ifdef UTF8
if(c >= 0x10000) {
hinsi[i++] = ((c >> (6 * 3)) & 7) | 0xf0;
j = 3;
} else if(c >= 0x800) {
hinsi[i++] = ((c >> (6 * 2)) & 15) | 0xe0;
j = 2;
} else if(c >= 0x80) {
hinsi[i++] = ((c >> (6 * 1)) & 31) | 0xc0;
j = 1;
} else {
hinsi[i++] = c;
}
for(k = 0; k < j; k++) {
if(i >= 126) error(TOOLONGHINSI);
hinsi[i++] = ((c >> (6 * (j - k - 1))) & 63) | 0x80;
}
#else
if(c > 0xffffff) {
if(i >= 126)
error(TOOLONGHINSI);
@ -198,13 +176,10 @@ retry:
error(TOOLONGHINSI);
hinsi[i++] = ((c >> 8) & 0xff);
}
#endif
if(i >= 127)
error(TOOLONGHINSI);
#ifndef UTF8
hinsi[i++] = (c & 0xff);
#endif
c = readchar();
c = readchar();
}
if(i == 0)
@ -215,8 +190,8 @@ retry:
i = cnvhinsi(hinsi + 1);
if(i > 0) {
fprintf(stderr,
#ifdef UTF8
"品詞 \"%s\" に括弧がついている\n"
#ifdef ENGLISH
"Verb \"%s\" has parenthis\n"
#else
"\311\312\273\354 \"%s\" \244\313\263\347\270\314\244\254\244\304\244\244\244\306\244\244\244\353\n"
#endif
@ -225,8 +200,8 @@ retry:
1);
} else if(!i) {
fprintf(stderr,
#ifdef UTF8
"\"%s\" がコード化できません\n"
#ifdef ENGLISH
"Cannot turn \"%s\" into code\n"
#else
"\"%s\" \244\254\245\263\241\274\245\311\262\275\244\307\244\255\244\336\244\273\244\363\n"
#endif
@ -240,8 +215,8 @@ retry:
i = cnvhinsi(hinsi);
if(!i) {
fprintf(stderr,
#ifdef UTF8
"\"%s\" がコード化できません\n"
#ifdef ENGLISH
"Cannot turn \"%s\" into code\n"
#else
"\"%s\" \244\254\245\263\241\274\245\311\262\275\244\307\244\255\244\336\244\273\244\363\n"
#endif

View file

@ -73,25 +73,23 @@
#define MAXHINDONUMBER 2000
#define MAXOFFSETNUMBER 1000
#ifdef UTF8
#define ILLEGALFORMAT "フォーマットが異常です"
#define TOOLONGYOMI "読み文字列が長すぎます"
#define TOOLONGKANJI "漢字文字列が長すぎます"
#define TOOLONGHINSI "品詞文字列が長すぎます"
#define TOOLONGGROUP "グループ名が異常です"
#define TOOLONGJOSI "助詞文字列が異常です"
#define BADHINSI "登録されていない品詞です"
#define BADGROUP "登録されていないグループです"
#define BADJOSI "登録されていない助詞です"
#define NOYOMISTRING "読み文字列が取得できません"
#define NOKANJISTRING "漢字文字列が取得できません"
#define TOOMANYATR "属性の数が多すぎます"
#define TOOMANYHINSI "品詞の数が多すぎます"
#define TOOMANYJOSI "助詞の数が多すぎます"
#define NODATAINMAIN "メイン辞書に存在しません"
#define TOOMANYTARGET "品詞を1つ指定してください"
#define HINSITANKAN cnvhinsi((u_char*)"単漢")
#ifdef ENGLISH
#define ILLEGALFORMAT "Illegal format"
#define TOOLONGYOMI "Pronunciation too long"
#define TOOLONGKANJI "Kanji string too long"
#define TOOLONGHINSI "Verb too long"
#define TOOLONGGROUP "Illegal group"
#define TOOLONGJOSI "Illegal particle"
#define BADHINSI "Non-registered verb"
#define BADGROUP "Non-registered group"
#define BADJOSI "Non-registered particle"
#define NOYOMISTRING "Cannot get pronunciation string"
#define NOKANJISTRING "Cannot get kanji string"
#define TOOMANYATR "Too many attributes"
#define TOOMANYHINSI "Too many verbs"
#define TOOMANYJOSI "Too many particles"
#define NODATAINMAIN "Non-existent in main dictionary"
#define TOOMANYTARGET "Specify only one verb"
#else
#define ILLEGALFORMAT "\245\325\245\251\241\274\245\336\245\303\245\310\244\254\260\333\276\357\244\307\244\271"
#define TOOLONGYOMI "\306\311\244\337\312\270\273\372\316\363\244\254\304\271\244\271\244\256\244\336\244\271"
@ -109,9 +107,9 @@
#define TOOMANYJOSI "\275\365\273\354\244\316\277\364\244\254\302\277\244\271\244\256\244\336\244\271"
#define NODATAINMAIN "\245\341\245\244\245\363\274\255\275\361\244\313\302\270\272\337\244\267\244\336\244\273\244\363"
#define TOOMANYTARGET "\311\312\273\354\244\362\243\261\244\304\273\330\304\352\244\267\244\306\244\257\244\300\244\265\244\244"
#endif
#define HINSITANKAN cnvhinsi((u_char*)"\303\261\264\301")
#endif
#define FALSE 0
#define TRUE (!FALSE)

View file

@ -124,5 +124,6 @@ int top_strcmp(int*, int*);
int last_strcmp(int*, int*);
int string_cmp(u_char*, int, u_char*, int);
void utf8_print(char*, int);
int unicode_to_eucjp(int utf);
#endif /* SJ3MKDIC_H */

View file

@ -37,6 +37,8 @@
#include "sj3mkdic.h"
#include "ucstable.h"
int bubun_str(u_char* p1, int l1, u_char* p2, int l2) {
u_char* p;
int l;
@ -199,3 +201,23 @@ void utf8_print(char* out, int c) {
out[i++] = 0;
}
int unicode_to_eucjp(int utf) {
if(0 <= utf && utf <= 0xffff) {
const T_BITMAP_INDEX* b = &utf16_to_euc_jp_table[(utf >> 8) & 0xff];
if(b->byType == 2) {
if(b->dwBitmapIndex != 0) {
b = &utf16_to_euc_jp_table[b->dwBitmapIndex + (utf & 0xff)];
if(b->byType == 3) {
return b->dwEucJpCode;
}
}
}
return utf;
}
return -1;
}

23
src/sj3mkdic/ucstable.h Normal file

File diff suppressed because one or more lines are too long