WIP Unicode

This commit is contained in:
Nishi 2026-06-19 04:55:57 +09:00
commit f0700d97a6
6 changed files with 30126 additions and 7 deletions

View file

@ -6,7 +6,9 @@ file(GLOB_RECURSE SJ3CORE_SRCS lib/sj3core/**.c)
add_library(sj3core STATIC ${SJ3CORE_SRCS})
target_include_directories(sj3core PUBLIC include/sj3common include/sj3core)
target_include_directories(sj3core PRIVATE ${CMAKE_CURRENT_BINARY_DIR})
target_compile_definitions(sj3core PRIVATE UTF8)
file(GLOB_RECURSE SJ3MKDIC_SRCS src/sj3mkdic/**.c)
add_executable(sj3mkdic ${SJ3MKDIC_SRCS})
target_include_directories(sj3mkdic PRIVATE include/sj3common include/sj3core)
target_compile_definitions(sj3mkdic PRIVATE UTF8)

29803
dict/visual.utf8.dic Normal file

File diff suppressed because it is too large Load diff

198
src/sj3mkdic/GramTable.utf8 Normal file
View file

@ -0,0 +1,198 @@
{ "カ五1", 91 },
{ "カ五2", 101 },
{ "カ五3", 131 },
{ "カ五4", 141 },
{ "カ五5", 111 },
{ "カ五6", 121 },
{ "カ五7", 151 },
{ "カ五8", 161 },
{ "カ五音便", 185 },
{ "カ変仮", 181 },
{ "カ変終体", 180 },
{ "カ変未", 178 },
{ "カ変命", 182 },
{ "カ変用", 179 },
{ "ガ五1", 92 },
{ "ガ五2", 102 },
{ "ガ五3", 132 },
{ "ガ五4", 142 },
{ "ガ五5", 112 },
{ "ガ五6", 122 },
{ "ガ五7", 152 },
{ "ガ五8", 162 },
{ "サ五1", 93 },
{ "サ五2", 103 },
{ "サ五3", 133 },
{ "サ五4", 143 },
{ "サ五5", 113 },
{ "サ五6", 123 },
{ "サ五7", 153 },
{ "サ五8", 163 },
{ "サ変", 80 },
{ "サ変仮", 175 },
{ "サ変終体", 174 },
{ "サ変未1", 171 },
{ "サ変未2", 172 },
{ "サ変未用", 173 },
{ "サ変命1", 176 },
{ "サ変命2", 177 },
{ "ザ変", 81 },
{ "タ五1", 94 },
{ "タ五2", 104 },
{ "タ五3", 134 },
{ "タ五4", 144 },
{ "タ五5", 114 },
{ "タ五6", 124 },
{ "タ五7", 154 },
{ "タ五8", 164 },
{ "ナ五", 95 },
{ "バ五1", 96 },
{ "バ五2", 105 },
{ "バ五3", 135 },
{ "バ五4", 145 },
{ "バ五5", 115 },
{ "バ五6", 125 },
{ "バ五7", 155 },
{ "バ五8", 165 },
{ "マ五1", 97 },
{ "マ五2", 106 },
{ "マ五3", 136 },
{ "マ五4", 146 },
{ "マ五5", 116 },
{ "マ五6", 126 },
{ "マ五7", 156 },
{ "マ五8", 166 },
{ "ラ五1", 98 },
{ "ラ五2", 107 },
{ "ラ五3", 137 },
{ "ラ五4", 147 },
{ "ラ五5", 117 },
{ "ラ五6", 127 },
{ "ラ五7", 157 },
{ "ラ五8", 167 },
{ "ワ五1", 99 },
{ "ワ五2", 108 },
{ "ワ五3", 138 },
{ "ワ五4", 148 },
{ "ワ五5", 118 },
{ "ワ五6", 128 },
{ "ワ五7", 158 },
{ "ワ五8", 168 },
{ "挨拶", 187 },
{ "衣服", -11 },
{ "一括", 200 },
{ "一段1", 90 },
{ "一段2", 100 },
{ "一段3", 130 },
{ "一段4", 140 },
{ "雨", -21 },
{ "音楽", -20 },
{ "海", -31 },
{ "開閉", -4 },
{ "活動", -5 },
{ "感動", 28 },
{ "企業", 23 },
{ "魚", -14 },
{ "形1", 60 },
{ "形10", 69 },
{ "形11", 70 },
{ "形2", 61 },
{ "形3", 62 },
{ "形4", 63 },
{ "形5", 64 },
{ "形6", 65 },
{ "形7", 66 },
{ "形8", 67 },
{ "形9", 68 },
{ "形動1", 71 },
{ "形動2", 72 },
{ "形動3", 73 },
{ "形動4", 74 },
{ "形動5", 75 },
{ "形動6", 76 },
{ "形動7", 77 },
{ "形動8", 78 },
{ "形動9", 79 },
{ "建築", -13 },
{ "県区", 25 },
{ "湖", -33 },
{ "国", -12 },
{ "試合", -10 },
{ "事件", -27 },
{ "時", -25 },
{ "酒", -23 },
{ "州", -29 },
{ "出版", -18 },
{ "所", -2 },
{ "助数", 29 },
{ "助数2", 54 },
{ "乗物", -17 },
{ "植物", -6 },
{ "食物", -16 },
{ "身体", -8 },
{ "人", -1 },
{ "数詞", 30 },
{ "数詞2", 55 },
{ "接続", 27 },
{ "接頭1", 31 },
{ "接頭2", 32 },
{ "接頭3", 33 },
{ "接頭4", 34 },
{ "接頭5", 35 },
{ "接尾1", 36 },
{ "接尾2", 37 },
{ "接尾3", 38 },
{ "接尾4", 39 },
{ "接尾5", 40 },
{ "接尾6", 41 },
{ "接尾7", 42 },
{ "接尾8", 43 },
{ "接尾9", 44 },
{ "川", -30 },
{ "組織", -3 },
{ "贈物", -26 },
{ "代1", 12 },
{ "代2", 13 },
{ "代3", 14 },
{ "代4", 15 },
{ "代5", 16 },
{ "代6", 17 },
{ "単漢", 189 },
{ "地位", -7 },
{ "地名", 24 },
{ "丁寧1", 183 },
{ "丁寧2", 184 },
{ "電車", -34 },
{ "都市", -24 },
{ "島", -32 },
{ "動物", -9 },
{ "特殊形容", 188 },
{ "特殊副", 186 },
{ "病気", -15 },
{ "苗字", 21 },
{ "副1", 45 },
{ "副2", 46 },
{ "副3", 47 },
{ "副4", 48 },
{ "副5", 49 },
{ "副6", 50 },
{ "副7", 51 },
{ "副8", 52 },
{ "副9", 53 },
{ "文字", -19 },
{ "法", -35 },
{ "名1", 1 },
{ "名10", 10 },
{ "名11", 11 },
{ "名2", 2 },
{ "名20", 18 },
{ "名3", 3 },
{ "名4", 4 },
{ "名5", 5 },
{ "名6", 6 },
{ "名7", 7 },
{ "名8", 8 },
{ "名9", 9 },
{ "名前", 22 },
{ "鳴物", -28 },
{ "連体", 26 }

View file

@ -40,7 +40,11 @@ static struct gram_code {
int code;
} gramtbl[] = {
#ifdef UTF8
#include "GramTable.utf8"
#else
#include "GramTable"
#endif
};
@ -61,10 +65,17 @@ int u_strcmp(u_char* a, u_char* b) {
}
int cnvhinsi(u_char* buf) {
int i;
#if 1
for(i = 0; i <= GramMax; i++) {
if(u_strcmp(buf, (u_char*)gramtbl[i].name) == 0) return gramtbl[i].code;
}
return 0;
#else
int min;
int max;
int mid;
int i;
min = 0;
max = GramMax;
@ -86,6 +97,7 @@ int cnvhinsi(u_char* buf) {
return (gramtbl[mid].code);
}
}
#endif
return 0;
}

View file

@ -55,8 +55,40 @@ error(char* s) {
exit(1);
}
static const char utf8_bytes[256] = {
0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0,
1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1,
1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1,
1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 0,
0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0,
0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0,
0, 0, 2, 2, 2, 2, 2, 2, 2, 2, 2, 2, 2, 2, 2, 2, 2, 2, 2, 2, 2, 2, 2, 2, 2, 2, 2, 2, 2, 2, 2, 2,
3, 3, 3, 3, 3, 3, 3, 3, 3, 3, 3, 3, 3, 3, 3, 3, 4, 4, 4, 4, 4, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0};
static int
readchar() {
#ifdef UTF8
int c[4];
int i = 0, j;
int n = 0;
do {
if((c[i++] = getch()) == EOF) return EOF;
} while(i < utf8_bytes[c[0]]);
if(i == 1) {
n = c[0];
} else {
n |= c[0] & (0xff >> (8 - i));
for(j = 1; j < i; j++) {
n = n << 6;
n |= c[j] & 63;
}
}
return n;
#else
int c1;
int c2;
@ -82,6 +114,7 @@ readchar() {
c1 = ((c1 & 0xffff) << 8) + (c2 & 0xff);
}
return c1;
#endif
}
static int
@ -114,10 +147,36 @@ retry:
}
for(i = 0; !Isblank(c);) {
int j = 0, k;
if(c == '\n' || c == ':')
break;
if(Isillegal(c))
error(ILLEGALFORMAT);
#ifdef UTF8
printf("%x\n", c);
if(c >= 0x10000) {
hinsi[i++] = ((c >> (6 * 3)) & 7) | 0xf0;
j = 3;
} else if(c >= 0x800) {
hinsi[i++] = ((c >> (6 * 2)) & 15) | 0xe0;
j = 2;
} else if(c >= 0x80) {
hinsi[i++] = ((c >> (6 * 1)) & 31) | 0xc0;
j = 1;
} else {
hinsi[i++] = c;
}
for(k = 0; k < j; k++) {
if(i >= 126) error(TOOLONGHINSI);
hinsi[i++] = ((c >> (6 * (j - k - 1))) & 63) | 0x80;
}
#else
if(c > 0xffffff) {
if(i >= 126)
error(TOOLONGHINSI);
@ -140,10 +199,13 @@ retry:
error(TOOLONGHINSI);
hinsi[i++] = ((c >> 8) & 0xff);
}
#endif
if(i >= 127)
error(TOOLONGHINSI);
#ifndef UTF8
hinsi[i++] = (c & 0xff);
c = readchar();
#endif
c = readchar();
}
if(i == 0)
@ -153,18 +215,39 @@ retry:
hinsi[i - 1] = 0;
i = cnvhinsi(hinsi + 1);
if(i > 0) {
fprintf(stderr, "\311\312\273\354 \"%s\" \244\313\263\347\270\314\244\254\244\304\244\244\244\306\244\244\244\353\n",
hinsi + 1);
fprintf(stderr,
#ifdef UTF8
"品詞 \"%s\" に括弧がついている\n"
#else
"\311\312\273\354 \"%s\" \244\313\263\347\270\314\244\254\244\304\244\244\244\306\244\244\244\353\n"
#endif
,
hinsi +
1);
} else if(!i) {
fprintf(stderr, "\"%s\" \244\254\245\263\241\274\245\311\262\275\244\307\244\255\244\336\244\273\244\363\n",
hinsi + 1);
fprintf(stderr,
#ifdef UTF8
"\"%s\" がコード化できません\n"
#else
"\"%s\" \244\254\245\263\241\274\245\311\262\275\244\307\244\255\244\336\244\273\244\363\n"
#endif
,
hinsi +
1);
goto retry;
}
} else {
hinsi[i] = 0;
i = cnvhinsi(hinsi);
if(!i) {
fprintf(stderr, "\"%s\" \244\254\245\263\241\274\245\311\262\275\244\307\244\255\244\336\244\273\244\363\n", hinsi);
fprintf(stderr,
#ifdef UTF8
"\"%s\" がコード化できません\n"
#else
"\"%s\" \244\254\245\263\241\274\245\311\262\275\244\307\244\255\244\336\244\273\244\363\n"
#endif
,
hinsi);
goto retry;
}
}

View file

@ -73,6 +73,26 @@
#define MAXHINDONUMBER 2000
#define MAXOFFSETNUMBER 1000
#ifdef UTF8
#define ILLEGALFORMAT "フォーマットが異常です"
#define TOOLONGYOMI "読み文字列が長すぎます"
#define TOOLONGKANJI "漢字文字列が長すぎます"
#define TOOLONGHINSI "品詞文字列が長すぎます"
#define TOOLONGGROUP "グループ名が異常です"
#define TOOLONGJOSI "助詞文字列が異常です"
#define BADHINSI "登録されていない品詞です"
#define BADGROUP "登録されていないグループです"
#define BADJOSI "登録されていない助詞です"
#define NOYOMISTRING "読み文字列が取得できません"
#define NOKANJISTRING "漢字文字列が取得できません"
#define TOOMANYATR "属性の数が多すぎます"
#define TOOMANYHINSI "品詞の数が多すぎます"
#define TOOMANYJOSI "助詞の数が多すぎます"
#define NODATAINMAIN "メイン辞書に存在しません"
#define TOOMANYTARGET "品詞を1つ指定してください"
#define HINSITANKAN cnvhinsi((u_char*)"単漢")
#else
#define ILLEGALFORMAT "\245\325\245\251\241\274\245\336\245\303\245\310\244\254\260\333\276\357\244\307\244\271"
#define TOOLONGYOMI "\306\311\244\337\312\270\273\372\316\363\244\254\304\271\244\271\244\256\244\336\244\271"
#define TOOLONGKANJI "\264\301\273\372\312\270\273\372\316\363\244\254\304\271\244\271\244\256\244\336\244\271"
@ -91,6 +111,7 @@
#define TOOMANYTARGET "\311\312\273\354\244\362\243\261\244\304\273\330\304\352\244\267\244\306\244\257\244\300\244\265\244\244"
#define HINSITANKAN cnvhinsi((u_char*)"\303\261\264\301")
#endif
#define FALSE 0
#define TRUE (!FALSE)