|
阅读:3888回复:14
合并用户码表
灌水积分贴,合并用户码表逻辑
1、合并用户码表会将用户码表中的信息合并到主码表中; 2、当主码表和分词库中有相同的字词时,会将主码表中对应的字词删除; 3、执行“合并用户码表”时会对当前的“输入法”方案进行合并,不会影响其他“输入法”方案; 4、每执行一次“合并用户码表”时,只会对执行一次上面的判断逻辑,比如:
|
|
最新喜欢: |
|
沙发#
发布于:2026-09-18 17:16
看你们讨论合并用户词库,讨论了好多细节,我的个人意见是不要合并用户词库,保留主码表+用户词库+分词库的形式,首先有主码表,然后才有用户词库,这种情况下不会出现候选重复的,再加上分词库,只要分词库是单行单义的,即便与主码表有部分词语撞库了,小小也会自动过滤重复候选,候选框是不会出现重复候选的。分词库可以进行排序,加速程序读取,但是不要合并成单行多义的格式,看似缩小了码表行数整体文件体积变小了,但是与主码表、用户词库撞库后,出现重复候选就是不可避免地了。而且不合并用户词库也有利于对主码表的维护,合并后就彻底脱离了主码表的更新。
|
|
|
板凳#
发布于:2026-09-18 00:14
非必要,不优化!
拼音码表优化实例 文件格式提示:优化时请确保码表文件、脚本文件的编码为utf8,使用的换行符为unix格式! 1、删除码表中表头(保存好,最后恢复),只保留数据部分。将以^开头的编码复制到新文件用于最后的合并; 2、修整码表,将码表中开头为格式为“{6}”这样格式的编码复制到新文件,用于最后的恢复使用;删除码表中每行开头部位非字母的部分,如“{6}”、“^”,方法如下: 2.1、在VIM中使用/^[^a-z]可以快速查找开头不是小写字母的行;
2.2、使用:%s/^{\d\+}//g可删除每行开头形式如{6}的部分。
2.3、使用:%s/^\^//g可删除每行开头的^符号。
3、使用下面的akw脚本(假设保存为了fen.ahk),将修整后的码表转换为一码一字词的形式。命令行这样写:awk -f fen.awk 修整后的码表文件.txt > 分开后的码表.txt {
key = $1 # 第一个字段是键
for (i = 2; i <= NF; i++) { # 遍历剩余字段(合并后的值)
print key, $i # 输出键 + 单个值
}
}
4、删除“分开后的码表.txt”文件中重复的行,在VIM中使用下面的脚本去重可以保持去重后各行的前后顺序不变。提取所有单个字的行到新文件,后续恢复码表使用。提取单个字的文件可以使用VIM命令:v/ .$/d删除所有词组行,保留单字即可。 function! Remove() " 不改变文件顺序删除重复靠后的行
let i=1|g/^/s//\=i.'|'/|let i+=1
sort! /^\d\{-}|/ " 将行号后面的内容进行倒序排列
g/^\d\{-}|\(.*\)$\n\d\{-}|\1$/d " 删除与下一行内容相同的行
sort n
%s/\d\{-}|//
endfunction
5、对去重后的“分开后的码表.txt”文件进行分词。分词后仔细查看文件exception.txt中分词失败的词条,人工校验。分词使用的源代码如下: /*
* 分词设计思路
* 1、拆字:将汉字词组拆分为单个汉字列表。
* 2、查码:针对每个汉字,在单字词典中查出其对应的所有编码。
* 3、剪枝匹配:从前向后逐个处理。关键优化是根据词组编码当前首字母,只尝试该汉字中以该字母开头的编码候选,过滤掉首字母不匹配的编码,大幅减少无效尝试。
* 4、输出:匹配成功则用 ' 连接各字对应编码输出;匹配失败(首字母对不上或长度不匹配)则原样输出原始词组编码。
*/
/*
* 词组编码智能分割工具 v3.8(二分查找版)
*
* 1. 读取完所有单字后,使用 qsort 对 dict 按汉字 UTF-8 字符串排序
* 2. 排序后立即 memset(used, 0, sizeof(used)),确保 used 下标与排序后的 dict 对应
* 3. 查找函数 find_dict 改用二分查找(bsearch)
* 4. 构建词典阶段使用线性查找(find_dict_linear),因为此时词典尚未排序
*
* 编译:gcc -o split_pinyin split_pinyin_v38.c
* 运行:split_pinyin data.txt output.txt exception.txt
*/
#include <stdio.h>
#include <stdlib.h>
#include <string.h>
#include <ctype.h>
#define MAX_CHAR_WORDS 90000
#define PINYIN_MAX_LEN 50
#define PINYIN_PER_CHAR 10
#define PHRASE_MAX_LEN 200
#define MAX_PHRASES 200000
#define MAX_LINE_LEN 4096
/* ---------- 数据结构 ---------- */
typedef struct {
char chinese[5]; /* UTF-8 单字符 */
char pinyins[PINYIN_PER_CHAR][PINYIN_MAX_LEN];
int count;
} CharPinyin;
typedef struct {
char encoding[PINYIN_MAX_LEN * 10];
char chinese[PHRASE_MAX_LEN * 4];
} PhraseRecord;
CharPinyin dict[MAX_CHAR_WORDS];
int dict_count = 0;
int used[MAX_CHAR_WORDS];
PhraseRecord phrases[MAX_PHRASES];
int phrase_count = 0;
/* ---------- UTF-8 基础工具 ---------- */
int utf8_char_bytes(const unsigned char *s) {
unsigned char c = s[0];
if (c < 0x80) return 1;
if (c >= 0xC2 && c <= 0xDF) return 2;
if (c >= 0xE0 && c <= 0xEF) return 3;
if (c >= 0xF0 && c <= 0xF4) return 4;
return 0;
}
int utf8_char_count(const char *str) {
int count = 0;
const unsigned char *p = (const unsigned char*)str;
while (*p) {
int bytes = utf8_char_bytes(p);
if (bytes == 0 || bytes > (int)strlen((const char*)p)) break;
count++;
p += bytes;
}
return count;
}
int is_single_char(const char *str) {
if (!str || !*str) return 0;
return utf8_char_count(str) == 1;
}
int split_chinese(const char *chinese, char *chars[], int max_chars) {
int count = 0;
const unsigned char *p = (const unsigned char*)chinese;
while (*p && count < max_chars) {
int bytes = utf8_char_bytes(p);
if (bytes == 0 || bytes > (int)strlen((const char*)p)) break;
if (bytes == 1) {
if (!isspace(*p)) {
chars[count] = malloc(2);
chars[count][0] = *p;
chars[count][1] = '\0';
count++;
}
p++;
} else {
chars[count] = malloc(bytes + 1);
strncpy(chars[count], (const char*)p, bytes);
chars[count][bytes] = '\0';
count++;
p += bytes;
}
}
return count;
}
/* ---------- 词典操作 ---------- */
/* 线性查找(仅用于词典构建阶段) */
CharPinyin* find_dict_linear(const char *chinese) {
for (int i = 0; i < dict_count; i++) {
if (strcmp(dict<i>.chinese, chinese) == 0)
return &dict<i>;
}
return NULL;
}
/* qsort 比较函数 */
int cmp_charpinyin(const void *a, const void *b) {
const CharPinyin *pa = (const CharPinyin*)a;
const CharPinyin *pb = (const CharPinyin*)b;
return strcmp(pa->chinese, pb->chinese);
}
/* 排序并重置 used */
void sort_dict_and_reset_used(void) {
qsort(dict, dict_count, sizeof(CharPinyin), cmp_charpinyin);
memset(used, 0, sizeof(used)); /* 排序后 used 下标与 dict 对应 */
}
/* 二分查找(排序后使用) */
CharPinyin* find_dict(const char *chinese) {
return bsearch(chinese, dict, dict_count, sizeof(CharPinyin), cmp_charpinyin);
}
/* 添加词典条目(构建期使用线性查找) */
void add_to_dict(const char *chinese, const char *encoding) {
CharPinyin *cp = find_dict_linear(chinese);
if (cp != NULL) {
for (int i = 0; i < cp->count; i++)
if (strcmp(cp->pinyins<i>, encoding) == 0) return;
if (cp->count < PINYIN_PER_CHAR)
strcpy(cp->pinyins[cp->count++], encoding);
return;
}
if (dict_count >= MAX_CHAR_WORDS) return;
strcpy(dict[dict_count].chinese, chinese);
strcpy(dict[dict_count].pinyins[0], encoding);
dict[dict_count].count = 1;
dict_count++;
}
/* ---------- 文件解析 ---------- */
void parse_file(const char *filename) {
FILE *file = fopen(filename, "rb");
if (!file) { printf("错误:无法打开 %s\n", filename); exit(1); }
char line[MAX_LINE_LEN];
int first_line = 1;
while (fgets(line, sizeof(line), file)) {
size_t len = strlen(line);
while (len > 0 && (line[len-1] == '\n' || line[len-1] == '\r'))
line[--len] = '\0';
if (len == 0) continue;
if (first_line) {
unsigned char *p = (unsigned char*)line;
if (p[0] == 0xEF && p[1] == 0xBB && p[2] == 0xBF) {
memmove(line, line + 3, len - 3 + 1);
len -= 3;
}
first_line = 0;
}
if (len == 0) continue;
char *space = strchr(line, ' ');
if (!space) space = strchr(line, '\t');
if (!space) continue;
*space = '\0';
char *encoding = line;
char *chinese = space + 1;
while (*chinese == ' ' || *chinese == '\t') chinese++;
if (strlen(chinese) == 0) continue;
if (is_single_char(chinese)) {
add_to_dict(chinese, encoding);
} else {
if (phrase_count < MAX_PHRASES) {
strcpy(phrases[phrase_count].encoding, encoding);
strcpy(phrases[phrase_count].chinese, chinese);
phrase_count++;
}
}
}
fclose(file);
/* 单字收集完毕,排序并重置 used */
sort_dict_and_reset_used();
printf("词典:%d个单字,词组:%d条\n", dict_count, phrase_count);
}
/* ---------- 匹配算法(与 v3.6 相同) ---------- */
int match_phrase(const char *encoding, char *chars[], int char_count,
int idx, int pos, char result[][PINYIN_MAX_LEN])
{
int len = strlen(encoding);
if (idx == char_count) return (pos == len) ? 1 : 0;
if (pos >= len) return 0;
CharPinyin *cp = find_dict(chars[idx]);
if (!cp) return 0;
char first = encoding[pos];
for (int i = 0; i < cp->count; i++) {
const char *py = cp->pinyins<i>;
int py_len = strlen(py);
if (py[0] != first) continue;
if (pos + py_len > len) continue;
if (strncmp(encoding + pos, py, py_len) == 0) {
strcpy(result[idx], py);
if (match_phrase(encoding, chars, char_count, idx + 1, pos + py_len, result))
return 1;
}
}
return 0;
}
int match_phrase_brutal(const char *encoding, char *chars[], int char_count,
int idx, int pos, char result[][PINYIN_MAX_LEN])
{
int len = strlen(encoding);
if (idx == char_count) return (pos == len) ? 1 : 0;
if (pos >= len) return 0;
CharPinyin *cp = find_dict(chars[idx]);
if (!cp) return 0;
for (int i = 0; i < cp->count; i++) {
const char *py = cp->pinyins<i>;
int py_len = strlen(py);
if (pos + py_len > len) continue;
if (strncmp(encoding + pos, py, py_len) == 0) {
strcpy(result[idx], py);
if (match_phrase_brutal(encoding, chars, char_count, idx + 1, pos + py_len, result))
return 1;
}
}
return 0;
}
void process_phrase(FILE *out, FILE *exc, const char *encoding, const char *chinese) {
char *chars[PHRASE_MAX_LEN];
int count = split_chinese(chinese, chars, PHRASE_MAX_LEN);
if (count == 0) { fprintf(out, "%s %s\n", encoding, chinese); return; }
/* 检查缺字(二分查找) */
for (int i = 0; i < count; i++) {
if (!find_dict(chars<i>)) {
fprintf(exc, "%s %s\n", encoding, chinese);
fprintf(exc, " --> 缺字:'%s' 不在词典中\n", chars<i>);
for (int k = 0; k < count; k++) free(chars[k]);
return;
}
}
char result[PHRASE_MAX_LEN][PINYIN_MAX_LEN];
int ret = match_phrase(encoding, chars, count, 0, 0, result);
if (ret) {
for (int i = 0; i < count; i++) {
if (i > 0) fputc('\'', out);
fprintf(out, "%s", result<i>);
}
fprintf(out, " %s\n", chinese);
} else {
char brutal_result[PHRASE_MAX_LEN][PINYIN_MAX_LEN];
int bret = match_phrase_brutal(encoding, chars, count, 0, 0, brutal_result);
if (bret) {
int pos = 0;
for (int i = 0; i < count; i++) {
char expected = encoding[pos];
const char *py = brutal_result<i>;
int py_len = strlen(py);
if (py[0] != expected) {
fprintf(exc, "%s %s\n", encoding, chinese);
fprintf(exc, " --> 剪枝不匹配:第 %d 个字 '%s' 的正确编码为 '%s',其首字母 '%c' 不等于当前位置 %d 的期望首字母 '%c'。\n",
i + 1, chars<i>, py, py[0], pos, expected);
break;
}
pos += py_len;
}
} else {
fprintf(exc, "%s %s\n", encoding, chinese);
fprintf(exc, " --> 无匹配组合。\n");
fprintf(exc, " 词组拆分后各字及编码:\n");
for (int i = 0; i < count; i++) {
CharPinyin *cp = find_dict(chars<i>);
fprintf(exc, " '%s': ", chars<i>);
if (cp) {
for (int j = 0; j < cp->count; j++)
fprintf(exc, "%s ", cp->pinyins[j]);
} else {
fprintf(exc, "(无编码)");
}
fprintf(exc, "\n");
}
}
}
for (int i = 0; i < count; i++) free(chars<i>);
}
/* ---------- 主函数 ---------- */
int main(int argc, char *argv[]) {
const char *infile = "data.txt";
const char *outfile = "output.txt";
const char *excfile = "exception.txt";
if (argc >= 2) infile = argv[1];
if (argc >= 3) outfile = argv[2];
if (argc >= 4) excfile = argv[3];
printf("输入:%s\n输出:%s\n异常:%s\n\n", infile, outfile, excfile);
parse_file(infile);
if (dict_count == 0) {
printf("词典为空,请检查输入文件。\n");
return 1;
}
FILE *out = fopen(outfile, "w");
if (!out) { printf("无法创建 %s\n", outfile); return 1; }
FILE *exc = fopen(excfile, "w");
if (!exc) { printf("无法创建 %s\n", excfile); return 1; }
/* 处理词组,标记 used */
for (int i = 0; i < phrase_count; i++) {
char *chars[PHRASE_MAX_LEN];
int count = split_chinese(phrases<i>.chinese, chars, PHRASE_MAX_LEN);
for (int j = 0; j < count; j++) {
CharPinyin *cp = find_dict(chars[j]);
if (cp) {
int idx = (int)(cp - dict); /* 排序后下标映射到 used */
used[idx] = 1;
}
}
for (int j = 0; j < count; j++) free(chars[j]);
process_phrase(out, exc, phrases<i>.encoding, phrases<i>.chinese);
}
/* 输出未被词组使用的单字 */
for (int i = 0; i < dict_count; i++) {
if (!used<i>) {
for (int j = 0; j < dict<i>.count; j++)
fprintf(exc, "%s %s\n", dict<i>.pinyins[j], dict<i>.chinese);
}
}
fclose(out);
fclose(exc);
printf("\n处理完毕!共 %d 条词组。\n", phrase_count);
printf("正常结果 -> %s\n", outfile);
printf("异常数据 -> %s\n", excfile);
return 0;
}
6、对上一步加入分割符的词库使用下面的awk脚本(假设保存为了he.awk)进行并,命令行这样写:awk -f he.awk 已添加分隔符的词组文件.txt > 合并后的文件.txt;对提取的单个汉字文件进行合并,命令这样写:awk -f he.awk 一编码一汉字单字.txt > 一编码多汉字.txt BEGIN {
FS = "[ \t]+" # 字段间可能有多个空格或制表符
}
{
key = $1
value = ""
for (i = 2; i <= NF; i++) {
value = value (i == 2 ? "" : " ") $i
}
if (!(key in key_index)) {
# 第一次见到这个键,记录索引
key_index[key] = ++key_count
order[key_count] = key
# 初始化值(不加前导空格)
values[key] = value
} else {
# 追加值,确保格式整洁
values[key] = values[key] " " value
}
}
END {
# 按原始顺序输出
for (i = 1; i <= key_count; i++) {
key = order<i>
print key, values[key]
}
}7、删除“合并后的文件.txt”中的分隔符,使用第2步中复制出来的新文件恢复排序(…将码表中开头为格式为“{6}”这样格式的编码复制到新文件…)。将其与“一编码多汉字.txt”、第1步中保存的新文件进行合并。将合并后的文件进行排序,VIM的命令是:sort
8、添加码表表头,保存。完成优化码表的操作; |
|
|
地板#
发布于:2026-09-18 00:07
pfox:楼主,能否把awk和im上传?我网上下载的awk加载第一个脚本运行出错,不知道是不是awk的问题。回到原帖脚本就是帖子里的内容,执行失败极有可能和文件的编码或换行符设置有关,请仔细检查。使用匹配的换行符,保持脚本和处理文件编码的一致 换行符小知识:换行符(linux/windows)与跨平台文件操作 一文搞懂编码Unicode、utf-8、ISO-8859-1、GBK、ASCII 、扩展ASCII码、GB2312等知识 |
|
|
4楼#
发布于:2026-09-17 09:49
|
|
|
5楼#
发布于:2026-09-15 21:46
盘古大陆:使用输入法自带的合并用户码表,发现了以下两点不太符合我习惯的地方。拼音码表不要随便使用“码表合并”和“码表优化功能”。原因见:https://yong.dgod.net/read.php?tid=15&fid=7。 最好不要用合并用户码表和码表优化功能,因为相同编码的但实际不同的字词,如(昏暗、湖南)这样的会合并到一块,导致输入不正常。 拼音相同但实际读音不同的词不要放在一行(比如敏感和明暗)。 因为输入法为了加载速度,不会在同一行中去检测这个分词的问题。 若想优化需要自行解决拼音分词的问题,正确分词后再进行合并或优化就能完美解决问题了。这里提供一个分词的思路。前提先把编码转换为一个编码对应一个字符串的格式,也就是下面这样的形式: a 啊 a 阿 a 呵 a 腌 a 嗄 a 锕 a 吖 aba 阿坝 aba 阿爸 abao 阿宝 第二步:存储单个汉字对应的所有编码(如“重” 对应的编码“chong”“zhong”); 第三步:构建一个树形结构存储编码字符串数据,并在每个节点做以下判断。 ①、编码和对应的字符串是否已完全匹配,完全匹配表示成功输出对应的编码,每个编码之间可以用逗号进行分割; ②、记录字符串中第一个字符中所有能完全匹配编码开头字符的编码,生成下一级节点; ③、第 ② 步中若一个匹配的都没有,则返回上一个分岔处继续进行匹配; ④、所有的节点都无法匹配,表示匹配失败。原样输入编码和字符串。 |
|
|
6楼#
发布于:2026-05-19 10:00
|
|
|
7楼#
发布于:2026-05-19 09:26
|
|
|
8楼#
发布于:2026-05-19 09:09
|
|
|
9楼#
发布于:2026-05-19 09:03
|
|
上一页
下一页