# 【已上传】《汉语大词典》2.0 光盘版 软件内提取数据

**URL:** <https://forum.freemdict.com/t/topic/26220>\
**Category:** 技术交流与词典编修\
**Tags:** 词源\
**Created:** [2024 年1 月 19 日 02:18 UTC](https://forum.freemdict.com/t/topic/26220 "2024-01-19T02:18:21Z")\
**Posts on this page:** 1\
**Showing post:** 18

<div class="post-metadata">

**Author:** ![naosei7dd8](https://forumcdn.freemdict.com/letter_avatar_proxy/v4/letter/n/ecae2f/32.png) [@naosei7dd8](https://forum.freemdict.com/u/naosei7dd8)\
**Post date:** [2024 年1 月 20 日 04:24 UTC](https://forum.freemdict.com/t/topic/26220/18 "2024-01-20T04:24:09Z")

</div>

# 汉语大词典 2.0 光盘版 提取全过程

## 数据

[汉语大词典 2.0 光盘版 解密数据](https://cloud.freemdict.com/index.php/s/zwES7YkZY33ENTe) 共563M

在提取前将论坛中[abs前辈发的iso](https://forum.freemdict.com/t/topic/7883/92)和我手中的iso做比较，在数据部分(\*.FPT+\*.DAT)校验了哈希一致性，因此理论上用同一套代码提取的数据也是一样的。

其中只解密了:

· HYDC1(字音字义组词举例)

· HYDC2(字音字义以及所有字典词信息与详细解释，即[此处](https://forum.freemdict.com/t/topic/15998)网盘的"汉语大词典源数据合并"部分)

· HYDC8(看着像成语与详细解释)

· HYDC9(二字词语与详细解释)

· HYDCF(三字词语与详细解释)

· HYDCG(四字及以上词语与详细解释)

剩下的数据没看懂是什么含义，就没有解密。光盘版的这个软件功能很多很全，可以通过各种方式检索字，估计其他的库里就放着注音、偏旁部首、笔画等信息，有兴趣的可以自己解密。

再转换过程中，存在一些问题，其中HYDC2存在多条空白记录(不确定是什么作用，估计可能因为当时光盘软件制作时录入的问题)，其他部分也有编码失败的问题，分析了其中一个例子，应该是光盘或软件制作时就有的问题(比如一个字节变成空格)，软件版查时也不能正确编码。我在代码里指定如果编码错误则将对应字节用"backslashreplace"方法，如"\x78"，全局搜索\x即可找到。

因此，记录错误有两种方式体现，一种是\x，一种是空白记录，这也许会给这版本词典修订提供一些线索。

如果不关心技术只关心数据看完这就可以划走了。

## 基础背景

[这里是来自abs前辈发现的成果](https://forum.freemdict.com/t/topic/7883/96)

首先根据上述的成果反编译出汉语大词典2.0的代码，以及数据库相关的知识abs前辈也已经补充上了。只不过有一点，还原数据看上去并不需要SSS文件(即索引)，索引是类似于查询加速的作用，有了索引就不用每一条遍历地查询，因此并不需要SSS文件，只要有全套FPT和DAT就可以了。

前辈卡在了VfpDecompress上，以及下面回复的linbai前辈提到VfpDeCompress 和 VfpCompress函数是olefx32.dll的内置函数。然而网上并没有关于这两个函数的资料，因此直接IDA启动！

## 代码

```python
import codecs
from dbfread import DBF
import sqlite3
import os
from tqdm import tqdm

# 辅助函数：向左和向右旋转一个字节
def rol(byte, shift):
    return ((byte << shift) & 0xff) | (byte >> (8 - shift))

def ror(byte, shift):
    return (byte >> shift) | ((byte << (8 - shift)) & 0xff)

def is_bytearray_all_digits(byte_array):
    # 定义ASCII码的数字范围
    ASCII_ZERO = 48
    ASCII_NINE = 57

    # 检查bytearray中的每个元素是否为ASCII码表示的数字
    return all(ASCII_ZERO <= byte <= ASCII_NINE for byte in byte_array)

# 自定义编解码器类
class CustomCodec(codecs.Codec):

    # 初始化随机数生成器的状态
    def __init__ (self):
        self.v2 = 1 # initial state for the LCG

    # 编码方法
    def encode(self, input, errors='strict'):
        return input.encode("utf-8"), len(input)

    # 解码方法
    def decode(self, input, errors='strict'):
        # print(bytearray(input)[0])
        if bytearray(input)[0] == ord("F"):
            result = bytearray(input).decode("gb18030")
            return result, len(result)
        if is_bytearray_all_digits(bytearray(input)):
            result = bytearray(input).decode("gb18030")
            return result, len(result)
        try:
            result = decode_HYDC(bytearray(input))
            result = result.decode("gb18030")
            return result, len(result)
        except UnicodeDecodeError:
            # print(bytearray(input))
            result = bytearray(input).strip().decode("gb18030",errors='backslashreplace')
            return result, len(result)

def decode_HYDC(input):
    v2 = 1
    data = input
    # 确保在数据的末尾有一个 null 字节（0）
    data.append(0)
    # 数据的总长度
    total_length = len(data) - 1
    if total_length > 0:
        tmp_length = total_length
        while tmp_length > 0:
            # 执行循环左移操作
            data[total_length - tmp_length] = rol(data[total_length - tmp_length], 3)
            # 更新 v2 的值
            v2 = (1199 * v2 + 19782) % 0x1DF0DD
            # 更新数据字节
            data[total_length - tmp_length] ^= int((v2 * 0.0000045866768) // 1)
            # 移动到下一个字节
            tmp_length -= 1
    # 移除末尾的 null 字节并返回去混淆后的字节串
    result = bytes(data[:-1])
    return result

# 注册编解码器
def custom_codec_search_function(encoding):
    if encoding == 'customcodec':
        return codecs.CodecInfo(
            name='customcodec',
            encode=CustomCodec().encode,
            decode=CustomCodec().decode,
        )
    return None

class SQLiteDBHandler:
    def __init__ (self, db_file):
        self.db_file = db_file
        self.conn = None
        self._create_connection()

    def _create_connection(self):
        """创建数据库连接"""
        try:
            self.conn = sqlite3.connect(self.db_file)
        except sqlite3.Error as e:
            print(e)

    def _create_table(self, data_dict, table_name):
        """根据dict的键创建表"""
        columns = ", ".join([f"{key} TEXT" for key in data_dict.keys()])
        create_table_sql = f"CREATE TABLE IF NOT EXISTS {table_name} ({columns});"

        try:
            cursor = self.conn.cursor()
            cursor.execute(create_table_sql)
        except sqlite3.Error as e:
            print(e)

    def _insert_values(self, data_dict, table_name):
        """插入字典数据到表"""
        placeholders = ", ".join(["?"] * len(data_dict))
        columns = ", ".join(data_dict.keys())
        values = list(data_dict.values())

        insert_sql = f"INSERT INTO {table_name} ({columns}) VALUES ({placeholders})"

        try:
            cursor = self.conn.cursor()
            cursor.execute(insert_sql, values)
            self.conn.commit()
        except sqlite3.Error as e:
            print(e)

    def close_connection(self):
        """关闭数据库连接"""
        if self.conn:
            self.conn.close()

    def write_dict_to_table(self, table_name, data_dicts):
        """将字典数据写入到指定的SQLite表"""
        if not data_dicts:
            print("No data provided to write to the table.")
            return

        # 假设所有字典在data_dicts中具有相同的键
        first_dict = data_dicts[0]
        if self.conn is not None:
            self._create_table(first_dict, table_name)
            for data_dict in data_dicts:
                self._insert_values(data_dict, table_name)
        else:
            print("Connection not established.")

def check_and_process_dict(d, keys_to_check):
    """
    检查并处理字典中指定键的值，如果是数字则调用process函数处理。

    :param d: 待处理的字典。
    :param keys_to_check: 需要检查的键的列表。
    :return: None，字典将直接被修改。
    """
    for key in keys_to_check:
        value = str(d.get(key, '')) # 确保值是字符串类型
        if value.isdigit(): # 如果值是数字
            d[key] = decode_HYDC(bytearray(value.encode("gb18030"))).decode("gb18030") # 调用处理函数
    return d

def find_dbf_files(directory):
    dbf_files = []
    # os.walk遍历目录
    for root, dirs, files in os.walk(directory):
        for file in files:
            if file.lower().endswith('.dat'): # 检查文件后缀是否为.dbf
                relative_path = os.path.relpath(os.path.join(root, file), ".")
                dbf_files.append(relative_path) # 将相对路径添加到列表
    return dbf_files

# content_key_list = ["F1", "F3", "F6", "F7", "F9", "F11"]
HYDC1_key_list = ["F1", "F8"]
HYDC2_key_list = ["F1", "F4", "F5"]
HYDC8_key_list = ["F1", "F3", "F6", "F7", "F9", "F11"]
HYDC9_key_list = ["F1", "F3", "F6", "F7"]
HYDCF_key_list = ["F1", "F3", "F6", "F7", "F9"]
HYDCG_key_list = ["F1", "F3", "F6", "F7", "F9", "F11"]

def extract(name, key_list):
    db_handler = SQLiteDBHandler(f'{name}.db')
    db_handler.write_dict_to_table(name,
                                   [check_and_process_dict(dict(record), key_list)
                                    for record in DBF(os.path.join("HYDC", f"{name}.DAT"), encoding="customcodec")]
                                   )
    db_handler.close_connection()

if __name__ == ' __main__':
    codecs.register(custom_codec_search_function)

    pbar = tqdm(total=6)
    extract("HYDC1", HYDC1_key_list)
    pbar.update(1)
    extract("HYDC2", HYDC2_key_list)
    pbar.update(1)
    extract("HYDC8", HYDC8_key_list)
    pbar.update(1)
    extract("HYDC9", HYDC9_key_list)
    pbar.update(1)
    extract("HYDCF", HYDCF_key_list)
    pbar.update(1)
    extract("HYDCG", HYDCG_key_list)
    pbar.update(1)

```

## 逆向部分

我用了ida提取了olefx32.dll中VfpDecompress的函数逻辑，拉到gpt4里写的python解密代码，并用ollydbg(毕竟是32位，吾爱的od比x64dbg好用)动态调试验证…这里的工作其实就不是什么写出来就有的普适的经验，我先把数据+脚本发出来，逆向过程我考虑一下怎么写比较好

---

_[View the full topic](https://forum.freemdict.com/t/topic/26220)._
