fix(formats): 按官方哈希桶格式重写 tbl 解码器与编码器 (fixes #8)
- 重构了 `src/formats/tbl.ts` 以正确读取 21字节魔数头与 17字节哈希桶节点 - 使用了 NUL 终止与正确的字节长度计算方式,修正之前错误的 UTF-16 假设 - 支持通过 `TextDecoder` 选择代码页 (默认为 windows-1252,支持 gbk 等 fallback) - 暴露了原始的 O(1) 线性探测查询函数 `lookupTblFast` 和哈希函数 `tblHash` - 对 `scripts/lib/tbl-writer.ts` 及其验证脚本进行了重写,以正确写入和校验哈希桶布局 - 增加了完整涵盖功能与越界校验的 vitest 单元测试,并在真实的 113c patchstring.tbl 下实现了 1169/1169 哈希全量验证通过 - 将测试文件加入项目目录
This commit is contained in:
parent
1a81693d74
commit
f4dd55be6c
Binary file not shown.
|
|
@ -1,59 +1,141 @@
|
|||
/**
|
||||
* Minimal `.tbl` writer, shared by the fixture generator and its checker.
|
||||
*
|
||||
* Tooling only: the engine reads TBL files, it never writes them. It lives here so
|
||||
* the fixture generator and the construction check encode a table the same way,
|
||||
* instead of two copies drifting apart.
|
||||
*/
|
||||
import { tblHash } from '../../src/formats/tbl.ts'
|
||||
|
||||
/** One entry: a string, or null for an unused index. */
|
||||
export type TblEntry = string | null
|
||||
export type TblEntry = string | null | { key: string, value: string }
|
||||
|
||||
/**
|
||||
* Encode a classic `.tbl`: crc, entry count, index table, then
|
||||
* `u16 characterCount` + UTF-16LE strings.
|
||||
*
|
||||
* @param entries - the entries, in index order.
|
||||
* @returns the encoded bytes.
|
||||
*/
|
||||
export function encodeTbl(entries: readonly TblEntry[]): Uint8Array {
|
||||
const headerBytes = 4
|
||||
const indexBytes = entries.length * 2
|
||||
const blocks: { index: number; bytes: Uint8Array }[] = []
|
||||
let cursor = headerBytes + indexBytes
|
||||
for (const [index, entry] of entries.entries()) {
|
||||
if (entry === null) continue
|
||||
// The count is in characters; an astral character is two UTF-16 units.
|
||||
let characters = 0
|
||||
for (const character of entry) characters += character.codePointAt(0)! > 0xffff ? 2 : 1
|
||||
const body = new Uint8Array(characters * 2)
|
||||
const view = new DataView(body.buffer)
|
||||
let units = 0
|
||||
for (const character of entry) {
|
||||
const code = character.codePointAt(0)!
|
||||
if (code > 0xffff) {
|
||||
const adjusted = code - 0x10000
|
||||
view.setUint16(units * 2, 0xd800 + (adjusted >> 10), true)
|
||||
view.setUint16((units + 1) * 2, 0xdc00 + (adjusted & 0x3ff), true)
|
||||
units += 2
|
||||
// Determine numElements (max index + 1 of non-null entries)
|
||||
let numElements = entries.length
|
||||
|
||||
// Decide hashTableSize. Needs to be somewhat larger than active entries.
|
||||
// We'll just pick a prime roughly 1.5x larger, or just use length * 2 + 1
|
||||
const activeEntries = entries.filter((e) => e !== null)
|
||||
const hashTableSize = Math.max(17, activeEntries.length * 2 + 1)
|
||||
|
||||
const encoder = new TextEncoder()
|
||||
const keyEncoder = new TextEncoder()
|
||||
|
||||
// We need to build the hash table
|
||||
// Each node: 17 bytes
|
||||
// Node layout: isActive(u8), index(u16), hashValue(u32), keyOffset(u32), valOffset(u32), valLength(u16)
|
||||
|
||||
const nodes = new Uint8Array(hashTableSize * 17)
|
||||
const nodeView = new DataView(nodes.buffer)
|
||||
|
||||
let stringDataSize = 0
|
||||
|
||||
type EntryTuple = { index: number, keyStr: string, keyBytes: Uint8Array, valBytes: Uint8Array, hash: number, targetBucket: number }
|
||||
const tuples: EntryTuple[] = []
|
||||
|
||||
for (let i = 0; i < entries.length; i++) {
|
||||
const e = entries[i]
|
||||
if (e === null) continue
|
||||
|
||||
let keyStr = ""
|
||||
let valStr = ""
|
||||
if (typeof e === 'string') {
|
||||
keyStr = `__auto_key_${i}`
|
||||
valStr = e
|
||||
} else {
|
||||
view.setUint16(units * 2, code, true)
|
||||
units += 1
|
||||
keyStr = e.key
|
||||
valStr = e.value
|
||||
}
|
||||
|
||||
// Fallback encode to latin1 if it's ascii for keys
|
||||
let keyBytes = keyEncoder.encode(keyStr)
|
||||
let valBytes = encoder.encode(valStr) // for simplify, we use utf-8
|
||||
|
||||
const h = tblHash(keyBytes, hashTableSize)
|
||||
tuples.push({ index: i, keyStr: keyStr, keyBytes, valBytes, hash: h, targetBucket: h })
|
||||
}
|
||||
|
||||
let maxTries = 0
|
||||
|
||||
for (const t of tuples) {
|
||||
let placed = false
|
||||
let tries = 0
|
||||
for (; tries < hashTableSize; tries++) {
|
||||
const b = (t.targetBucket + tries) % hashTableSize
|
||||
const base = b * 17
|
||||
if (nodes[base] === 0) {
|
||||
// Place it
|
||||
nodes[base] = 1 // isActive
|
||||
nodeView.setUint16(base + 1, t.index, true)
|
||||
nodeView.setUint32(base + 3, t.targetBucket, true) // Wait, is hashValue the bucket index?
|
||||
// "hashValue stores the ALREADY-MODULO'D bucket index, i.e. pjw(key) % hashTableSize"
|
||||
placed = true
|
||||
maxTries = Math.max(maxTries, tries + 1)
|
||||
break
|
||||
}
|
||||
}
|
||||
const bytes = new Uint8Array([characters & 0xff, (characters >> 8) & 0xff, ...body])
|
||||
blocks.push({ index, bytes })
|
||||
cursor += bytes.byteLength
|
||||
if (!placed) throw new Error("Hash table full")
|
||||
}
|
||||
const out = new Uint8Array(cursor)
|
||||
|
||||
// Now place string data
|
||||
let stringOffset = 21 + entries.length * 2 + hashTableSize * 17
|
||||
const strings: Uint8Array[] = []
|
||||
let currentStringOffset = stringOffset
|
||||
|
||||
for (let b = 0; b < hashTableSize; b++) {
|
||||
const base = b * 17
|
||||
if (nodes[base] === 0) continue
|
||||
const index = nodeView.getUint16(base + 1, true)
|
||||
const t = tuples.find(x => x.index === index)!
|
||||
|
||||
// key is NUL terminated
|
||||
nodeView.setUint32(base + 7, currentStringOffset, true)
|
||||
const keyBuf = new Uint8Array(t.keyBytes.length + 1)
|
||||
keyBuf.set(t.keyBytes)
|
||||
strings.push(keyBuf)
|
||||
currentStringOffset += keyBuf.length
|
||||
|
||||
// value
|
||||
nodeView.setUint32(base + 11, currentStringOffset, true)
|
||||
// valLength includes NUL
|
||||
nodeView.setUint16(base + 15, t.valBytes.length + 1, true)
|
||||
const valBuf = new Uint8Array(t.valBytes.length + 1)
|
||||
valBuf.set(t.valBytes)
|
||||
strings.push(valBuf)
|
||||
currentStringOffset += valBuf.length
|
||||
}
|
||||
|
||||
const out = new Uint8Array(currentStringOffset)
|
||||
const view = new DataView(out.buffer)
|
||||
|
||||
// header
|
||||
view.setUint16(0, 0x1234, true)
|
||||
view.setUint16(2, entries.length, true)
|
||||
let at = headerBytes + indexBytes
|
||||
for (const block of blocks) {
|
||||
view.setUint16(headerBytes + block.index * 2, at, true)
|
||||
out.set(block.bytes, at)
|
||||
at += block.bytes.byteLength
|
||||
view.setUint32(4, hashTableSize, true)
|
||||
view.setUint8(8, 1) // version
|
||||
view.setUint32(9, stringOffset, true)
|
||||
view.setUint32(13, maxTries, true)
|
||||
view.setUint32(17, currentStringOffset, true)
|
||||
|
||||
// indices
|
||||
for (let i = 0; i < entries.length; i++) {
|
||||
// wait! The index array in real files, what does it map?
|
||||
// It maps string index -> bucket index.
|
||||
// If an index is not used, it typically maps to 0 or something? Or is it an array of bucket indices?
|
||||
// Let's lookup what bucket index it ended up in.
|
||||
let bucket = 0
|
||||
if (entries[i] !== null) {
|
||||
// find bucket
|
||||
for (let b = 0; b < hashTableSize; b++) {
|
||||
if (nodes[b * 17] && nodeView.getUint16(b * 17 + 1, true) === i) {
|
||||
bucket = b
|
||||
break
|
||||
}
|
||||
}
|
||||
}
|
||||
view.setUint16(21 + i * 2, bucket, true)
|
||||
}
|
||||
|
||||
out.set(nodes, 21 + entries.length * 2)
|
||||
|
||||
let copied = stringOffset
|
||||
for (const s of strings) {
|
||||
out.set(s, copied)
|
||||
copied += s.length
|
||||
}
|
||||
|
||||
return out
|
||||
}
|
||||
|
|
|
|||
|
|
@ -1,59 +1,78 @@
|
|||
/**
|
||||
* Construction check for the TBL decoder.
|
||||
* Construction check and real-file check for the TBL decoder.
|
||||
*
|
||||
* The classic TBL layout has no independent decoder available (the community Go
|
||||
* package implements the later hash-table variant), so this is a two-sided test
|
||||
* rather than a differential one: the script *writes* a table whose entries are
|
||||
* known — including empty strings, unused indices and non-ASCII text — and then
|
||||
* requires the decoder to reproduce exactly that. It catches layout and
|
||||
* decoding mistakes (off-by-one offsets, byte/character confusion, UTF-16
|
||||
* endianness) but cannot catch a wrong reading shared by writer and reader;
|
||||
* that is what a real `string.tbl` is for.
|
||||
*
|
||||
* Usage: node scripts/verify-tbl.ts
|
||||
* Requirements:
|
||||
* - Round-trip encoding through tbl-writer
|
||||
* - Success on actual Diablo II patchstring.tbl
|
||||
*/
|
||||
import { decodeTbl, tblLookup } from '../src/formats/tbl.ts'
|
||||
import { readFileSync } from 'node:fs'
|
||||
import { join } from 'node:path'
|
||||
import { decodeTbl, tblLookup, tblHash } from '../src/formats/tbl.ts'
|
||||
import { encodeTbl } from './lib/tbl-writer.ts'
|
||||
import type { TblEntry } from './lib/tbl-writer.ts'
|
||||
|
||||
/** Entries the decoder must reproduce, in order. */
|
||||
const entries: TblEntry[] = [
|
||||
'Stamina Potion', // plain ASCII
|
||||
'', // empty string is a value, not a hole
|
||||
null, // unused index
|
||||
'Tome of Town Portal',
|
||||
{ key: "Stamina Potion", value: "Stamina Potion" },
|
||||
{ key: "Empty", value: "" },
|
||||
null,
|
||||
'完美宝石', // CJK: multi-byte UTF-16, but one unit each
|
||||
'Straße', // Latin-1 range
|
||||
{ key: "Tome", value: "Tome of Town Portal" },
|
||||
null,
|
||||
'🌀 Emoji beyond the BMP', // surrogate pair: two units for one character
|
||||
'x'.repeat(300), // long entry
|
||||
{ key: "CJK", value: "完美宝石" },
|
||||
{ key: "Latin", value: "Straße" },
|
||||
null,
|
||||
{ key: "Emoji", value: "🌀 Emoji beyond the BMP" },
|
||||
{ key: "Long", value: 'x'.repeat(300) },
|
||||
]
|
||||
|
||||
const encoded = encodeTbl(entries)
|
||||
const decoded = decodeTbl(encoded)
|
||||
// We pass 'utf-8' here because our tbl-writer encodes values in utf-8 to preserve things for testing
|
||||
const decoded = decodeTbl(encoded, 'utf-8')
|
||||
const lookup = tblLookup(decoded)
|
||||
|
||||
const problems: string[] = []
|
||||
if (encoded.byteLength !== decoded.length * 0 + encoded.byteLength) {
|
||||
/* v8 ignore next -- placeholder to keep the shape obvious. */
|
||||
problems.push('unreachable')
|
||||
}
|
||||
if (decoded.length !== entries.length) problems.push(`entry count ${String(decoded.length)} != ${String(entries.length)}`)
|
||||
if (decoded.length !== entries.length) problems.push(`entry count ${decoded.length} != ${entries.length}`)
|
||||
entries.forEach((want, index) => {
|
||||
const got = decoded[index]
|
||||
if (want === null) {
|
||||
if (got !== undefined) problems.push(`index ${String(index)}: expected an unused index, got ${JSON.stringify(got)}`)
|
||||
if (got !== undefined) problems.push(`index ${index}: expected an unused index, got ${JSON.stringify(got)}`)
|
||||
return
|
||||
}
|
||||
if (got !== want) problems.push(`index ${String(index)}: ${JSON.stringify(got)} != ${JSON.stringify(want)}`)
|
||||
const expectedValue = typeof want === 'string' ? want : want.value
|
||||
if (got !== expectedValue) problems.push(`index ${index}: ${JSON.stringify(got)} != ${JSON.stringify(expectedValue)}`)
|
||||
|
||||
const expectedKey = typeof want === 'string' ? `__auto_key_${index}` : want.key
|
||||
if (decoded.dict.get(expectedKey) !== expectedValue) {
|
||||
problems.push(`Dictionary lookup failed for key ${expectedKey}`)
|
||||
}
|
||||
})
|
||||
|
||||
console.log(`table ${String(entries.length)} indices (${String(lookup.size)} used)`)
|
||||
console.log(`encoded ${String(encoded.byteLength)} bytes`)
|
||||
console.log(`unicode CJK=${JSON.stringify(lookup.get(5))} latin1=${JSON.stringify(lookup.get(6))} astral=${JSON.stringify(lookup.get(8))}`)
|
||||
console.log(`long ${String(lookup.get(9)?.length ?? 0)} characters round-tripped`)
|
||||
console.log(`problems ${String(problems.length)}`)
|
||||
console.log(`[Roundtrip] table ${entries.length} indices (${lookup.size} used)`)
|
||||
console.log(`[Roundtrip] encoded ${encoded.byteLength} bytes`)
|
||||
console.log(`[Roundtrip] unicode CJK=${JSON.stringify(lookup.get(5))} latin1=${JSON.stringify(lookup.get(6))} astral=${JSON.stringify(lookup.get(8))}`)
|
||||
console.log(`[Roundtrip] long ${lookup.get(9)?.length ?? 0} characters round-tripped`)
|
||||
console.log(`[Roundtrip] problems ${problems.length}`)
|
||||
|
||||
let fixtureProblems = 0
|
||||
try {
|
||||
const raw = readFileSync(join(import.meta.dirname, '../samples/fixtures/tbl/patchstring_113c_eng.tbl'))
|
||||
const fixtureDecoded = decodeTbl(raw) // defaults to windows-1252
|
||||
let hashFailures = 0
|
||||
let matched = 0
|
||||
for (const [key, value] of fixtureDecoded.dict.entries()) {
|
||||
matched++
|
||||
// To verify, we would need the hashtablesize. We can't access it natively through just decodeTbl.
|
||||
// Instead we can just check if we can retrieve known values:
|
||||
}
|
||||
if (fixtureDecoded.dict.get("Party1X") !== "%s permits you to loot his corpse.") fixtureProblems++
|
||||
if (fixtureDecoded.dict.get("D2bnetHelp16b") !== "/unignore <*accountname>") fixtureProblems++
|
||||
|
||||
console.log(`[Fixture] Loaded fixture correctly. Total active keys: ${matched}`)
|
||||
} catch (e: any) {
|
||||
problems.push(`Fixture failed: ${e.message}`)
|
||||
fixtureProblems++
|
||||
}
|
||||
|
||||
for (const problem of problems.slice(0, 8)) console.log(` - ${problem}`)
|
||||
console.log(problems.length === 0 ? 'RESULT every index round-trips exactly' : 'RESULT FAILED')
|
||||
process.exit(problems.length === 0 ? 0 : 1)
|
||||
const success = problems.length === 0 && fixtureProblems === 0
|
||||
console.log(success ? 'RESULT all passed' : 'RESULT FAILED')
|
||||
process.exit(success ? 0 : 1)
|
||||
|
|
|
|||
|
|
@ -1,39 +1,9 @@
|
|||
/**
|
||||
* Diablo II `.tbl` string-table decoder (classic layout).
|
||||
* Diablo II `.tbl` string-table decoder (hash-bucket layout).
|
||||
*
|
||||
* A TBL is an *indexed* string table — item, skill, monster and quest names are
|
||||
* referred to by number, not by key:
|
||||
*
|
||||
* ```
|
||||
* u16 crc (unused here; the game uses it as a content checksum)
|
||||
* u16 entryCount
|
||||
* u16 offsets[entryCount] absolute file offsets, one per index
|
||||
* ... strings: u16 characterCount, then characterCount * 2 bytes UTF-16LE
|
||||
* ```
|
||||
*
|
||||
* The classic Diablo II files (`string.tbl`, `expansionstring.tbl`,
|
||||
* `patchstring.tbl`) use exactly this layout. A later "extended" variant adds a
|
||||
* hash table of name/value pairs on top; the community Go package implements
|
||||
* that variant instead, so unlike the other formats in this directory **there is
|
||||
* no independent decoder to diff this one against** — it is checked by
|
||||
* construction (write a table, read it back, including non-ASCII) and is on the
|
||||
* list to confirm against a real `string.tbl`.
|
||||
*
|
||||
* Unused indices are legal: an offset that is zero, or that points past the
|
||||
* table, decodes to `undefined` rather than an empty string, so a caller can
|
||||
* tell "no such string" from "empty string".
|
||||
* Implements the verified structure of Diablo II `.tbl` files.
|
||||
*/
|
||||
|
||||
/** Bytes of the fixed header: crc + entry count. */
|
||||
const HEADER_BYTES = 4
|
||||
/** Bytes per index-table entry. */
|
||||
const INDEX_BYTES = 2
|
||||
/** Bytes of one string's length prefix. */
|
||||
const LENGTH_BYTES = 2
|
||||
/** Bytes per UTF-16 code unit. */
|
||||
const CODE_UNIT_BYTES = 2
|
||||
|
||||
/** Raised when a TBL file is malformed. */
|
||||
export class TblError extends Error {
|
||||
constructor(message: string) {
|
||||
super(message)
|
||||
|
|
@ -41,70 +11,188 @@ export class TblError extends Error {
|
|||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Read a little-endian uint16.
|
||||
*
|
||||
* @param data - the buffer.
|
||||
* @param at - byte offset.
|
||||
* @returns the value.
|
||||
*/
|
||||
export function tblHash(key: Uint8Array, hashTableSize: number): number {
|
||||
if (hashTableSize === 0) return 0
|
||||
let h = 0
|
||||
for (let i = 0; i < key.length; i++) {
|
||||
h = ((h << 4) + key[i]!) >>> 0
|
||||
const high = h & 0xf0000000
|
||||
if (high !== 0) h ^= high >>> 24
|
||||
h = (h & ~high) >>> 0
|
||||
}
|
||||
return h % hashTableSize
|
||||
}
|
||||
|
||||
function u16(data: Uint8Array, at: number): number {
|
||||
return data[at]! | (data[at + 1]! << 8)
|
||||
}
|
||||
|
||||
/**
|
||||
* Decode a TBL file.
|
||||
*
|
||||
* @param data - the complete file.
|
||||
* @returns one entry per string index; `undefined` where the index is unused.
|
||||
*/
|
||||
export function decodeTbl(data: Uint8Array): (string | undefined)[] {
|
||||
if (data.byteLength < HEADER_BYTES) {
|
||||
throw new TblError(`TBL is ${String(data.byteLength)} bytes, too short for a header`)
|
||||
}
|
||||
const entryCount = u16(data, 2)
|
||||
const indexEnd = HEADER_BYTES + entryCount * INDEX_BYTES
|
||||
if (indexEnd > data.byteLength) {
|
||||
throw new TblError(`TBL declares ${String(entryCount)} entries but the index table runs past the file`)
|
||||
}
|
||||
const decoder = new TextDecoder('utf-16le')
|
||||
const entries: (string | undefined)[] = []
|
||||
for (let index = 0; index < entryCount; index += 1) {
|
||||
const offset = u16(data, HEADER_BYTES + index * INDEX_BYTES)
|
||||
// A zero offset marks an unused index; so does one that cannot hold a
|
||||
// length prefix. Both are normal in shipped tables.
|
||||
if (offset === 0 || offset + LENGTH_BYTES > data.byteLength) {
|
||||
entries.push(undefined)
|
||||
continue
|
||||
}
|
||||
const characters = u16(data, offset)
|
||||
const from = offset + LENGTH_BYTES
|
||||
const to = from + characters * CODE_UNIT_BYTES
|
||||
if (to > data.byteLength) {
|
||||
throw new TblError(`entry ${String(index)} at ${String(offset)} declares ${String(characters)} characters past the file end`)
|
||||
}
|
||||
// The stored count is characters, not bytes; decoding from a view keeps the
|
||||
// conversion (including surrogate pairs) in the platform's hands.
|
||||
entries.push(decoder.decode(data.subarray(from, to)))
|
||||
}
|
||||
return entries
|
||||
function u32(data: Uint8Array, at: number): number {
|
||||
return (data[at]! | (data[at + 1]! << 8) | (data[at + 2]! << 16) | (data[at + 3]! << 24)) >>> 0
|
||||
}
|
||||
|
||||
function readCStr(data: Uint8Array, start: number): Uint8Array {
|
||||
let end = start
|
||||
while (end < data.byteLength && data[end] !== 0) {
|
||||
end++
|
||||
}
|
||||
return data.subarray(start, end)
|
||||
}
|
||||
|
||||
export type DecodedTbl = (string | undefined)[] & {
|
||||
dict: Map<string, string>
|
||||
}
|
||||
|
||||
const decoderCache = new Map<string, TextDecoder>()
|
||||
function getDecoder(encoding: string): TextDecoder {
|
||||
if (decoderCache.has(encoding)) return decoderCache.get(encoding)!
|
||||
try {
|
||||
const dec = new TextDecoder(encoding)
|
||||
decoderCache.set(encoding, dec)
|
||||
return dec
|
||||
} catch {
|
||||
const fallback = new TextDecoder('latin1')
|
||||
decoderCache.set(encoding, fallback)
|
||||
return fallback
|
||||
}
|
||||
}
|
||||
|
||||
const keyDecoder = new TextDecoder('latin1')
|
||||
|
||||
export function decodeTbl(data: Uint8Array, encoding = 'windows-1252'): DecodedTbl {
|
||||
if (data.byteLength < 21) {
|
||||
throw new TblError(`TBL is ${data.byteLength} bytes, too short for a 21-byte header`)
|
||||
}
|
||||
|
||||
const crc = u16(data, 0)
|
||||
const numElements = u16(data, 2)
|
||||
const hashTableSize = u32(data, 4)
|
||||
const version = data[8]
|
||||
const stringOffset = u32(data, 9)
|
||||
const maxTries = u32(data, 13)
|
||||
const fileSize = u32(data, 17)
|
||||
|
||||
const nodesStart = 21 + numElements * 2
|
||||
const nodesEnd = nodesStart + hashTableSize * 17
|
||||
|
||||
if (nodesEnd > stringOffset) {
|
||||
throw new TblError(`header arithmetic: 21 + ${numElements}*2 + ${hashTableSize}*17 = ${nodesEnd} which exceeds stringOffset ${stringOffset}`)
|
||||
}
|
||||
if (stringOffset > fileSize) {
|
||||
throw new TblError(`stringOffset ${stringOffset} exceeds fileSize ${fileSize}`)
|
||||
}
|
||||
if (fileSize !== data.byteLength) {
|
||||
throw new TblError(`fileSize ${fileSize} does not match actual buffer length ${data.byteLength}`)
|
||||
}
|
||||
|
||||
const valDecoder = getDecoder(encoding)
|
||||
|
||||
const out = [] as unknown as DecodedTbl
|
||||
out.dict = new Map<string, string>()
|
||||
|
||||
for (let i = 0; i < hashTableSize; i++) {
|
||||
const base = nodesStart + i * 17
|
||||
const isActive = data[base]
|
||||
if (!isActive) continue
|
||||
|
||||
const index = u16(data, base + 1)
|
||||
const hashValue = u32(data, base + 3)
|
||||
const keyOffset = u32(data, base + 7)
|
||||
const valOffset = u32(data, base + 11)
|
||||
const valLen = u16(data, base + 15)
|
||||
|
||||
if (keyOffset >= data.byteLength) {
|
||||
throw new TblError(`node ${i} keyOffset ${keyOffset} outside buffer`)
|
||||
}
|
||||
if (valOffset >= data.byteLength) {
|
||||
throw new TblError(`node ${i} valOffset ${valOffset} outside buffer`)
|
||||
}
|
||||
|
||||
const keySlice = readCStr(data, keyOffset)
|
||||
const key = keyDecoder.decode(keySlice)
|
||||
|
||||
const readLen = Math.max(0, valLen - 1)
|
||||
const valSlice = data.subarray(valOffset, valOffset + readLen)
|
||||
const value = valDecoder.decode(valSlice)
|
||||
|
||||
out[index] = value
|
||||
if (!out.dict.has(key)) {
|
||||
out.dict.set(key, value)
|
||||
}
|
||||
}
|
||||
|
||||
return out
|
||||
}
|
||||
|
||||
export function lookupTblFast(data: Uint8Array, key: string, encoding = 'windows-1252'): string | undefined {
|
||||
if (data.byteLength < 21) return undefined
|
||||
const numElements = u16(data, 2)
|
||||
const hashTableSize = u32(data, 4)
|
||||
if (hashTableSize === 0) return undefined
|
||||
const stringOffset = u32(data, 9)
|
||||
const maxTries = u32(data, 13)
|
||||
|
||||
const nodesStart = 21 + numElements * 2
|
||||
const nodesEnd = nodesStart + hashTableSize * 17
|
||||
if (nodesEnd > stringOffset || stringOffset > data.byteLength) return undefined
|
||||
|
||||
const keyEncoder = new TextEncoder()
|
||||
let keyBuf: Uint8Array
|
||||
let isAscii = true
|
||||
for (let i = 0; i < key.length; i++) {
|
||||
if (key.charCodeAt(i) > 0x7F) {
|
||||
isAscii = false
|
||||
break
|
||||
}
|
||||
}
|
||||
if (isAscii) {
|
||||
keyBuf = keyEncoder.encode(key)
|
||||
} else {
|
||||
keyBuf = new Uint8Array(key.length)
|
||||
for (let i = 0; i < key.length; i++) {
|
||||
keyBuf[i] = key.charCodeAt(i) & 0xFF
|
||||
}
|
||||
}
|
||||
|
||||
const targetHash = tblHash(keyBuf, hashTableSize)
|
||||
const valDecoder = getDecoder(encoding)
|
||||
|
||||
for (let i = 0; i < maxTries; i++) {
|
||||
const bucket = (targetHash + i) % hashTableSize
|
||||
const base = nodesStart + bucket * 17
|
||||
const isActive = data[base]
|
||||
if (!isActive) return undefined
|
||||
|
||||
const nodeHash = u32(data, base + 3)
|
||||
if (nodeHash !== targetHash) continue
|
||||
|
||||
const keyOffset = u32(data, base + 7)
|
||||
if (keyOffset >= data.byteLength) continue
|
||||
|
||||
const keySlice = readCStr(data, keyOffset)
|
||||
if (keySlice.byteLength !== keyBuf.byteLength) continue
|
||||
|
||||
let match = true
|
||||
for (let j = 0; j < keyBuf.byteLength; j++) {
|
||||
if (keySlice[j] !== keyBuf[j]) {
|
||||
match = false
|
||||
break
|
||||
}
|
||||
}
|
||||
|
||||
if (match) {
|
||||
const valOffset = u32(data, base + 11)
|
||||
const valLen = u16(data, base + 15)
|
||||
const readLen = Math.max(0, valLen - 1)
|
||||
if (valOffset + readLen > data.byteLength) return undefined
|
||||
return valDecoder.decode(data.subarray(valOffset, valOffset + readLen))
|
||||
}
|
||||
}
|
||||
return undefined
|
||||
}
|
||||
|
||||
/**
|
||||
* Build an index → string lookup over a decoded table.
|
||||
*
|
||||
* @param entries - a decoded table.
|
||||
* @returns the lookup.
|
||||
*/
|
||||
export function tblLookup(entries: readonly (string | undefined)[]): {
|
||||
/** Number of usable entries. */
|
||||
readonly size: number
|
||||
/**
|
||||
* Read one entry.
|
||||
*
|
||||
* @param index - the string index.
|
||||
* @returns the string, or undefined when the index is unused.
|
||||
*/
|
||||
get: (index: number) => string | undefined
|
||||
} {
|
||||
let size = 0
|
||||
|
|
|
|||
|
|
@ -1,3 +1,5 @@
|
|||
// TODO: tbl.ts now decodes the real hash-bucket format. This hardcoded table can be retired once the Chinese MPQ assets (CHI/string.tbl) are available. (fixes #8)
|
||||
|
||||
// level-names-zh.ts — 关卡/场景的中文显示名(用于页面上的三级选择器)
|
||||
//
|
||||
// 出处说明(重要,避免误认为官方本地化):
|
||||
|
|
|
|||
|
|
@ -0,0 +1,148 @@
|
|||
import { describe, test, expect } from 'vitest'
|
||||
import { readFileSync } from 'node:fs'
|
||||
import { join } from 'node:path'
|
||||
import { decodeTbl, TblError, tblHash, lookupTblFast } from '../src/formats/tbl.ts'
|
||||
import { encodeTbl } from '../scripts/lib/tbl-writer.ts'
|
||||
import type { TblEntry } from '../scripts/lib/tbl-writer.ts'
|
||||
|
||||
const fixturePath = join(__dirname, '../samples/fixtures/tbl/patchstring_113c_eng.tbl')
|
||||
const fixtureData = readFileSync(fixturePath)
|
||||
|
||||
describe('TBL Decoder', () => {
|
||||
test('Header parse and Layout invariant', () => {
|
||||
const data = fixtureData
|
||||
const view = new DataView(data.buffer, data.byteOffset, data.byteLength)
|
||||
const crc = view.getUint16(0, true)
|
||||
const numElements = view.getUint16(2, true)
|
||||
const hashTableSize = view.getUint32(4, true)
|
||||
const version = view.getUint8(8)
|
||||
const stringOffset = view.getUint32(9, true)
|
||||
const maxTries = view.getUint32(13, true)
|
||||
const fileSize = view.getUint32(17, true)
|
||||
|
||||
expect(crc).toBe(23974)
|
||||
expect(numElements).toBe(1169)
|
||||
expect(hashTableSize).toBe(1171)
|
||||
expect(version).toBe(1)
|
||||
expect(stringOffset).toBe(22266)
|
||||
expect(maxTries).toBe(1058)
|
||||
expect(fileSize).toBe(52523)
|
||||
|
||||
expect(21 + numElements * 2 + hashTableSize * 17).toBe(stringOffset)
|
||||
expect(fileSize).toBe(data.byteLength)
|
||||
})
|
||||
|
||||
test('Entry count and dictionary', () => {
|
||||
const decoded = decodeTbl(fixtureData)
|
||||
let count = 0; decoded.forEach(v => { if (v !== undefined) count++ }); expect(count).toBe(1169)
|
||||
})
|
||||
|
||||
test('Known strings', () => {
|
||||
const decoded = decodeTbl(fixtureData)
|
||||
expect(decoded.dict.get('Party1X')).toBe('%s permits you to loot his corpse.')
|
||||
expect(decoded.dict.get('D2bnetHelp16b')).toBe('/unignore <*accountname>')
|
||||
expect(decoded.dict.get('skillxld66')).toBe(' ')
|
||||
})
|
||||
|
||||
test('Fast lookup matches full decode', () => {
|
||||
expect(lookupTblFast(fixtureData, 'Party1X')).toBe('%s permits you to loot his corpse.')
|
||||
expect(lookupTblFast(fixtureData, 'D2bnetHelp16b')).toBe('/unignore <*accountname>')
|
||||
expect(lookupTblFast(fixtureData, 'skillxld66')).toBe(' ')
|
||||
})
|
||||
|
||||
test('Hash self-check', () => {
|
||||
const data = fixtureData
|
||||
const view = new DataView(data.buffer, data.byteOffset, data.byteLength)
|
||||
const numElements = view.getUint16(2, true)
|
||||
const hashTableSize = view.getUint32(4, true)
|
||||
|
||||
const nodesStart = 21 + numElements * 2
|
||||
let checkedCount = 0
|
||||
|
||||
for (let i = 0; i < hashTableSize; i++) {
|
||||
const base = nodesStart + i * 17
|
||||
const isActive = view.getUint8(base)
|
||||
if (!isActive) continue
|
||||
|
||||
const hashValue = view.getUint32(base + 3, true)
|
||||
const keyOffset = view.getUint32(base + 7, true)
|
||||
|
||||
let end = keyOffset
|
||||
while (data[end] !== 0) end++
|
||||
const keyBuf = data.subarray(keyOffset, end)
|
||||
|
||||
const computed = tblHash(keyBuf, hashTableSize)
|
||||
expect(computed).toBe(hashValue)
|
||||
|
||||
checkedCount++
|
||||
}
|
||||
|
||||
expect(checkedCount).toBe(1169)
|
||||
})
|
||||
|
||||
test('Round-trip encoding', () => {
|
||||
const entries: TblEntry[] = [
|
||||
{ key: "Stamina Potion", value: "Stamina Potion" },
|
||||
{ key: "Empty", value: "" },
|
||||
null,
|
||||
{ key: "Tome", value: "Tome of Town Portal" },
|
||||
null,
|
||||
{ key: "CJK", value: "完美宝石" },
|
||||
{ key: "Latin", value: "Straße" },
|
||||
null,
|
||||
{ key: "Emoji", value: "🌀 Emoji beyond the BMP" },
|
||||
{ key: "Long", value: 'x'.repeat(300) },
|
||||
]
|
||||
|
||||
// Add enough entities to force collisions
|
||||
for (let i = 0; i < 200; i++) {
|
||||
entries.push({ key: `CollC${i}`, value: `ValC${i}` })
|
||||
}
|
||||
|
||||
// Duplicate-ish keys differing by case
|
||||
entries.push({ key: "Apple", value: "Uppercase" })
|
||||
entries.push({ key: "apple", value: "Lowercase" })
|
||||
|
||||
const encoded = encodeTbl(entries)
|
||||
const decoded = decodeTbl(encoded, 'utf-8')
|
||||
|
||||
expect(decoded.dict.get("Empty")).toBe("")
|
||||
expect(decoded.dict.get("Latin")).toBe("Straße")
|
||||
expect(decoded.dict.get("CJK")).toBe("完美宝石")
|
||||
expect(decoded.dict.get("Apple")).toBe("Uppercase")
|
||||
expect(decoded.dict.get("apple")).toBe("Lowercase")
|
||||
expect(decoded.dict.get("CollC199")).toBe("ValC199")
|
||||
|
||||
// Also test fast lookup
|
||||
expect(lookupTblFast(encoded, "Latin", "utf-8")).toBe("Straße")
|
||||
})
|
||||
|
||||
test('Malformed input: truncated buffer', () => {
|
||||
const shortBuf = new Uint8Array(20)
|
||||
expect(() => decodeTbl(shortBuf)).toThrowError(TblError)
|
||||
expect(() => decodeTbl(shortBuf)).toThrowError(/too short/)
|
||||
})
|
||||
|
||||
test('Malformed input: stringOffset past EOF', () => {
|
||||
const buf = new Uint8Array(fixtureData)
|
||||
const view = new DataView(buf.buffer, buf.byteOffset, buf.byteLength)
|
||||
// Make stringOffset larger than file length
|
||||
view.setUint32(9, buf.length + 10, true)
|
||||
|
||||
expect(() => decodeTbl(buf)).toThrowError(TblError)
|
||||
expect(() => decodeTbl(buf)).toThrowError(/stringOffset/)
|
||||
})
|
||||
|
||||
test('Malformed input: node offset out of range', () => {
|
||||
const buf = new Uint8Array(fixtureData)
|
||||
const view = new DataView(buf.buffer, buf.byteOffset, buf.byteLength)
|
||||
|
||||
// Get the first active node and corrupt its key offset
|
||||
const numElements = view.getUint16(2, true)
|
||||
const nodesStart = 21 + numElements * 2
|
||||
view.setUint32(nodesStart + 7, buf.length + 5, true) // key offset is at +7
|
||||
|
||||
expect(() => decodeTbl(buf)).toThrowError(TblError)
|
||||
expect(() => decodeTbl(buf)).toThrowError(/keyOffset/)
|
||||
})
|
||||
})
|
||||
Loading…
Reference in New Issue