fix(formats): 按官方哈希桶格式重写 tbl 解码器与编码器 (fixes #8)

- 重构了 `src/formats/tbl.ts` 以正确读取 21字节魔数头与 17字节哈希桶节点
- 使用了 NUL 终止与正确的字节长度计算方式,修正之前错误的 UTF-16 假设
- 支持通过 `TextDecoder` 选择代码页 (默认为 windows-1252,支持 gbk 等 fallback)
- 暴露了原始的 O(1) 线性探测查询函数 `lookupTblFast` 和哈希函数 `tblHash`
- 对 `scripts/lib/tbl-writer.ts` 及其验证脚本进行了重写,以正确写入和校验哈希桶布局
- 增加了完整涵盖功能与越界校验的 vitest 单元测试,并在真实的 113c patchstring.tbl 下实现了 1169/1169 哈希全量验证通过
- 将测试文件加入项目目录
This commit is contained in:
troytt 2026-09-14 11:56:23 +00:00
parent 1a81693d74
commit f4dd55be6c
6 changed files with 511 additions and 172 deletions

Binary file not shown.

View File

@ -1,59 +1,141 @@
/** import { tblHash } from '../../src/formats/tbl.ts'
* Minimal `.tbl` writer, shared by the fixture generator and its checker.
*
* Tooling only: the engine reads TBL files, it never writes them. It lives here so
* the fixture generator and the construction check encode a table the same way,
* instead of two copies drifting apart.
*/
/** One entry: a string, or null for an unused index. */ export type TblEntry = string | null | { key: string, value: string }
export type TblEntry = string | null
/**
* Encode a classic `.tbl`: crc, entry count, index table, then
* `u16 characterCount` + UTF-16LE strings.
*
* @param entries - the entries, in index order.
* @returns the encoded bytes.
*/
export function encodeTbl(entries: readonly TblEntry[]): Uint8Array { export function encodeTbl(entries: readonly TblEntry[]): Uint8Array {
const headerBytes = 4 // Determine numElements (max index + 1 of non-null entries)
const indexBytes = entries.length * 2 let numElements = entries.length
const blocks: { index: number; bytes: Uint8Array }[] = []
let cursor = headerBytes + indexBytes // Decide hashTableSize. Needs to be somewhat larger than active entries.
for (const [index, entry] of entries.entries()) { // We'll just pick a prime roughly 1.5x larger, or just use length * 2 + 1
if (entry === null) continue const activeEntries = entries.filter((e) => e !== null)
// The count is in characters; an astral character is two UTF-16 units. const hashTableSize = Math.max(17, activeEntries.length * 2 + 1)
let characters = 0
for (const character of entry) characters += character.codePointAt(0)! > 0xffff ? 2 : 1 const encoder = new TextEncoder()
const body = new Uint8Array(characters * 2) const keyEncoder = new TextEncoder()
const view = new DataView(body.buffer)
let units = 0 // We need to build the hash table
for (const character of entry) { // Each node: 17 bytes
const code = character.codePointAt(0)! // Node layout: isActive(u8), index(u16), hashValue(u32), keyOffset(u32), valOffset(u32), valLength(u16)
if (code > 0xffff) {
const adjusted = code - 0x10000 const nodes = new Uint8Array(hashTableSize * 17)
view.setUint16(units * 2, 0xd800 + (adjusted >> 10), true) const nodeView = new DataView(nodes.buffer)
view.setUint16((units + 1) * 2, 0xdc00 + (adjusted & 0x3ff), true)
units += 2 let stringDataSize = 0
type EntryTuple = { index: number, keyStr: string, keyBytes: Uint8Array, valBytes: Uint8Array, hash: number, targetBucket: number }
const tuples: EntryTuple[] = []
for (let i = 0; i < entries.length; i++) {
const e = entries[i]
if (e === null) continue
let keyStr = ""
let valStr = ""
if (typeof e === 'string') {
keyStr = `__auto_key_${i}`
valStr = e
} else { } else {
view.setUint16(units * 2, code, true) keyStr = e.key
units += 1 valStr = e.value
}
// Fallback encode to latin1 if it's ascii for keys
let keyBytes = keyEncoder.encode(keyStr)
let valBytes = encoder.encode(valStr) // for simplify, we use utf-8
const h = tblHash(keyBytes, hashTableSize)
tuples.push({ index: i, keyStr: keyStr, keyBytes, valBytes, hash: h, targetBucket: h })
}
let maxTries = 0
for (const t of tuples) {
let placed = false
let tries = 0
for (; tries < hashTableSize; tries++) {
const b = (t.targetBucket + tries) % hashTableSize
const base = b * 17
if (nodes[base] === 0) {
// Place it
nodes[base] = 1 // isActive
nodeView.setUint16(base + 1, t.index, true)
nodeView.setUint32(base + 3, t.targetBucket, true) // Wait, is hashValue the bucket index?
// "hashValue stores the ALREADY-MODULO'D bucket index, i.e. pjw(key) % hashTableSize"
placed = true
maxTries = Math.max(maxTries, tries + 1)
break
} }
} }
const bytes = new Uint8Array([characters & 0xff, (characters >> 8) & 0xff, ...body]) if (!placed) throw new Error("Hash table full")
blocks.push({ index, bytes })
cursor += bytes.byteLength
} }
const out = new Uint8Array(cursor)
// Now place string data
let stringOffset = 21 + entries.length * 2 + hashTableSize * 17
const strings: Uint8Array[] = []
let currentStringOffset = stringOffset
for (let b = 0; b < hashTableSize; b++) {
const base = b * 17
if (nodes[base] === 0) continue
const index = nodeView.getUint16(base + 1, true)
const t = tuples.find(x => x.index === index)!
// key is NUL terminated
nodeView.setUint32(base + 7, currentStringOffset, true)
const keyBuf = new Uint8Array(t.keyBytes.length + 1)
keyBuf.set(t.keyBytes)
strings.push(keyBuf)
currentStringOffset += keyBuf.length
// value
nodeView.setUint32(base + 11, currentStringOffset, true)
// valLength includes NUL
nodeView.setUint16(base + 15, t.valBytes.length + 1, true)
const valBuf = new Uint8Array(t.valBytes.length + 1)
valBuf.set(t.valBytes)
strings.push(valBuf)
currentStringOffset += valBuf.length
}
const out = new Uint8Array(currentStringOffset)
const view = new DataView(out.buffer) const view = new DataView(out.buffer)
// header
view.setUint16(0, 0x1234, true) view.setUint16(0, 0x1234, true)
view.setUint16(2, entries.length, true) view.setUint16(2, entries.length, true)
let at = headerBytes + indexBytes view.setUint32(4, hashTableSize, true)
for (const block of blocks) { view.setUint8(8, 1) // version
view.setUint16(headerBytes + block.index * 2, at, true) view.setUint32(9, stringOffset, true)
out.set(block.bytes, at) view.setUint32(13, maxTries, true)
at += block.bytes.byteLength view.setUint32(17, currentStringOffset, true)
// indices
for (let i = 0; i < entries.length; i++) {
// wait! The index array in real files, what does it map?
// It maps string index -> bucket index.
// If an index is not used, it typically maps to 0 or something? Or is it an array of bucket indices?
// Let's lookup what bucket index it ended up in.
let bucket = 0
if (entries[i] !== null) {
// find bucket
for (let b = 0; b < hashTableSize; b++) {
if (nodes[b * 17] && nodeView.getUint16(b * 17 + 1, true) === i) {
bucket = b
break
} }
}
}
view.setUint16(21 + i * 2, bucket, true)
}
out.set(nodes, 21 + entries.length * 2)
let copied = stringOffset
for (const s of strings) {
out.set(s, copied)
copied += s.length
}
return out return out
} }

View File

@ -1,59 +1,78 @@
/** /**
* Construction check for the TBL decoder. * Construction check and real-file check for the TBL decoder.
* *
* The classic TBL layout has no independent decoder available (the community Go * Requirements:
* package implements the later hash-table variant), so this is a two-sided test * - Round-trip encoding through tbl-writer
* rather than a differential one: the script *writes* a table whose entries are * - Success on actual Diablo II patchstring.tbl
* known — including empty strings, unused indices and non-ASCII text — and then
* requires the decoder to reproduce exactly that. It catches layout and
* decoding mistakes (off-by-one offsets, byte/character confusion, UTF-16
* endianness) but cannot catch a wrong reading shared by writer and reader;
* that is what a real `string.tbl` is for.
*
* Usage: node scripts/verify-tbl.ts
*/ */
import { decodeTbl, tblLookup } from '../src/formats/tbl.ts' import { readFileSync } from 'node:fs'
import { join } from 'node:path'
import { decodeTbl, tblLookup, tblHash } from '../src/formats/tbl.ts'
import { encodeTbl } from './lib/tbl-writer.ts' import { encodeTbl } from './lib/tbl-writer.ts'
import type { TblEntry } from './lib/tbl-writer.ts' import type { TblEntry } from './lib/tbl-writer.ts'
/** Entries the decoder must reproduce, in order. */
const entries: TblEntry[] = [ const entries: TblEntry[] = [
'Stamina Potion', // plain ASCII { key: "Stamina Potion", value: "Stamina Potion" },
'', // empty string is a value, not a hole { key: "Empty", value: "" },
null, // unused index
'Tome of Town Portal',
null, null,
'完美宝石', // CJK: multi-byte UTF-16, but one unit each { key: "Tome", value: "Tome of Town Portal" },
'Straße', // Latin-1 range
null, null,
'🌀 Emoji beyond the BMP', // surrogate pair: two units for one character { key: "CJK", value: "完美宝石" },
'x'.repeat(300), // long entry { key: "Latin", value: "Straße" },
null,
{ key: "Emoji", value: "🌀 Emoji beyond the BMP" },
{ key: "Long", value: 'x'.repeat(300) },
] ]
const encoded = encodeTbl(entries) const encoded = encodeTbl(entries)
const decoded = decodeTbl(encoded) // We pass 'utf-8' here because our tbl-writer encodes values in utf-8 to preserve things for testing
const decoded = decodeTbl(encoded, 'utf-8')
const lookup = tblLookup(decoded) const lookup = tblLookup(decoded)
const problems: string[] = [] const problems: string[] = []
if (encoded.byteLength !== decoded.length * 0 + encoded.byteLength) { if (decoded.length !== entries.length) problems.push(`entry count ${decoded.length} != ${entries.length}`)
/* v8 ignore next -- placeholder to keep the shape obvious. */
problems.push('unreachable')
}
if (decoded.length !== entries.length) problems.push(`entry count ${String(decoded.length)} != ${String(entries.length)}`)
entries.forEach((want, index) => { entries.forEach((want, index) => {
const got = decoded[index] const got = decoded[index]
if (want === null) { if (want === null) {
if (got !== undefined) problems.push(`index ${String(index)}: expected an unused index, got ${JSON.stringify(got)}`) if (got !== undefined) problems.push(`index ${index}: expected an unused index, got ${JSON.stringify(got)}`)
return return
} }
if (got !== want) problems.push(`index ${String(index)}: ${JSON.stringify(got)} != ${JSON.stringify(want)}`) const expectedValue = typeof want === 'string' ? want : want.value
if (got !== expectedValue) problems.push(`index ${index}: ${JSON.stringify(got)} != ${JSON.stringify(expectedValue)}`)
const expectedKey = typeof want === 'string' ? `__auto_key_${index}` : want.key
if (decoded.dict.get(expectedKey) !== expectedValue) {
problems.push(`Dictionary lookup failed for key ${expectedKey}`)
}
}) })
console.log(`table ${String(entries.length)} indices (${String(lookup.size)} used)`) console.log(`[Roundtrip] table ${entries.length} indices (${lookup.size} used)`)
console.log(`encoded ${String(encoded.byteLength)} bytes`) console.log(`[Roundtrip] encoded ${encoded.byteLength} bytes`)
console.log(`unicode CJK=${JSON.stringify(lookup.get(5))} latin1=${JSON.stringify(lookup.get(6))} astral=${JSON.stringify(lookup.get(8))}`) console.log(`[Roundtrip] unicode CJK=${JSON.stringify(lookup.get(5))} latin1=${JSON.stringify(lookup.get(6))} astral=${JSON.stringify(lookup.get(8))}`)
console.log(`long ${String(lookup.get(9)?.length ?? 0)} characters round-tripped`) console.log(`[Roundtrip] long ${lookup.get(9)?.length ?? 0} characters round-tripped`)
console.log(`problems ${String(problems.length)}`) console.log(`[Roundtrip] problems ${problems.length}`)
let fixtureProblems = 0
try {
const raw = readFileSync(join(import.meta.dirname, '../samples/fixtures/tbl/patchstring_113c_eng.tbl'))
const fixtureDecoded = decodeTbl(raw) // defaults to windows-1252
let hashFailures = 0
let matched = 0
for (const [key, value] of fixtureDecoded.dict.entries()) {
matched++
// To verify, we would need the hashtablesize. We can't access it natively through just decodeTbl.
// Instead we can just check if we can retrieve known values:
}
if (fixtureDecoded.dict.get("Party1X") !== "%s permits you to loot his corpse.") fixtureProblems++
if (fixtureDecoded.dict.get("D2bnetHelp16b") !== "/unignore <*accountname>") fixtureProblems++
console.log(`[Fixture] Loaded fixture correctly. Total active keys: ${matched}`)
} catch (e: any) {
problems.push(`Fixture failed: ${e.message}`)
fixtureProblems++
}
for (const problem of problems.slice(0, 8)) console.log(` - ${problem}`) for (const problem of problems.slice(0, 8)) console.log(` - ${problem}`)
console.log(problems.length === 0 ? 'RESULT every index round-trips exactly' : 'RESULT FAILED') const success = problems.length === 0 && fixtureProblems === 0
process.exit(problems.length === 0 ? 0 : 1) console.log(success ? 'RESULT all passed' : 'RESULT FAILED')
process.exit(success ? 0 : 1)

View File

@ -1,39 +1,9 @@
/** /**
* Diablo II `.tbl` string-table decoder (classic layout). * Diablo II `.tbl` string-table decoder (hash-bucket layout).
* *
* A TBL is an *indexed* string table — item, skill, monster and quest names are * Implements the verified structure of Diablo II `.tbl` files.
* referred to by number, not by key:
*
* ```
* u16 crc (unused here; the game uses it as a content checksum)
* u16 entryCount
* u16 offsets[entryCount] absolute file offsets, one per index
* ... strings: u16 characterCount, then characterCount * 2 bytes UTF-16LE
* ```
*
* The classic Diablo II files (`string.tbl`, `expansionstring.tbl`,
* `patchstring.tbl`) use exactly this layout. A later "extended" variant adds a
* hash table of name/value pairs on top; the community Go package implements
* that variant instead, so unlike the other formats in this directory **there is
* no independent decoder to diff this one against** — it is checked by
* construction (write a table, read it back, including non-ASCII) and is on the
* list to confirm against a real `string.tbl`.
*
* Unused indices are legal: an offset that is zero, or that points past the
* table, decodes to `undefined` rather than an empty string, so a caller can
* tell "no such string" from "empty string".
*/ */
/** Bytes of the fixed header: crc + entry count. */
const HEADER_BYTES = 4
/** Bytes per index-table entry. */
const INDEX_BYTES = 2
/** Bytes of one string's length prefix. */
const LENGTH_BYTES = 2
/** Bytes per UTF-16 code unit. */
const CODE_UNIT_BYTES = 2
/** Raised when a TBL file is malformed. */
export class TblError extends Error { export class TblError extends Error {
constructor(message: string) { constructor(message: string) {
super(message) super(message)
@ -41,70 +11,188 @@ export class TblError extends Error {
} }
} }
/** export function tblHash(key: Uint8Array, hashTableSize: number): number {
* Read a little-endian uint16. if (hashTableSize === 0) return 0
* let h = 0
* @param data - the buffer. for (let i = 0; i < key.length; i++) {
* @param at - byte offset. h = ((h << 4) + key[i]!) >>> 0
* @returns the value. const high = h & 0xf0000000
*/ if (high !== 0) h ^= high >>> 24
h = (h & ~high) >>> 0
}
return h % hashTableSize
}
function u16(data: Uint8Array, at: number): number { function u16(data: Uint8Array, at: number): number {
return data[at]! | (data[at + 1]! << 8) return data[at]! | (data[at + 1]! << 8)
} }
/** function u32(data: Uint8Array, at: number): number {
* Decode a TBL file. return (data[at]! | (data[at + 1]! << 8) | (data[at + 2]! << 16) | (data[at + 3]! << 24)) >>> 0
* }
* @param data - the complete file.
* @returns one entry per string index; `undefined` where the index is unused. function readCStr(data: Uint8Array, start: number): Uint8Array {
*/ let end = start
export function decodeTbl(data: Uint8Array): (string | undefined)[] { while (end < data.byteLength && data[end] !== 0) {
if (data.byteLength < HEADER_BYTES) { end++
throw new TblError(`TBL is ${String(data.byteLength)} bytes, too short for a header`) }
} return data.subarray(start, end)
const entryCount = u16(data, 2) }
const indexEnd = HEADER_BYTES + entryCount * INDEX_BYTES
if (indexEnd > data.byteLength) { export type DecodedTbl = (string | undefined)[] & {
throw new TblError(`TBL declares ${String(entryCount)} entries but the index table runs past the file`) dict: Map<string, string>
} }
const decoder = new TextDecoder('utf-16le')
const entries: (string | undefined)[] = [] const decoderCache = new Map<string, TextDecoder>()
for (let index = 0; index < entryCount; index += 1) { function getDecoder(encoding: string): TextDecoder {
const offset = u16(data, HEADER_BYTES + index * INDEX_BYTES) if (decoderCache.has(encoding)) return decoderCache.get(encoding)!
// A zero offset marks an unused index; so does one that cannot hold a try {
// length prefix. Both are normal in shipped tables. const dec = new TextDecoder(encoding)
if (offset === 0 || offset + LENGTH_BYTES > data.byteLength) { decoderCache.set(encoding, dec)
entries.push(undefined) return dec
continue } catch {
} const fallback = new TextDecoder('latin1')
const characters = u16(data, offset) decoderCache.set(encoding, fallback)
const from = offset + LENGTH_BYTES return fallback
const to = from + characters * CODE_UNIT_BYTES }
if (to > data.byteLength) { }
throw new TblError(`entry ${String(index)} at ${String(offset)} declares ${String(characters)} characters past the file end`)
} const keyDecoder = new TextDecoder('latin1')
// The stored count is characters, not bytes; decoding from a view keeps the
// conversion (including surrogate pairs) in the platform's hands. export function decodeTbl(data: Uint8Array, encoding = 'windows-1252'): DecodedTbl {
entries.push(decoder.decode(data.subarray(from, to))) if (data.byteLength < 21) {
} throw new TblError(`TBL is ${data.byteLength} bytes, too short for a 21-byte header`)
return entries }
const crc = u16(data, 0)
const numElements = u16(data, 2)
const hashTableSize = u32(data, 4)
const version = data[8]
const stringOffset = u32(data, 9)
const maxTries = u32(data, 13)
const fileSize = u32(data, 17)
const nodesStart = 21 + numElements * 2
const nodesEnd = nodesStart + hashTableSize * 17
if (nodesEnd > stringOffset) {
throw new TblError(`header arithmetic: 21 + ${numElements}*2 + ${hashTableSize}*17 = ${nodesEnd} which exceeds stringOffset ${stringOffset}`)
}
if (stringOffset > fileSize) {
throw new TblError(`stringOffset ${stringOffset} exceeds fileSize ${fileSize}`)
}
if (fileSize !== data.byteLength) {
throw new TblError(`fileSize ${fileSize} does not match actual buffer length ${data.byteLength}`)
}
const valDecoder = getDecoder(encoding)
const out = [] as unknown as DecodedTbl
out.dict = new Map<string, string>()
for (let i = 0; i < hashTableSize; i++) {
const base = nodesStart + i * 17
const isActive = data[base]
if (!isActive) continue
const index = u16(data, base + 1)
const hashValue = u32(data, base + 3)
const keyOffset = u32(data, base + 7)
const valOffset = u32(data, base + 11)
const valLen = u16(data, base + 15)
if (keyOffset >= data.byteLength) {
throw new TblError(`node ${i} keyOffset ${keyOffset} outside buffer`)
}
if (valOffset >= data.byteLength) {
throw new TblError(`node ${i} valOffset ${valOffset} outside buffer`)
}
const keySlice = readCStr(data, keyOffset)
const key = keyDecoder.decode(keySlice)
const readLen = Math.max(0, valLen - 1)
const valSlice = data.subarray(valOffset, valOffset + readLen)
const value = valDecoder.decode(valSlice)
out[index] = value
if (!out.dict.has(key)) {
out.dict.set(key, value)
}
}
return out
}
export function lookupTblFast(data: Uint8Array, key: string, encoding = 'windows-1252'): string | undefined {
if (data.byteLength < 21) return undefined
const numElements = u16(data, 2)
const hashTableSize = u32(data, 4)
if (hashTableSize === 0) return undefined
const stringOffset = u32(data, 9)
const maxTries = u32(data, 13)
const nodesStart = 21 + numElements * 2
const nodesEnd = nodesStart + hashTableSize * 17
if (nodesEnd > stringOffset || stringOffset > data.byteLength) return undefined
const keyEncoder = new TextEncoder()
let keyBuf: Uint8Array
let isAscii = true
for (let i = 0; i < key.length; i++) {
if (key.charCodeAt(i) > 0x7F) {
isAscii = false
break
}
}
if (isAscii) {
keyBuf = keyEncoder.encode(key)
} else {
keyBuf = new Uint8Array(key.length)
for (let i = 0; i < key.length; i++) {
keyBuf[i] = key.charCodeAt(i) & 0xFF
}
}
const targetHash = tblHash(keyBuf, hashTableSize)
const valDecoder = getDecoder(encoding)
for (let i = 0; i < maxTries; i++) {
const bucket = (targetHash + i) % hashTableSize
const base = nodesStart + bucket * 17
const isActive = data[base]
if (!isActive) return undefined
const nodeHash = u32(data, base + 3)
if (nodeHash !== targetHash) continue
const keyOffset = u32(data, base + 7)
if (keyOffset >= data.byteLength) continue
const keySlice = readCStr(data, keyOffset)
if (keySlice.byteLength !== keyBuf.byteLength) continue
let match = true
for (let j = 0; j < keyBuf.byteLength; j++) {
if (keySlice[j] !== keyBuf[j]) {
match = false
break
}
}
if (match) {
const valOffset = u32(data, base + 11)
const valLen = u16(data, base + 15)
const readLen = Math.max(0, valLen - 1)
if (valOffset + readLen > data.byteLength) return undefined
return valDecoder.decode(data.subarray(valOffset, valOffset + readLen))
}
}
return undefined
} }
/**
* Build an index → string lookup over a decoded table.
*
* @param entries - a decoded table.
* @returns the lookup.
*/
export function tblLookup(entries: readonly (string | undefined)[]): { export function tblLookup(entries: readonly (string | undefined)[]): {
/** Number of usable entries. */
readonly size: number readonly size: number
/**
* Read one entry.
*
* @param index - the string index.
* @returns the string, or undefined when the index is unused.
*/
get: (index: number) => string | undefined get: (index: number) => string | undefined
} { } {
let size = 0 let size = 0

View File

@ -1,3 +1,5 @@
// TODO: tbl.ts now decodes the real hash-bucket format. This hardcoded table can be retired once the Chinese MPQ assets (CHI/string.tbl) are available. (fixes #8)
// level-names-zh.ts — 关卡/场景的中文显示名(用于页面上的三级选择器) // level-names-zh.ts — 关卡/场景的中文显示名(用于页面上的三级选择器)
// //
// 出处说明(重要,避免误认为官方本地化): // 出处说明(重要,避免误认为官方本地化):

148
tests/tbl.test.ts Normal file
View File

@ -0,0 +1,148 @@
import { describe, test, expect } from 'vitest'
import { readFileSync } from 'node:fs'
import { join } from 'node:path'
import { decodeTbl, TblError, tblHash, lookupTblFast } from '../src/formats/tbl.ts'
import { encodeTbl } from '../scripts/lib/tbl-writer.ts'
import type { TblEntry } from '../scripts/lib/tbl-writer.ts'
const fixturePath = join(__dirname, '../samples/fixtures/tbl/patchstring_113c_eng.tbl')
const fixtureData = readFileSync(fixturePath)
describe('TBL Decoder', () => {
test('Header parse and Layout invariant', () => {
const data = fixtureData
const view = new DataView(data.buffer, data.byteOffset, data.byteLength)
const crc = view.getUint16(0, true)
const numElements = view.getUint16(2, true)
const hashTableSize = view.getUint32(4, true)
const version = view.getUint8(8)
const stringOffset = view.getUint32(9, true)
const maxTries = view.getUint32(13, true)
const fileSize = view.getUint32(17, true)
expect(crc).toBe(23974)
expect(numElements).toBe(1169)
expect(hashTableSize).toBe(1171)
expect(version).toBe(1)
expect(stringOffset).toBe(22266)
expect(maxTries).toBe(1058)
expect(fileSize).toBe(52523)
expect(21 + numElements * 2 + hashTableSize * 17).toBe(stringOffset)
expect(fileSize).toBe(data.byteLength)
})
test('Entry count and dictionary', () => {
const decoded = decodeTbl(fixtureData)
let count = 0; decoded.forEach(v => { if (v !== undefined) count++ }); expect(count).toBe(1169)
})
test('Known strings', () => {
const decoded = decodeTbl(fixtureData)
expect(decoded.dict.get('Party1X')).toBe('%s permits you to loot his corpse.')
expect(decoded.dict.get('D2bnetHelp16b')).toBe('/unignore <*accountname>')
expect(decoded.dict.get('skillxld66')).toBe(' ')
})
test('Fast lookup matches full decode', () => {
expect(lookupTblFast(fixtureData, 'Party1X')).toBe('%s permits you to loot his corpse.')
expect(lookupTblFast(fixtureData, 'D2bnetHelp16b')).toBe('/unignore <*accountname>')
expect(lookupTblFast(fixtureData, 'skillxld66')).toBe(' ')
})
test('Hash self-check', () => {
const data = fixtureData
const view = new DataView(data.buffer, data.byteOffset, data.byteLength)
const numElements = view.getUint16(2, true)
const hashTableSize = view.getUint32(4, true)
const nodesStart = 21 + numElements * 2
let checkedCount = 0
for (let i = 0; i < hashTableSize; i++) {
const base = nodesStart + i * 17
const isActive = view.getUint8(base)
if (!isActive) continue
const hashValue = view.getUint32(base + 3, true)
const keyOffset = view.getUint32(base + 7, true)
let end = keyOffset
while (data[end] !== 0) end++
const keyBuf = data.subarray(keyOffset, end)
const computed = tblHash(keyBuf, hashTableSize)
expect(computed).toBe(hashValue)
checkedCount++
}
expect(checkedCount).toBe(1169)
})
test('Round-trip encoding', () => {
const entries: TblEntry[] = [
{ key: "Stamina Potion", value: "Stamina Potion" },
{ key: "Empty", value: "" },
null,
{ key: "Tome", value: "Tome of Town Portal" },
null,
{ key: "CJK", value: "完美宝石" },
{ key: "Latin", value: "Straße" },
null,
{ key: "Emoji", value: "🌀 Emoji beyond the BMP" },
{ key: "Long", value: 'x'.repeat(300) },
]
// Add enough entities to force collisions
for (let i = 0; i < 200; i++) {
entries.push({ key: `CollC${i}`, value: `ValC${i}` })
}
// Duplicate-ish keys differing by case
entries.push({ key: "Apple", value: "Uppercase" })
entries.push({ key: "apple", value: "Lowercase" })
const encoded = encodeTbl(entries)
const decoded = decodeTbl(encoded, 'utf-8')
expect(decoded.dict.get("Empty")).toBe("")
expect(decoded.dict.get("Latin")).toBe("Straße")
expect(decoded.dict.get("CJK")).toBe("完美宝石")
expect(decoded.dict.get("Apple")).toBe("Uppercase")
expect(decoded.dict.get("apple")).toBe("Lowercase")
expect(decoded.dict.get("CollC199")).toBe("ValC199")
// Also test fast lookup
expect(lookupTblFast(encoded, "Latin", "utf-8")).toBe("Straße")
})
test('Malformed input: truncated buffer', () => {
const shortBuf = new Uint8Array(20)
expect(() => decodeTbl(shortBuf)).toThrowError(TblError)
expect(() => decodeTbl(shortBuf)).toThrowError(/too short/)
})
test('Malformed input: stringOffset past EOF', () => {
const buf = new Uint8Array(fixtureData)
const view = new DataView(buf.buffer, buf.byteOffset, buf.byteLength)
// Make stringOffset larger than file length
view.setUint32(9, buf.length + 10, true)
expect(() => decodeTbl(buf)).toThrowError(TblError)
expect(() => decodeTbl(buf)).toThrowError(/stringOffset/)
})
test('Malformed input: node offset out of range', () => {
const buf = new Uint8Array(fixtureData)
const view = new DataView(buf.buffer, buf.byteOffset, buf.byteLength)
// Get the first active node and corrupt its key offset
const numElements = view.getUint16(2, true)
const nodesStart = 21 + numElements * 2
view.setUint32(nodesStart + 7, buf.length + 5, true) // key offset is at +7
expect(() => decodeTbl(buf)).toThrowError(TblError)
expect(() => decodeTbl(buf)).toThrowError(/keyOffset/)
})
})