diff --git a/samples/fixtures/tbl/patchstring_113c_eng.tbl b/samples/fixtures/tbl/patchstring_113c_eng.tbl new file mode 100644 index 0000000..3af7a84 Binary files /dev/null and b/samples/fixtures/tbl/patchstring_113c_eng.tbl differ diff --git a/scripts/lib/tbl-writer.ts b/scripts/lib/tbl-writer.ts index 76a8a14..ebfe7f8 100644 --- a/scripts/lib/tbl-writer.ts +++ b/scripts/lib/tbl-writer.ts @@ -1,59 +1,141 @@ -/** - * Minimal `.tbl` writer, shared by the fixture generator and its checker. - * - * Tooling only: the engine reads TBL files, it never writes them. It lives here so - * the fixture generator and the construction check encode a table the same way, - * instead of two copies drifting apart. - */ +import { tblHash } from '../../src/formats/tbl.ts' -/** One entry: a string, or null for an unused index. */ -export type TblEntry = string | null +export type TblEntry = string | null | { key: string, value: string } -/** - * Encode a classic `.tbl`: crc, entry count, index table, then - * `u16 characterCount` + UTF-16LE strings. - * - * @param entries - the entries, in index order. - * @returns the encoded bytes. - */ export function encodeTbl(entries: readonly TblEntry[]): Uint8Array { - const headerBytes = 4 - const indexBytes = entries.length * 2 - const blocks: { index: number; bytes: Uint8Array }[] = [] - let cursor = headerBytes + indexBytes - for (const [index, entry] of entries.entries()) { - if (entry === null) continue - // The count is in characters; an astral character is two UTF-16 units. - let characters = 0 - for (const character of entry) characters += character.codePointAt(0)! > 0xffff ? 2 : 1 - const body = new Uint8Array(characters * 2) - const view = new DataView(body.buffer) - let units = 0 - for (const character of entry) { - const code = character.codePointAt(0)! - if (code > 0xffff) { - const adjusted = code - 0x10000 - view.setUint16(units * 2, 0xd800 + (adjusted >> 10), true) - view.setUint16((units + 1) * 2, 0xdc00 + (adjusted & 0x3ff), true) - units += 2 - } else { - view.setUint16(units * 2, code, true) - units += 1 + // Determine numElements (max index + 1 of non-null entries) + let numElements = entries.length + + // Decide hashTableSize. Needs to be somewhat larger than active entries. + // We'll just pick a prime roughly 1.5x larger, or just use length * 2 + 1 + const activeEntries = entries.filter((e) => e !== null) + const hashTableSize = Math.max(17, activeEntries.length * 2 + 1) + + const encoder = new TextEncoder() + const keyEncoder = new TextEncoder() + + // We need to build the hash table + // Each node: 17 bytes + // Node layout: isActive(u8), index(u16), hashValue(u32), keyOffset(u32), valOffset(u32), valLength(u16) + + const nodes = new Uint8Array(hashTableSize * 17) + const nodeView = new DataView(nodes.buffer) + + let stringDataSize = 0 + + type EntryTuple = { index: number, keyStr: string, keyBytes: Uint8Array, valBytes: Uint8Array, hash: number, targetBucket: number } + const tuples: EntryTuple[] = [] + + for (let i = 0; i < entries.length; i++) { + const e = entries[i] + if (e === null) continue + + let keyStr = "" + let valStr = "" + if (typeof e === 'string') { + keyStr = `__auto_key_${i}` + valStr = e + } else { + keyStr = e.key + valStr = e.value + } + + // Fallback encode to latin1 if it's ascii for keys + let keyBytes = keyEncoder.encode(keyStr) + let valBytes = encoder.encode(valStr) // for simplify, we use utf-8 + + const h = tblHash(keyBytes, hashTableSize) + tuples.push({ index: i, keyStr: keyStr, keyBytes, valBytes, hash: h, targetBucket: h }) + } + + let maxTries = 0 + + for (const t of tuples) { + let placed = false + let tries = 0 + for (; tries < hashTableSize; tries++) { + const b = (t.targetBucket + tries) % hashTableSize + const base = b * 17 + if (nodes[base] === 0) { + // Place it + nodes[base] = 1 // isActive + nodeView.setUint16(base + 1, t.index, true) + nodeView.setUint32(base + 3, t.targetBucket, true) // Wait, is hashValue the bucket index? + // "hashValue stores the ALREADY-MODULO'D bucket index, i.e. pjw(key) % hashTableSize" + placed = true + maxTries = Math.max(maxTries, tries + 1) + break } } - const bytes = new Uint8Array([characters & 0xff, (characters >> 8) & 0xff, ...body]) - blocks.push({ index, bytes }) - cursor += bytes.byteLength + if (!placed) throw new Error("Hash table full") } - const out = new Uint8Array(cursor) + + // Now place string data + let stringOffset = 21 + entries.length * 2 + hashTableSize * 17 + const strings: Uint8Array[] = [] + let currentStringOffset = stringOffset + + for (let b = 0; b < hashTableSize; b++) { + const base = b * 17 + if (nodes[base] === 0) continue + const index = nodeView.getUint16(base + 1, true) + const t = tuples.find(x => x.index === index)! + + // key is NUL terminated + nodeView.setUint32(base + 7, currentStringOffset, true) + const keyBuf = new Uint8Array(t.keyBytes.length + 1) + keyBuf.set(t.keyBytes) + strings.push(keyBuf) + currentStringOffset += keyBuf.length + + // value + nodeView.setUint32(base + 11, currentStringOffset, true) + // valLength includes NUL + nodeView.setUint16(base + 15, t.valBytes.length + 1, true) + const valBuf = new Uint8Array(t.valBytes.length + 1) + valBuf.set(t.valBytes) + strings.push(valBuf) + currentStringOffset += valBuf.length + } + + const out = new Uint8Array(currentStringOffset) const view = new DataView(out.buffer) + + // header view.setUint16(0, 0x1234, true) view.setUint16(2, entries.length, true) - let at = headerBytes + indexBytes - for (const block of blocks) { - view.setUint16(headerBytes + block.index * 2, at, true) - out.set(block.bytes, at) - at += block.bytes.byteLength + view.setUint32(4, hashTableSize, true) + view.setUint8(8, 1) // version + view.setUint32(9, stringOffset, true) + view.setUint32(13, maxTries, true) + view.setUint32(17, currentStringOffset, true) + + // indices + for (let i = 0; i < entries.length; i++) { + // wait! The index array in real files, what does it map? + // It maps string index -> bucket index. + // If an index is not used, it typically maps to 0 or something? Or is it an array of bucket indices? + // Let's lookup what bucket index it ended up in. + let bucket = 0 + if (entries[i] !== null) { + // find bucket + for (let b = 0; b < hashTableSize; b++) { + if (nodes[b * 17] && nodeView.getUint16(b * 17 + 1, true) === i) { + bucket = b + break + } + } + } + view.setUint16(21 + i * 2, bucket, true) } + + out.set(nodes, 21 + entries.length * 2) + + let copied = stringOffset + for (const s of strings) { + out.set(s, copied) + copied += s.length + } + return out } diff --git a/scripts/verify-tbl.ts b/scripts/verify-tbl.ts index ca43d63..f4f59b8 100644 --- a/scripts/verify-tbl.ts +++ b/scripts/verify-tbl.ts @@ -1,59 +1,78 @@ /** - * Construction check for the TBL decoder. + * Construction check and real-file check for the TBL decoder. * - * The classic TBL layout has no independent decoder available (the community Go - * package implements the later hash-table variant), so this is a two-sided test - * rather than a differential one: the script *writes* a table whose entries are - * known — including empty strings, unused indices and non-ASCII text — and then - * requires the decoder to reproduce exactly that. It catches layout and - * decoding mistakes (off-by-one offsets, byte/character confusion, UTF-16 - * endianness) but cannot catch a wrong reading shared by writer and reader; - * that is what a real `string.tbl` is for. - * - * Usage: node scripts/verify-tbl.ts + * Requirements: + * - Round-trip encoding through tbl-writer + * - Success on actual Diablo II patchstring.tbl */ -import { decodeTbl, tblLookup } from '../src/formats/tbl.ts' +import { readFileSync } from 'node:fs' +import { join } from 'node:path' +import { decodeTbl, tblLookup, tblHash } from '../src/formats/tbl.ts' import { encodeTbl } from './lib/tbl-writer.ts' import type { TblEntry } from './lib/tbl-writer.ts' -/** Entries the decoder must reproduce, in order. */ const entries: TblEntry[] = [ - 'Stamina Potion', // plain ASCII - '', // empty string is a value, not a hole - null, // unused index - 'Tome of Town Portal', + { key: "Stamina Potion", value: "Stamina Potion" }, + { key: "Empty", value: "" }, null, - '完美宝石', // CJK: multi-byte UTF-16, but one unit each - 'Straße', // Latin-1 range + { key: "Tome", value: "Tome of Town Portal" }, null, - '🌀 Emoji beyond the BMP', // surrogate pair: two units for one character - 'x'.repeat(300), // long entry + { key: "CJK", value: "完美宝石" }, + { key: "Latin", value: "Straße" }, + null, + { key: "Emoji", value: "🌀 Emoji beyond the BMP" }, + { key: "Long", value: 'x'.repeat(300) }, ] const encoded = encodeTbl(entries) -const decoded = decodeTbl(encoded) +// We pass 'utf-8' here because our tbl-writer encodes values in utf-8 to preserve things for testing +const decoded = decodeTbl(encoded, 'utf-8') const lookup = tblLookup(decoded) const problems: string[] = [] -if (encoded.byteLength !== decoded.length * 0 + encoded.byteLength) { - /* v8 ignore next -- placeholder to keep the shape obvious. */ - problems.push('unreachable') -} -if (decoded.length !== entries.length) problems.push(`entry count ${String(decoded.length)} != ${String(entries.length)}`) +if (decoded.length !== entries.length) problems.push(`entry count ${decoded.length} != ${entries.length}`) entries.forEach((want, index) => { const got = decoded[index] if (want === null) { - if (got !== undefined) problems.push(`index ${String(index)}: expected an unused index, got ${JSON.stringify(got)}`) + if (got !== undefined) problems.push(`index ${index}: expected an unused index, got ${JSON.stringify(got)}`) return } - if (got !== want) problems.push(`index ${String(index)}: ${JSON.stringify(got)} != ${JSON.stringify(want)}`) + const expectedValue = typeof want === 'string' ? want : want.value + if (got !== expectedValue) problems.push(`index ${index}: ${JSON.stringify(got)} != ${JSON.stringify(expectedValue)}`) + + const expectedKey = typeof want === 'string' ? `__auto_key_${index}` : want.key + if (decoded.dict.get(expectedKey) !== expectedValue) { + problems.push(`Dictionary lookup failed for key ${expectedKey}`) + } }) -console.log(`table ${String(entries.length)} indices (${String(lookup.size)} used)`) -console.log(`encoded ${String(encoded.byteLength)} bytes`) -console.log(`unicode CJK=${JSON.stringify(lookup.get(5))} latin1=${JSON.stringify(lookup.get(6))} astral=${JSON.stringify(lookup.get(8))}`) -console.log(`long ${String(lookup.get(9)?.length ?? 0)} characters round-tripped`) -console.log(`problems ${String(problems.length)}`) +console.log(`[Roundtrip] table ${entries.length} indices (${lookup.size} used)`) +console.log(`[Roundtrip] encoded ${encoded.byteLength} bytes`) +console.log(`[Roundtrip] unicode CJK=${JSON.stringify(lookup.get(5))} latin1=${JSON.stringify(lookup.get(6))} astral=${JSON.stringify(lookup.get(8))}`) +console.log(`[Roundtrip] long ${lookup.get(9)?.length ?? 0} characters round-tripped`) +console.log(`[Roundtrip] problems ${problems.length}`) + +let fixtureProblems = 0 +try { + const raw = readFileSync(join(import.meta.dirname, '../samples/fixtures/tbl/patchstring_113c_eng.tbl')) + const fixtureDecoded = decodeTbl(raw) // defaults to windows-1252 + let hashFailures = 0 + let matched = 0 + for (const [key, value] of fixtureDecoded.dict.entries()) { + matched++ + // To verify, we would need the hashtablesize. We can't access it natively through just decodeTbl. + // Instead we can just check if we can retrieve known values: + } + if (fixtureDecoded.dict.get("Party1X") !== "%s permits you to loot his corpse.") fixtureProblems++ + if (fixtureDecoded.dict.get("D2bnetHelp16b") !== "/unignore <*accountname>") fixtureProblems++ + + console.log(`[Fixture] Loaded fixture correctly. Total active keys: ${matched}`) +} catch (e: any) { + problems.push(`Fixture failed: ${e.message}`) + fixtureProblems++ +} + for (const problem of problems.slice(0, 8)) console.log(` - ${problem}`) -console.log(problems.length === 0 ? 'RESULT every index round-trips exactly' : 'RESULT FAILED') -process.exit(problems.length === 0 ? 0 : 1) +const success = problems.length === 0 && fixtureProblems === 0 +console.log(success ? 'RESULT all passed' : 'RESULT FAILED') +process.exit(success ? 0 : 1) diff --git a/src/formats/tbl.ts b/src/formats/tbl.ts index be43c02..b096e9e 100644 --- a/src/formats/tbl.ts +++ b/src/formats/tbl.ts @@ -1,39 +1,9 @@ /** - * Diablo II `.tbl` string-table decoder (classic layout). - * - * A TBL is an *indexed* string table — item, skill, monster and quest names are - * referred to by number, not by key: - * - * ``` - * u16 crc (unused here; the game uses it as a content checksum) - * u16 entryCount - * u16 offsets[entryCount] absolute file offsets, one per index - * ... strings: u16 characterCount, then characterCount * 2 bytes UTF-16LE - * ``` - * - * The classic Diablo II files (`string.tbl`, `expansionstring.tbl`, - * `patchstring.tbl`) use exactly this layout. A later "extended" variant adds a - * hash table of name/value pairs on top; the community Go package implements - * that variant instead, so unlike the other formats in this directory **there is - * no independent decoder to diff this one against** — it is checked by - * construction (write a table, read it back, including non-ASCII) and is on the - * list to confirm against a real `string.tbl`. - * - * Unused indices are legal: an offset that is zero, or that points past the - * table, decodes to `undefined` rather than an empty string, so a caller can - * tell "no such string" from "empty string". + * Diablo II `.tbl` string-table decoder (hash-bucket layout). + * + * Implements the verified structure of Diablo II `.tbl` files. */ -/** Bytes of the fixed header: crc + entry count. */ -const HEADER_BYTES = 4 -/** Bytes per index-table entry. */ -const INDEX_BYTES = 2 -/** Bytes of one string's length prefix. */ -const LENGTH_BYTES = 2 -/** Bytes per UTF-16 code unit. */ -const CODE_UNIT_BYTES = 2 - -/** Raised when a TBL file is malformed. */ export class TblError extends Error { constructor(message: string) { super(message) @@ -41,70 +11,188 @@ export class TblError extends Error { } } -/** - * Read a little-endian uint16. - * - * @param data - the buffer. - * @param at - byte offset. - * @returns the value. - */ +export function tblHash(key: Uint8Array, hashTableSize: number): number { + if (hashTableSize === 0) return 0 + let h = 0 + for (let i = 0; i < key.length; i++) { + h = ((h << 4) + key[i]!) >>> 0 + const high = h & 0xf0000000 + if (high !== 0) h ^= high >>> 24 + h = (h & ~high) >>> 0 + } + return h % hashTableSize +} + function u16(data: Uint8Array, at: number): number { return data[at]! | (data[at + 1]! << 8) } -/** - * Decode a TBL file. - * - * @param data - the complete file. - * @returns one entry per string index; `undefined` where the index is unused. - */ -export function decodeTbl(data: Uint8Array): (string | undefined)[] { - if (data.byteLength < HEADER_BYTES) { - throw new TblError(`TBL is ${String(data.byteLength)} bytes, too short for a header`) - } - const entryCount = u16(data, 2) - const indexEnd = HEADER_BYTES + entryCount * INDEX_BYTES - if (indexEnd > data.byteLength) { - throw new TblError(`TBL declares ${String(entryCount)} entries but the index table runs past the file`) - } - const decoder = new TextDecoder('utf-16le') - const entries: (string | undefined)[] = [] - for (let index = 0; index < entryCount; index += 1) { - const offset = u16(data, HEADER_BYTES + index * INDEX_BYTES) - // A zero offset marks an unused index; so does one that cannot hold a - // length prefix. Both are normal in shipped tables. - if (offset === 0 || offset + LENGTH_BYTES > data.byteLength) { - entries.push(undefined) - continue - } - const characters = u16(data, offset) - const from = offset + LENGTH_BYTES - const to = from + characters * CODE_UNIT_BYTES - if (to > data.byteLength) { - throw new TblError(`entry ${String(index)} at ${String(offset)} declares ${String(characters)} characters past the file end`) - } - // The stored count is characters, not bytes; decoding from a view keeps the - // conversion (including surrogate pairs) in the platform's hands. - entries.push(decoder.decode(data.subarray(from, to))) - } - return entries +function u32(data: Uint8Array, at: number): number { + return (data[at]! | (data[at + 1]! << 8) | (data[at + 2]! << 16) | (data[at + 3]! << 24)) >>> 0 +} + +function readCStr(data: Uint8Array, start: number): Uint8Array { + let end = start + while (end < data.byteLength && data[end] !== 0) { + end++ + } + return data.subarray(start, end) +} + +export type DecodedTbl = (string | undefined)[] & { + dict: Map +} + +const decoderCache = new Map() +function getDecoder(encoding: string): TextDecoder { + if (decoderCache.has(encoding)) return decoderCache.get(encoding)! + try { + const dec = new TextDecoder(encoding) + decoderCache.set(encoding, dec) + return dec + } catch { + const fallback = new TextDecoder('latin1') + decoderCache.set(encoding, fallback) + return fallback + } +} + +const keyDecoder = new TextDecoder('latin1') + +export function decodeTbl(data: Uint8Array, encoding = 'windows-1252'): DecodedTbl { + if (data.byteLength < 21) { + throw new TblError(`TBL is ${data.byteLength} bytes, too short for a 21-byte header`) + } + + const crc = u16(data, 0) + const numElements = u16(data, 2) + const hashTableSize = u32(data, 4) + const version = data[8] + const stringOffset = u32(data, 9) + const maxTries = u32(data, 13) + const fileSize = u32(data, 17) + + const nodesStart = 21 + numElements * 2 + const nodesEnd = nodesStart + hashTableSize * 17 + + if (nodesEnd > stringOffset) { + throw new TblError(`header arithmetic: 21 + ${numElements}*2 + ${hashTableSize}*17 = ${nodesEnd} which exceeds stringOffset ${stringOffset}`) + } + if (stringOffset > fileSize) { + throw new TblError(`stringOffset ${stringOffset} exceeds fileSize ${fileSize}`) + } + if (fileSize !== data.byteLength) { + throw new TblError(`fileSize ${fileSize} does not match actual buffer length ${data.byteLength}`) + } + + const valDecoder = getDecoder(encoding) + + const out = [] as unknown as DecodedTbl + out.dict = new Map() + + for (let i = 0; i < hashTableSize; i++) { + const base = nodesStart + i * 17 + const isActive = data[base] + if (!isActive) continue + + const index = u16(data, base + 1) + const hashValue = u32(data, base + 3) + const keyOffset = u32(data, base + 7) + const valOffset = u32(data, base + 11) + const valLen = u16(data, base + 15) + + if (keyOffset >= data.byteLength) { + throw new TblError(`node ${i} keyOffset ${keyOffset} outside buffer`) + } + if (valOffset >= data.byteLength) { + throw new TblError(`node ${i} valOffset ${valOffset} outside buffer`) + } + + const keySlice = readCStr(data, keyOffset) + const key = keyDecoder.decode(keySlice) + + const readLen = Math.max(0, valLen - 1) + const valSlice = data.subarray(valOffset, valOffset + readLen) + const value = valDecoder.decode(valSlice) + + out[index] = value + if (!out.dict.has(key)) { + out.dict.set(key, value) + } + } + + return out +} + +export function lookupTblFast(data: Uint8Array, key: string, encoding = 'windows-1252'): string | undefined { + if (data.byteLength < 21) return undefined + const numElements = u16(data, 2) + const hashTableSize = u32(data, 4) + if (hashTableSize === 0) return undefined + const stringOffset = u32(data, 9) + const maxTries = u32(data, 13) + + const nodesStart = 21 + numElements * 2 + const nodesEnd = nodesStart + hashTableSize * 17 + if (nodesEnd > stringOffset || stringOffset > data.byteLength) return undefined + + const keyEncoder = new TextEncoder() + let keyBuf: Uint8Array + let isAscii = true + for (let i = 0; i < key.length; i++) { + if (key.charCodeAt(i) > 0x7F) { + isAscii = false + break + } + } + if (isAscii) { + keyBuf = keyEncoder.encode(key) + } else { + keyBuf = new Uint8Array(key.length) + for (let i = 0; i < key.length; i++) { + keyBuf[i] = key.charCodeAt(i) & 0xFF + } + } + + const targetHash = tblHash(keyBuf, hashTableSize) + const valDecoder = getDecoder(encoding) + + for (let i = 0; i < maxTries; i++) { + const bucket = (targetHash + i) % hashTableSize + const base = nodesStart + bucket * 17 + const isActive = data[base] + if (!isActive) return undefined + + const nodeHash = u32(data, base + 3) + if (nodeHash !== targetHash) continue + + const keyOffset = u32(data, base + 7) + if (keyOffset >= data.byteLength) continue + + const keySlice = readCStr(data, keyOffset) + if (keySlice.byteLength !== keyBuf.byteLength) continue + + let match = true + for (let j = 0; j < keyBuf.byteLength; j++) { + if (keySlice[j] !== keyBuf[j]) { + match = false + break + } + } + + if (match) { + const valOffset = u32(data, base + 11) + const valLen = u16(data, base + 15) + const readLen = Math.max(0, valLen - 1) + if (valOffset + readLen > data.byteLength) return undefined + return valDecoder.decode(data.subarray(valOffset, valOffset + readLen)) + } + } + return undefined } -/** - * Build an index → string lookup over a decoded table. - * - * @param entries - a decoded table. - * @returns the lookup. - */ export function tblLookup(entries: readonly (string | undefined)[]): { - /** Number of usable entries. */ readonly size: number - /** - * Read one entry. - * - * @param index - the string index. - * @returns the string, or undefined when the index is unused. - */ get: (index: number) => string | undefined } { let size = 0 diff --git a/src/game/level-names-zh.ts b/src/game/level-names-zh.ts index 9459040..4bc7544 100644 --- a/src/game/level-names-zh.ts +++ b/src/game/level-names-zh.ts @@ -1,3 +1,5 @@ +// TODO: tbl.ts now decodes the real hash-bucket format. This hardcoded table can be retired once the Chinese MPQ assets (CHI/string.tbl) are available. (fixes #8) + // level-names-zh.ts — 关卡/场景的中文显示名(用于页面上的三级选择器) // // 出处说明(重要,避免误认为官方本地化): diff --git a/tests/tbl.test.ts b/tests/tbl.test.ts new file mode 100644 index 0000000..6672f95 --- /dev/null +++ b/tests/tbl.test.ts @@ -0,0 +1,148 @@ +import { describe, test, expect } from 'vitest' +import { readFileSync } from 'node:fs' +import { join } from 'node:path' +import { decodeTbl, TblError, tblHash, lookupTblFast } from '../src/formats/tbl.ts' +import { encodeTbl } from '../scripts/lib/tbl-writer.ts' +import type { TblEntry } from '../scripts/lib/tbl-writer.ts' + +const fixturePath = join(__dirname, '../samples/fixtures/tbl/patchstring_113c_eng.tbl') +const fixtureData = readFileSync(fixturePath) + +describe('TBL Decoder', () => { + test('Header parse and Layout invariant', () => { + const data = fixtureData + const view = new DataView(data.buffer, data.byteOffset, data.byteLength) + const crc = view.getUint16(0, true) + const numElements = view.getUint16(2, true) + const hashTableSize = view.getUint32(4, true) + const version = view.getUint8(8) + const stringOffset = view.getUint32(9, true) + const maxTries = view.getUint32(13, true) + const fileSize = view.getUint32(17, true) + + expect(crc).toBe(23974) + expect(numElements).toBe(1169) + expect(hashTableSize).toBe(1171) + expect(version).toBe(1) + expect(stringOffset).toBe(22266) + expect(maxTries).toBe(1058) + expect(fileSize).toBe(52523) + + expect(21 + numElements * 2 + hashTableSize * 17).toBe(stringOffset) + expect(fileSize).toBe(data.byteLength) + }) + + test('Entry count and dictionary', () => { + const decoded = decodeTbl(fixtureData) + let count = 0; decoded.forEach(v => { if (v !== undefined) count++ }); expect(count).toBe(1169) + }) + + test('Known strings', () => { + const decoded = decodeTbl(fixtureData) + expect(decoded.dict.get('Party1X')).toBe('%s permits you to loot his corpse.') + expect(decoded.dict.get('D2bnetHelp16b')).toBe('/unignore <*accountname>') + expect(decoded.dict.get('skillxld66')).toBe(' ') + }) + + test('Fast lookup matches full decode', () => { + expect(lookupTblFast(fixtureData, 'Party1X')).toBe('%s permits you to loot his corpse.') + expect(lookupTblFast(fixtureData, 'D2bnetHelp16b')).toBe('/unignore <*accountname>') + expect(lookupTblFast(fixtureData, 'skillxld66')).toBe(' ') + }) + + test('Hash self-check', () => { + const data = fixtureData + const view = new DataView(data.buffer, data.byteOffset, data.byteLength) + const numElements = view.getUint16(2, true) + const hashTableSize = view.getUint32(4, true) + + const nodesStart = 21 + numElements * 2 + let checkedCount = 0 + + for (let i = 0; i < hashTableSize; i++) { + const base = nodesStart + i * 17 + const isActive = view.getUint8(base) + if (!isActive) continue + + const hashValue = view.getUint32(base + 3, true) + const keyOffset = view.getUint32(base + 7, true) + + let end = keyOffset + while (data[end] !== 0) end++ + const keyBuf = data.subarray(keyOffset, end) + + const computed = tblHash(keyBuf, hashTableSize) + expect(computed).toBe(hashValue) + + checkedCount++ + } + + expect(checkedCount).toBe(1169) + }) + + test('Round-trip encoding', () => { + const entries: TblEntry[] = [ + { key: "Stamina Potion", value: "Stamina Potion" }, + { key: "Empty", value: "" }, + null, + { key: "Tome", value: "Tome of Town Portal" }, + null, + { key: "CJK", value: "完美宝石" }, + { key: "Latin", value: "Straße" }, + null, + { key: "Emoji", value: "🌀 Emoji beyond the BMP" }, + { key: "Long", value: 'x'.repeat(300) }, + ] + + // Add enough entities to force collisions + for (let i = 0; i < 200; i++) { + entries.push({ key: `CollC${i}`, value: `ValC${i}` }) + } + + // Duplicate-ish keys differing by case + entries.push({ key: "Apple", value: "Uppercase" }) + entries.push({ key: "apple", value: "Lowercase" }) + + const encoded = encodeTbl(entries) + const decoded = decodeTbl(encoded, 'utf-8') + + expect(decoded.dict.get("Empty")).toBe("") + expect(decoded.dict.get("Latin")).toBe("Straße") + expect(decoded.dict.get("CJK")).toBe("完美宝石") + expect(decoded.dict.get("Apple")).toBe("Uppercase") + expect(decoded.dict.get("apple")).toBe("Lowercase") + expect(decoded.dict.get("CollC199")).toBe("ValC199") + + // Also test fast lookup + expect(lookupTblFast(encoded, "Latin", "utf-8")).toBe("Straße") + }) + + test('Malformed input: truncated buffer', () => { + const shortBuf = new Uint8Array(20) + expect(() => decodeTbl(shortBuf)).toThrowError(TblError) + expect(() => decodeTbl(shortBuf)).toThrowError(/too short/) + }) + + test('Malformed input: stringOffset past EOF', () => { + const buf = new Uint8Array(fixtureData) + const view = new DataView(buf.buffer, buf.byteOffset, buf.byteLength) + // Make stringOffset larger than file length + view.setUint32(9, buf.length + 10, true) + + expect(() => decodeTbl(buf)).toThrowError(TblError) + expect(() => decodeTbl(buf)).toThrowError(/stringOffset/) + }) + + test('Malformed input: node offset out of range', () => { + const buf = new Uint8Array(fixtureData) + const view = new DataView(buf.buffer, buf.byteOffset, buf.byteLength) + + // Get the first active node and corrupt its key offset + const numElements = view.getUint16(2, true) + const nodesStart = 21 + numElements * 2 + view.setUint32(nodesStart + 7, buf.length + 5, true) // key offset is at +7 + + expect(() => decodeTbl(buf)).toThrowError(TblError) + expect(() => decodeTbl(buf)).toThrowError(/keyOffset/) + }) +})