/** * Diablo II `.tbl` string-table decoder (classic layout). * * A TBL is an *indexed* string table — item, skill, monster and quest names are * referred to by number, not by key: * * ``` * u16 crc (unused here; the game uses it as a content checksum) * u16 entryCount * u16 offsets[entryCount] absolute file offsets, one per index * ... strings: u16 characterCount, then characterCount * 2 bytes UTF-16LE * ``` * * The classic Diablo II files (`string.tbl`, `expansionstring.tbl`, * `patchstring.tbl`) use exactly this layout. A later "extended" variant adds a * hash table of name/value pairs on top; the community Go package implements * that variant instead, so unlike the other formats in this directory **there is * no independent decoder to diff this one against** — it is checked by * construction (write a table, read it back, including non-ASCII) and is on the * list to confirm against a real `string.tbl`. * * Unused indices are legal: an offset that is zero, or that points past the * table, decodes to `undefined` rather than an empty string, so a caller can * tell "no such string" from "empty string". */ /** Bytes of the fixed header: crc + entry count. */ const HEADER_BYTES = 4 /** Bytes per index-table entry. */ const INDEX_BYTES = 2 /** Bytes of one string's length prefix. */ const LENGTH_BYTES = 2 /** Bytes per UTF-16 code unit. */ const CODE_UNIT_BYTES = 2 /** Raised when a TBL file is malformed. */ export class TblError extends Error { constructor(message: string) { super(message) this.name = 'TblError' } } /** * Read a little-endian uint16. * * @param data - the buffer. * @param at - byte offset. * @returns the value. */ function u16(data: Uint8Array, at: number): number { return data[at]! | (data[at + 1]! << 8) } /** * Decode a TBL file. * * @param data - the complete file. * @returns one entry per string index; `undefined` where the index is unused. */ export function decodeTbl(data: Uint8Array): (string | undefined)[] { if (data.byteLength < HEADER_BYTES) { throw new TblError(`TBL is ${String(data.byteLength)} bytes, too short for a header`) } const entryCount = u16(data, 2) const indexEnd = HEADER_BYTES + entryCount * INDEX_BYTES if (indexEnd > data.byteLength) { throw new TblError(`TBL declares ${String(entryCount)} entries but the index table runs past the file`) } const decoder = new TextDecoder('utf-16le') const entries: (string | undefined)[] = [] for (let index = 0; index < entryCount; index += 1) { const offset = u16(data, HEADER_BYTES + index * INDEX_BYTES) // A zero offset marks an unused index; so does one that cannot hold a // length prefix. Both are normal in shipped tables. if (offset === 0 || offset + LENGTH_BYTES > data.byteLength) { entries.push(undefined) continue } const characters = u16(data, offset) const from = offset + LENGTH_BYTES const to = from + characters * CODE_UNIT_BYTES if (to > data.byteLength) { throw new TblError(`entry ${String(index)} at ${String(offset)} declares ${String(characters)} characters past the file end`) } // The stored count is characters, not bytes; decoding from a view keeps the // conversion (including surrogate pairs) in the platform's hands. entries.push(decoder.decode(data.subarray(from, to))) } return entries } /** * Build an index → string lookup over a decoded table. * * @param entries - a decoded table. * @returns the lookup. */ export function tblLookup(entries: readonly (string | undefined)[]): { /** Number of usable entries. */ readonly size: number /** * Read one entry. * * @param index - the string index. * @returns the string, or undefined when the index is unused. */ get: (index: number) => string | undefined } { let size = 0 for (const entry of entries) if (entry !== undefined) size += 1 return { size, get: index => (index >= 0 && index < entries.length ? entries[index] : undefined), } }