diablo2-web/src/formats/tbl.ts

117 lines
4.0 KiB
TypeScript

/**
* Diablo II `.tbl` string-table decoder (classic layout).
*
* A TBL is an *indexed* string table — item, skill, monster and quest names are
* referred to by number, not by key:
*
* ```
* u16 crc (unused here; the game uses it as a content checksum)
* u16 entryCount
* u16 offsets[entryCount] absolute file offsets, one per index
* ... strings: u16 characterCount, then characterCount * 2 bytes UTF-16LE
* ```
*
* The classic Diablo II files (`string.tbl`, `expansionstring.tbl`,
* `patchstring.tbl`) use exactly this layout. A later "extended" variant adds a
* hash table of name/value pairs on top; the community Go package implements
* that variant instead, so unlike the other formats in this directory **there is
* no independent decoder to diff this one against** — it is checked by
* construction (write a table, read it back, including non-ASCII) and is on the
* list to confirm against a real `string.tbl`.
*
* Unused indices are legal: an offset that is zero, or that points past the
* table, decodes to `undefined` rather than an empty string, so a caller can
* tell "no such string" from "empty string".
*/
/** Bytes of the fixed header: crc + entry count. */
const HEADER_BYTES = 4
/** Bytes per index-table entry. */
const INDEX_BYTES = 2
/** Bytes of one string's length prefix. */
const LENGTH_BYTES = 2
/** Bytes per UTF-16 code unit. */
const CODE_UNIT_BYTES = 2
/** Raised when a TBL file is malformed. */
export class TblError extends Error {
constructor(message: string) {
super(message)
this.name = 'TblError'
}
}
/**
* Read a little-endian uint16.
*
* @param data - the buffer.
* @param at - byte offset.
* @returns the value.
*/
function u16(data: Uint8Array, at: number): number {
return data[at]! | (data[at + 1]! << 8)
}
/**
* Decode a TBL file.
*
* @param data - the complete file.
* @returns one entry per string index; `undefined` where the index is unused.
*/
export function decodeTbl(data: Uint8Array): (string | undefined)[] {
if (data.byteLength < HEADER_BYTES) {
throw new TblError(`TBL is ${String(data.byteLength)} bytes, too short for a header`)
}
const entryCount = u16(data, 2)
const indexEnd = HEADER_BYTES + entryCount * INDEX_BYTES
if (indexEnd > data.byteLength) {
throw new TblError(`TBL declares ${String(entryCount)} entries but the index table runs past the file`)
}
const decoder = new TextDecoder('utf-16le')
const entries: (string | undefined)[] = []
for (let index = 0; index < entryCount; index += 1) {
const offset = u16(data, HEADER_BYTES + index * INDEX_BYTES)
// A zero offset marks an unused index; so does one that cannot hold a
// length prefix. Both are normal in shipped tables.
if (offset === 0 || offset + LENGTH_BYTES > data.byteLength) {
entries.push(undefined)
continue
}
const characters = u16(data, offset)
const from = offset + LENGTH_BYTES
const to = from + characters * CODE_UNIT_BYTES
if (to > data.byteLength) {
throw new TblError(`entry ${String(index)} at ${String(offset)} declares ${String(characters)} characters past the file end`)
}
// The stored count is characters, not bytes; decoding from a view keeps the
// conversion (including surrogate pairs) in the platform's hands.
entries.push(decoder.decode(data.subarray(from, to)))
}
return entries
}
/**
* Build an index → string lookup over a decoded table.
*
* @param entries - a decoded table.
* @returns the lookup.
*/
export function tblLookup(entries: readonly (string | undefined)[]): {
/** Number of usable entries. */
readonly size: number
/**
* Read one entry.
*
* @param index - the string index.
* @returns the string, or undefined when the index is unused.
*/
get: (index: number) => string | undefined
} {
let size = 0
for (const entry of entries) if (entry !== undefined) size += 1
return {
size,
get: index => (index >= 0 && index < entries.length ? entries[index] : undefined),
}
}