Canonicalize reward IDs during case-insensitive mapping

This commit is contained in:
Golumpa 2026-08-19 13:10:33 +01:00
parent 5fe39b96ae
commit 7f7ad80b46
9 changed files with 64 additions and 38 deletions

View file

@ -84,6 +84,12 @@ Current stable pool IDs:
Avoid using `banner.name`, `reward_name`, `reward_type` or `reward_rank` as primary IDs. They Avoid using `banner.name`, `reward_name`, `reward_type` or `reward_rank` as primary IDs. They
are useful display fields, but may change when mapping files are updated. are useful display fields, but may change when mapping files are updated.
Known `reward_id` values are matched case-insensitively and emitted using the
canonical casing from the bundled mapping. This prevents packet/asset casing
differences from dropping display metadata or producing different downstream
IDs. An unknown reward keeps the casing decoded from the packet. Reward IDs are
not UID inputs, so canonicalizing their casing does not change pull UIDs.
## Stability notes ## Stability notes
Every JSON record is exported with a stable `uid`. Re-scanning deeper history can Every JSON record is exported with a stable `uid`. Re-scanning deeper history can

View file

@ -101,6 +101,8 @@ Decoded fields:
Arc/Gashapon history uses the same scheme at `ASCII * 2`. Arc/Gashapon history uses the same scheme at `ASCII * 2`.
- The decoder decodes the key to its id string and looks up display metadata - The decoder decodes the key to its id string and looks up display metadata
in `mappings/arcs.json`, `mappings/characters.json`, and `mappings/items.json`. in `mappings/arcs.json`, `mappings/characters.json`, and `mappings/items.json`.
Lookups are case-insensitive across Monopoly, Arc, and Mystery Box decoders;
known IDs are emitted using mapping-canonical casing.
Achievement display metadata is generated separately in Achievement display metadata is generated separately in
`mappings/achievements.json`. `mappings/achievements.json`.
Unknown rewards still export their decoded id with empty name/rank. Unknown rewards still export their decoded id with empty name/rank.

View file

@ -21,7 +21,7 @@ from nte_history_exporter.constants import (
) )
from nte_history_exporter.decoder.boundary import select_continuous_run_from_page_1 from nte_history_exporter.decoder.boundary import select_continuous_run_from_page_1
from nte_history_exporter.decoder.run import fmt_packet_time from nte_history_exporter.decoder.run import fmt_packet_time
from nte_history_exporter.mappings import ARC_META from nte_history_exporter.mappings import ARC_META_BY_CASEFOLD
from nte_history_exporter.decoder.structured_protocol import ( from nte_history_exporter.decoder.structured_protocol import (
StructuredProtocolAssembler, StructuredProtocolAssembler,
StructuredRecord, StructuredRecord,
@ -95,7 +95,7 @@ def _parse_legacy_arc_response(response: bytes) -> list[dict[str, Any]]:
timestamp_raw = response[pos : pos + 8] timestamp_raw = response[pos : pos + 8]
pos += 8 pos += 8
arc_id = decode_arc_key(name_raw) or name_raw.hex() arc_id = decode_arc_key(name_raw) or name_raw.hex()
meta = ARC_META.get(arc_id, {}) canonical_id, meta = _resolve_arc_metadata(arc_id)
try: try:
ticks, unix_seconds, timestamp_decoded = decode_arc_timestamp(timestamp_raw) ticks, unix_seconds, timestamp_decoded = decode_arc_timestamp(timestamp_raw)
except ValueError: except ValueError:
@ -107,7 +107,7 @@ def _parse_legacy_arc_response(response: bytes) -> list[dict[str, Any]]:
"record_len": pos - start, "record_len": pos - start,
"reward_key_hex": name_raw.hex(), "reward_key_hex": name_raw.hex(),
"reward_type": "arc", "reward_type": "arc",
"reward_id": arc_id, "reward_id": canonical_id,
"reward_name": meta.get("name", "UNKNOWN"), "reward_name": meta.get("name", "UNKNOWN"),
"reward_rank": meta.get("rank", ""), "reward_rank": meta.get("rank", ""),
"type_key_hex": type_raw.hex(), "type_key_hex": type_raw.hex(),
@ -122,18 +122,15 @@ def _parse_legacy_arc_response(response: bytes) -> list[dict[str, Any]]:
return records return records
def _arc_metadata(arc_id: str) -> dict[str, Any]: def _resolve_arc_metadata(arc_id: str) -> tuple[str, dict[str, Any]]:
direct = ARC_META.get(arc_id) canonical_id, meta = ARC_META_BY_CASEFOLD.get(arc_id.casefold(), (arc_id, {}))
if direct is not None: return canonical_id, meta
return direct
folded = arc_id.casefold()
return next((meta for item_id, meta in ARC_META.items() if item_id.casefold() == folded), {})
def structured_arc_rows(structured_rows: list[StructuredRecord]) -> list[dict[str, Any]]: def structured_arc_rows(structured_rows: list[StructuredRecord]) -> list[dict[str, Any]]:
records = [] records = []
for structured in structured_rows: for structured in structured_rows:
meta = _arc_metadata(structured.item_id) canonical_id, meta = _resolve_arc_metadata(structured.item_id)
records.append( records.append(
{ {
"record_start": structured.record_start, "record_start": structured.record_start,
@ -141,7 +138,7 @@ def structured_arc_rows(structured_rows: list[StructuredRecord]) -> list[dict[st
"record_len": structured.record_end - structured.record_start, "record_len": structured.record_end - structured.record_start,
"reward_key_hex": "", "reward_key_hex": "",
"reward_type": "arc", "reward_type": "arc",
"reward_id": structured.item_id, "reward_id": canonical_id,
"reward_name": meta.get("name", "UNKNOWN"), "reward_name": meta.get("name", "UNKNOWN"),
"reward_rank": meta.get("rank", ""), "reward_rank": meta.get("rank", ""),
"type_key_hex": "", "type_key_hex": "",

View file

@ -24,16 +24,11 @@ from nte_history_exporter.constants import (
from nte_history_exporter.decoder.boundary import select_continuous_run_from_page_1 from nte_history_exporter.decoder.boundary import select_continuous_run_from_page_1
from nte_history_exporter.decoder.protocol import infer_reward_type from nte_history_exporter.decoder.protocol import infer_reward_type
from nte_history_exporter.decoder.run import fmt_packet_time from nte_history_exporter.decoder.run import fmt_packet_time
from nte_history_exporter.mappings import REWARDS_BY_ID from nte_history_exporter.mappings import REWARDS_BY_CASEFOLD
DOTNET_TICKS_PER_SECOND = 10_000_000 DOTNET_TICKS_PER_SECOND = 10_000_000
MAX_RECORDS_PER_BLOCK = 100 MAX_RECORDS_PER_BLOCK = 100
MAX_REWARD_ID_LENGTH = 256 MAX_REWARD_ID_LENGTH = 256
REWARDS_BY_CASEFOLDED_ID = {
reward_id.casefold(): reward for reward_id, reward in REWARDS_BY_ID.items()
}
def is_mystery_box_history_request(content: bytes) -> bool: def is_mystery_box_history_request(content: bytes) -> bool:
if len(content) < MYSTERY_BOX_HISTORY_REQUEST_LENGTH: if len(content) < MYSTERY_BOX_HISTORY_REQUEST_LENGTH:
return False return False
@ -108,16 +103,15 @@ def _parse_view(data: bytes) -> list[dict[str, Any]]:
timestamp_raw = data[pos : pos + 8] timestamp_raw = data[pos : pos + 8]
pos += 8 pos += 8
ticks, unix_seconds, timestamp_decoded = _decode_timestamp(timestamp_raw) ticks, unix_seconds, timestamp_decoded = _decode_timestamp(timestamp_raw)
reward = REWARDS_BY_ID.get(reward_id) or REWARDS_BY_CASEFOLDED_ID.get( reward = REWARDS_BY_CASEFOLD.get(reward_id.casefold(), {})
reward_id.casefold(), {} canonical_reward_id = reward.get("id", reward_id)
)
rows.append( rows.append(
{ {
"record_start": record_start, "record_start": record_start,
"record_end": pos, "record_end": pos,
"record_len": pos - record_start, "record_len": pos - record_start,
"reward_type": reward.get("type") or infer_reward_type(reward_id), "reward_type": reward.get("type") or infer_reward_type(reward_id),
"reward_id": reward_id, "reward_id": canonical_reward_id,
"reward_name": reward.get("name", ""), "reward_name": reward.get("name", ""),
"reward_rank": reward.get("rank"), "reward_rank": reward.get("rank"),
"quantity": quantity, "quantity": quantity,

View file

@ -15,7 +15,7 @@ from nte_history_exporter.constants import (
TIMESTAMP_TICKS_PER_SECOND, TIMESTAMP_TICKS_PER_SECOND,
VALID_DICE_FIELDS, VALID_DICE_FIELDS,
) )
from nte_history_exporter.mappings import REWARDS_BY_ID from nte_history_exporter.mappings import REWARDS_BY_CASEFOLD
from nte_history_exporter.decoder.structured_protocol import ( from nte_history_exporter.decoder.structured_protocol import (
StructuredRecord, StructuredRecord,
parse_structured_records, parse_structured_records,
@ -61,11 +61,12 @@ def decode_reward_key(raw: bytes) -> str:
def infer_reward_type(reward_id: str) -> str: def infer_reward_type(reward_id: str) -> str:
if not reward_id: if not reward_id:
return "" return ""
if reward_id.startswith("fork_"): folded = reward_id.casefold()
if folded.startswith("fork_"):
return "arc" return "arc"
if reward_id.isdigit(): if reward_id.isdigit():
return "character" return "character"
if reward_id.startswith("Fashion_"): if folded.startswith("fashion_"):
return "cosmetic" return "cosmetic"
return "item" return "item"
@ -165,7 +166,7 @@ def classify_result_type(
) -> tuple[str, int | None]: ) -> tuple[str, int | None]:
if dice is None or dice_offset is None: if dice is None or dice_offset is None:
return "unknown", None return "unknown", None
if reward_id == "Dice_ticket_01" and WARP_PIECE_CHASE_PATTERN in chunk_without_marker: if reward_id.casefold() == "dice_ticket_01" and WARP_PIECE_CHASE_PATTERN in chunk_without_marker:
return "chase_reward", -4 return "chase_reward", -4
if dice == 0: if dice == 0:
return "points_gift", 0 return "points_gift", 0
@ -182,13 +183,14 @@ def classify_result_type(
def guess_quantity(chunk_hex: str, reward_id: str, result_type: str | None = None) -> int | None: def guess_quantity(chunk_hex: str, reward_id: str, result_type: str | None = None) -> int | None:
if reward_id == "Dice_ticket_01": folded = reward_id.casefold()
if folded == "dice_ticket_01":
if result_type == "chase_reward": if result_type == "chase_reward":
return 30 return 30
return 4 return 4
if reward_id == "DiceNormal": if folded == "dicenormal":
return 1 return 1
if reward_id == "Dice_ticket_02": if folded == "dice_ticket_02":
if "c8b0d4c0" in chunk_hex: if "c8b0d4c0" in chunk_hex:
return 50 return 50
if "c8b0ccc0" in chunk_hex: if "c8b0ccc0" in chunk_hex:
@ -266,7 +268,8 @@ def _decode_aligned_response_records(response_content: bytes) -> list[dict[str,
elif result_type == "chase_reward": elif result_type == "chase_reward":
dice = -4 dice = -4
dice_raw = -4 dice_raw = -4
reward = REWARDS_BY_ID.get(reward_id, {}) reward = _reward_metadata(reward_id)
canonical_reward_id = reward.get("id", reward_id)
timestamp_raw = response_content[marker_offset + len(marker) : marker_offset + len(marker) + 8] timestamp_raw = response_content[marker_offset + len(marker) : marker_offset + len(marker) + 8]
timestamp_ticks, timestamp_unix, timestamp_decoded = decode_history_timestamp(timestamp_raw) timestamp_ticks, timestamp_unix, timestamp_decoded = decode_history_timestamp(timestamp_raw)
chunk_hex = chunk.hex() chunk_hex = chunk.hex()
@ -288,7 +291,7 @@ def _decode_aligned_response_records(response_content: bytes) -> list[dict[str,
"dice_offset_in_record": dice_offset, "dice_offset_in_record": dice_offset,
"reward_key_hex": key_hex, "reward_key_hex": key_hex,
"reward_type": reward.get("type") or infer_reward_type(reward_id), "reward_type": reward.get("type") or infer_reward_type(reward_id),
"reward_id": reward_id, "reward_id": canonical_reward_id,
"reward_name": reward.get("name", ""), "reward_name": reward.get("name", ""),
"reward_rank": reward.get("rank"), "reward_rank": reward.get("rank"),
"quantity": guess_quantity(chunk_hex, reward_id, result_type), "quantity": guess_quantity(chunk_hex, reward_id, result_type),
@ -335,8 +338,8 @@ def _enrich_heuristic_rows(
return heuristic_rows return heuristic_rows
for heuristic, structured in zip(heuristic_rows, structured_rows): for heuristic, structured in zip(heuristic_rows, structured_rows):
if not heuristic.get("reward_id"): if not heuristic.get("reward_id"):
heuristic["reward_id"] = structured.item_id
reward = _reward_metadata(structured.item_id) reward = _reward_metadata(structured.item_id)
heuristic["reward_id"] = reward.get("id", structured.item_id)
heuristic["reward_type"] = reward.get("type") or infer_reward_type(structured.item_id) heuristic["reward_type"] = reward.get("type") or infer_reward_type(structured.item_id)
heuristic["reward_name"] = reward.get("name", "") heuristic["reward_name"] = reward.get("name", "")
heuristic["reward_rank"] = reward.get("rank") heuristic["reward_rank"] = reward.get("rank")
@ -361,11 +364,7 @@ def _enrich_heuristic_rows(
def _reward_metadata(reward_id: str) -> dict[str, Any]: def _reward_metadata(reward_id: str) -> dict[str, Any]:
direct = REWARDS_BY_ID.get(reward_id) return REWARDS_BY_CASEFOLD.get(reward_id.casefold(), {})
if direct is not None:
return direct
folded = reward_id.casefold()
return next((meta for item_id, meta in REWARDS_BY_ID.items() if item_id.casefold() == folded), {})
def structured_monopoly_rows(structured_rows: list[StructuredRecord]) -> list[dict[str, Any]]: def structured_monopoly_rows(structured_rows: list[StructuredRecord]) -> list[dict[str, Any]]:
@ -373,6 +372,7 @@ def structured_monopoly_rows(structured_rows: list[StructuredRecord]) -> list[di
for row_index, structured in enumerate(structured_rows, start=1): for row_index, structured in enumerate(structured_rows, start=1):
dice, dice_raw, result_type, result_source = _structured_result(structured.roll_points_raw) dice, dice_raw, result_type, result_source = _structured_result(structured.roll_points_raw)
reward = _reward_metadata(structured.item_id) reward = _reward_metadata(structured.item_id)
canonical_reward_id = reward.get("id", structured.item_id)
rows.append( rows.append(
{ {
"row": row_index, "row": row_index,
@ -391,7 +391,7 @@ def structured_monopoly_rows(structured_rows: list[StructuredRecord]) -> list[di
"dice_offset_in_record": None, "dice_offset_in_record": None,
"reward_key_hex": "", "reward_key_hex": "",
"reward_type": reward.get("type") or infer_reward_type(structured.item_id), "reward_type": reward.get("type") or infer_reward_type(structured.item_id),
"reward_id": structured.item_id, "reward_id": canonical_reward_id,
"reward_name": reward.get("name", ""), "reward_name": reward.get("name", ""),
"reward_rank": reward.get("rank"), "reward_rank": reward.get("rank"),
"quantity": structured.count, "quantity": structured.count,

View file

@ -14,6 +14,9 @@ def load_mapping_file(filename: str) -> dict[str, Any]:
ARC_META: dict[str, dict[str, Any]] = load_mapping_file("arcs.json") ARC_META: dict[str, dict[str, Any]] = load_mapping_file("arcs.json")
ARC_META_BY_CASEFOLD = {
arc_id.casefold(): (arc_id, meta) for arc_id, meta in ARC_META.items()
}
CHARACTERS: dict[str, dict[str, Any]] = load_mapping_file("characters.json") CHARACTERS: dict[str, dict[str, Any]] = load_mapping_file("characters.json")
ITEMS: dict[str, dict[str, Any]] = load_mapping_file("items.json") ITEMS: dict[str, dict[str, Any]] = load_mapping_file("items.json")
ACHIEVEMENTS: dict[str, dict[str, Any]] = load_mapping_file("achievements.json") ACHIEVEMENTS: dict[str, dict[str, Any]] = load_mapping_file("achievements.json")
@ -36,3 +39,6 @@ for _character_id, _meta in CHARACTERS.items():
} }
for _item_id, _meta in ITEMS.items(): for _item_id, _meta in ITEMS.items():
REWARDS_BY_ID[_item_id] = {"type": _meta["type"], "id": _item_id, "name": _meta["name"], "rank": _meta.get("rank")} REWARDS_BY_ID[_item_id] = {"type": _meta["type"], "id": _item_id, "name": _meta["name"], "rank": _meta.get("rank")}
REWARDS_BY_CASEFOLD = {
reward_id.casefold(): reward for reward_id, reward in REWARDS_BY_ID.items()
}

View file

@ -1,8 +1,14 @@
from tests.support import * # noqa: F401,F403 from tests.support import * # noqa: F401,F403
from nte_history_exporter.decoder.arc import annotate_arc_groups from nte_history_exporter.decoder.arc import annotate_arc_groups, _resolve_arc_metadata
class ArcDecodingTests(unittest.TestCase): class ArcDecodingTests(unittest.TestCase):
def test_arc_mapping_lookup_is_case_insensitive_and_canonical(self):
reward_id, metadata = _resolve_arc_metadata("fork_Wushoutieyu")
self.assertEqual(reward_id, "fork_wushoutieyu")
self.assertEqual(metadata, {"name": "Raging Flames", "rank": "S"})
def test_page_split_timestamp_variants_form_one_ten_pull(self): def test_page_split_timestamp_variants_form_one_ten_pull(self):
rows = [ rows = [
{ {

View file

@ -71,6 +71,15 @@ class MysteryBoxDecodingTests(unittest.TestCase):
self.assertEqual(rows[2]["reward_name"], "Beetle Coin") self.assertEqual(rows[2]["reward_name"], "Beetle Coin")
self.assertEqual(rows[2]["reward_rank"], "B") self.assertEqual(rows[2]["reward_rank"], "B")
def test_reward_mapping_lookup_is_case_insensitive_and_canonical(self):
timestamp = datetime(2026, 7, 8, 19, 9, 33, tzinfo=timezone.utc)
rows = parse_mystery_box_response(
mystery_box_response([("Vehicle039", 1, timestamp)])
)
self.assertEqual(rows[0]["reward_id"], "vehicle039")
self.assertEqual(rows[0]["reward_name"], "Draco")
def test_live_session_accepts_partial_final_page(self): def test_live_session_accepts_partial_final_page(self):
session = LiveHistorySession("192.168.0.10") session = LiveHistorySession("192.168.0.10")
request = mystery_box_request(3) request = mystery_box_request(3)

View file

@ -157,6 +157,12 @@ class StructuredProtocolTests(unittest.TestCase):
self.assertEqual(rows[0]["secondary_quantity"], 5) self.assertEqual(rows[0]["secondary_quantity"], 5)
self.assertEqual(rows[0]["structured_pool_id"], "CardPool_Character") self.assertEqual(rows[0]["structured_pool_id"], "CardPool_Character")
def test_structured_reward_mapping_is_case_insensitive_and_canonical(self):
row = decode_response_records(monopoly_payload("DICENORMAL,1"))[0]
self.assertEqual(row["reward_id"], "DiceNormal")
self.assertEqual(row["reward_name"], "Fabricated Dice")
def test_structured_monopoly_parser_falls_back_when_heuristic_returns_no_rows(self): def test_structured_monopoly_parser_falls_back_when_heuristic_returns_no_rows(self):
payload = monopoly_payload("Dice_ticket_02,50", roll_points=0) payload = monopoly_payload("Dice_ticket_02,50", roll_points=0)