Stabilize UID generation and page-first dice parsing

This commit is contained in:
Golumpa 2026-06-28 16:23:02 +01:00
parent 241a15e584
commit 98b2165b84
6 changed files with 82 additions and 23 deletions

View file

@ -92,18 +92,22 @@ nice names or ranks.
## UID generation ## UID generation
The `uid` is the first 32 hex characters of `sha256(source)`. We generate our own roll UID as the game does not send their own, so to make things trackable and to help prevent duplicates we create our own UID with a selection of feilds making each entry in the history 100% unique. The way it is done means on a rescan the same UID is generated for the same history item even if it is further in the history and you have pulled more since the last scan. The `uid` is the first 32 hex characters of `sha256(source)`. The game does not
appear to send a stable pull/reward row ID, so the exporter builds one from the
history pool, raw packet timestamp, and ordinal inside that timestamp group.
Decoded content such as dice result, reward ID, and quantity is intentionally
excluded so decoder fixes do not change the identity of an already-captured row.
Monopoly source: Monopoly source:
```text ```text
nte|monopoly|pool_group_id|timestamp_raw|timestamp_group_ordinal|roll_result|reward_key_hex|quantity nte|monopoly|pool_group_id|timestamp_raw|timestamp_group_ordinal
``` ```
Arc source: Arc source:
```text ```text
nte|gashapon|pool_group_id|timestamp_raw|timestamp_group_ordinal|reward_key_hex nte|gashapon|pool_group_id|timestamp_raw|timestamp_group_ordinal
``` ```
## Example records ## Example records

View file

@ -25,6 +25,9 @@ Decoded fields:
- `roll_result = first u32 / 4` - `roll_result = first u32 / 4`
- `roll_result = 0` means Points Gift - `roll_result = 0` means Points Gift
- Some page-first records include a one-byte page prefix and an extra `0x14`
field before the real dice u32. In that shape, the real dice u32 is at offset
9 within the record chunk, not the earlier `0x14` field.
- Some page-first records have a short prefix before the record body. For these, a hidden signed source flag immediately after the visible dice field overrides the visible dice: - Some page-first records have a short prefix before the record body. For these, a hidden signed source flag immediately after the visible dice field overrides the visible dice:
- `source_flag = 0` means Points Gift - `source_flag = 0` means Points Gift
- `source_flag = -4` means Chase Reward - `source_flag = -4` means Chase Reward

View file

@ -146,8 +146,8 @@ def build_arc_rows_from_pairs(pairs: list[tuple]) -> list[dict[str, Any]]:
return rows return rows
def make_arc_uid(timestamp_raw: str, ordinal: int, arc_key_hex: str) -> str: def make_arc_uid(timestamp_raw: str, ordinal: int) -> str:
source = "|".join([GAME_UID_PART, ARC_SYSTEM, ARC_BANNER_ID, timestamp_raw, str(ordinal), arc_key_hex]) source = "|".join([GAME_UID_PART, ARC_SYSTEM, ARC_BANNER_ID, timestamp_raw, str(ordinal)])
return hashlib.sha256(source.encode("utf-8")).hexdigest()[:32] return hashlib.sha256(source.encode("utf-8")).hexdigest()[:32]
@ -165,7 +165,7 @@ def annotate_arc_groups(rows: list[dict[str, Any]]) -> None:
row["timestamp_group_index"] = group_index row["timestamp_group_index"] = group_index
row["timestamp_group_ordinal"] = ordinal row["timestamp_group_ordinal"] = ordinal
row["timestamp_group_size_seen"] = len(indexes) row["timestamp_group_size_seen"] = len(indexes)
row["uid"] = make_arc_uid(timestamp_raw, ordinal, row["reward_key_hex"]) row["uid"] = make_arc_uid(timestamp_raw, ordinal)
row["uid_status"] = "stable" row["uid_status"] = "stable"
row["export_record"] = True row["export_record"] = True
row["skip_reason"] = "" row["skip_reason"] = ""

View file

@ -14,9 +14,6 @@ def make_uid(record: dict[str, Any], ordinal: int) -> str:
str(record.get("pool_group_id", BANNER_ID)), str(record.get("pool_group_id", BANNER_ID)),
str(record.get("timestamp_raw_hex", "")), str(record.get("timestamp_raw_hex", "")),
str(ordinal), str(ordinal),
str(record.get("dice", "")),
str(record.get("reward_key_hex", "")),
str(record.get("quantity", "")),
] ]
) )
return hashlib.sha256(source.encode("utf-8")).hexdigest()[:32] return hashlib.sha256(source.encode("utf-8")).hexdigest()[:32]

View file

@ -132,7 +132,20 @@ def extract_key(chunk_without_marker: bytes) -> str:
return "" if best is None else chunk_without_marker[best:].hex() return "" if best is None else chunk_without_marker[best:].hex()
def _page_first_prefixed_dice_raw(chunk_without_marker: bytes) -> int | None:
if len(chunk_without_marker) >= 17 and chunk_without_marker[0] == 0:
prefix_field = struct.unpack_from("<I", chunk_without_marker, 5)[0]
prefixed_dice_raw = struct.unpack_from("<I", chunk_without_marker, 9)[0]
if prefix_field == 20 and prefixed_dice_raw in VALID_DICE_FIELDS:
return prefixed_dice_raw
return None
def extract_dice(chunk_without_marker: bytes) -> tuple[int | None, int | None, int | None]: def extract_dice(chunk_without_marker: bytes) -> tuple[int | None, int | None, int | None]:
prefixed_dice_raw = _page_first_prefixed_dice_raw(chunk_without_marker)
if prefixed_dice_raw is not None:
return (0 if prefixed_dice_raw == 0 else prefixed_dice_raw // 4), prefixed_dice_raw, 9
for off in range(0, min(16, max(0, len(chunk_without_marker) - 3))): for off in range(0, min(16, max(0, len(chunk_without_marker) - 3))):
val = struct.unpack_from("<I", chunk_without_marker, off)[0] val = struct.unpack_from("<I", chunk_without_marker, off)[0]
if val in VALID_DICE_FIELDS: if val in VALID_DICE_FIELDS:
@ -201,16 +214,23 @@ def _decode_aligned_response_records(response_content: bytes) -> list[dict[str,
chunk = response_content[prev:marker_offset] chunk = response_content[prev:marker_offset]
full_record = response_content[prev : marker_offset + len(marker) + 8] full_record = response_content[prev : marker_offset + len(marker) + 8]
dice, dice_raw, dice_offset = extract_dice(chunk) dice, dice_raw, dice_offset = extract_dice(chunk)
if dice is None and len(chunk) > 32:
embedded_candidates = []
original_key = extract_key(chunk) original_key = extract_key(chunk)
original_key_bytes = bytes.fromhex(original_key) if original_key else b"" original_key_bytes = bytes.fromhex(original_key) if original_key else b""
key_position = chunk.find(original_key_bytes) if original_key_bytes else len(chunk)
should_try_embedded_trim = (
dice is None
or (_page_first_prefixed_dice_raw(chunk) is None and key_position > 32)
)
if should_try_embedded_trim and len(chunk) > 32:
embedded_candidates = []
for trim in range(1, min(96, len(chunk))): for trim in range(1, min(96, len(chunk))):
candidate = chunk[trim:] candidate = chunk[trim:]
candidate_dice, candidate_raw, candidate_offset = extract_dice(candidate) candidate_dice, candidate_raw, candidate_offset = extract_dice(candidate)
candidate_is_page_first = _page_first_prefixed_dice_raw(candidate) is not None
if ( if (
candidate_dice is not None candidate_dice is not None
and candidate_offset in (0, 5) and candidate_offset in (0, 5, 9)
and (dice is None or candidate_is_page_first)
and extract_key(candidate) == original_key and extract_key(candidate) == original_key
): ):
key_count = candidate.count(original_key_bytes) if original_key_bytes else 0 key_count = candidate.count(original_key_bytes) if original_key_bytes else 0
@ -218,6 +238,7 @@ def _decode_aligned_response_records(response_content: bytes) -> list[dict[str,
embedded_candidates.append( embedded_candidates.append(
( (
-key_count, -key_count,
not candidate_is_page_first,
candidate_offset != 5, candidate_offset != 5,
key_position, key_position,
-trim, -trim,
@ -229,7 +250,7 @@ def _decode_aligned_response_records(response_content: bytes) -> list[dict[str,
) )
) )
if embedded_candidates: if embedded_candidates:
_, _, _, _, trim, chunk, dice, dice_raw, dice_offset = min(embedded_candidates) _, _, _, _, _, trim, chunk, dice, dice_raw, dice_offset = min(embedded_candidates)
record_start = prev + trim record_start = prev + trim
full_record = response_content[prev + trim : marker_offset + len(marker) + 8] full_record = response_content[prev + trim : marker_offset + len(marker) + 8]
key_hex = extract_key(chunk) key_hex = extract_key(chunk)

View file

@ -300,9 +300,9 @@ class BoundaryExportTests(unittest.TestCase):
def test_uid_source_matches_v4_reference(self): def test_uid_source_matches_v4_reference(self):
rows = load_reference_csv("monopoly_history_poc_10_all_44_pages_v4.csv") rows = load_reference_csv("monopoly_history_poc_10_all_44_pages_v4.csv")
first = rows[0] first = rows[0]
self.assertEqual(make_uid(first, 0), "5adcf52282e15445466863b271f3b745") self.assertEqual(make_uid(first, 0), "f2c72f0a80b79216bf15661521620693")
def test_uid_uses_detected_pool_group_id(self): def test_uid_uses_pool_timestamp_and_ordinal_only(self):
row = { row = {
"pool_group_id": "Lottery_LimitedCharacter", "pool_group_id": "Lottery_LimitedCharacter",
"timestamp_raw_hex": "40e93247c3097b23", "timestamp_raw_hex": "40e93247c3097b23",
@ -310,7 +310,18 @@ class BoundaryExportTests(unittest.TestCase):
"reward_key_hex": "10a58d957dd1a58dad95d17dc1c800", "reward_key_hex": "10a58d957dd1a58dad95d17dc1c800",
"quantity": 50, "quantity": 50,
} }
self.assertNotEqual(make_uid(row, 0), "5adcf52282e15445466863b271f3b745") changed_content = {
**row,
"dice": 1,
"reward_key_hex": "98bdc9ad7dd9a5b99501",
"quantity": 1,
}
changed_pool = {**row, "pool_group_id": "Lottery_Permanent"}
self.assertEqual(make_uid(row, 0), "74a9ef4aacde549dfe8e8e7cc6ddd65b")
self.assertEqual(make_uid(changed_content, 0), make_uid(row, 0))
self.assertNotEqual(make_uid(changed_pool, 0), make_uid(row, 0))
self.assertNotEqual(make_uid(row, 1), make_uid(row, 0))
def test_pages_1_to_5_exports_every_row(self): def test_pages_1_to_5_exports_every_row(self):
rows = load_reference_csv("monopoly_history_poc_13_pages_1_to_5_v4.csv") rows = load_reference_csv("monopoly_history_poc_13_pages_1_to_5_v4.csv")
@ -592,7 +603,7 @@ class BoundaryExportTests(unittest.TestCase):
self.assertEqual(decoded["dice_raw_u32"], -4) self.assertEqual(decoded["dice_raw_u32"], -4)
self.assertEqual(decoded["reward_id"], "Dice_ticket_01") self.assertEqual(decoded["reward_id"], "Dice_ticket_01")
self.assertEqual(decoded["quantity"], 30) self.assertEqual(decoded["quantity"], 30)
self.assertEqual(make_uid(decoded, int(reference["timestamp_group_ordinal"])), "7d035ec098f856f81b403ea538810145") self.assertEqual(make_uid(decoded, int(reference["timestamp_group_ordinal"])), "23dc293f39f18e94a81ccaaf7e1a67eb")
def test_warp_piece_chase_subrecord_without_prefix_marker_is_chase_reward(self): def test_warp_piece_chase_subrecord_without_prefix_marker_is_chase_reward(self):
decoded = decode_single_record( decoded = decode_single_record(
@ -608,6 +619,32 @@ class BoundaryExportTests(unittest.TestCase):
self.assertEqual(decoded["reward_name"], "Warp Piece") self.assertEqual(decoded["reward_name"], "Warp Piece")
self.assertEqual(decoded["quantity"], 30) self.assertEqual(decoded["quantity"], 30)
def test_page_first_prefix_uses_real_dice_field(self):
cases = [
(
"003006000014000000040000002800000098bdc9ad7dd9a5b995010000000008000000"
"3c00000010a58d957dd1a58dad95d17dc1c4002800000098bdc9ad7dd9a5b995014c"
"0000000c85c99141bdbdb17d0da185c9858dd195c901c0dd53bd2b137b23",
1,
"fork_vine",
),
(
"00c8060000140000001000000014000000c4c0d4d400000000000400000014000000"
"c4c0d4d400440000000c85c99141bdbdb17d3995dd49bdb1950100d929e115087b23",
4,
"1055",
),
]
for record_hex, expected_dice, reward_id in cases:
with self.subTest(reward_id=reward_id):
decoded = decode_single_record(record_hex)
self.assertEqual(decoded["dice"], expected_dice)
self.assertEqual(decoded["dice_raw_u32"], expected_dice * 4)
self.assertEqual(decoded["dice_offset_in_record"], 9)
self.assertEqual(decoded["result_type"], "dice")
self.assertEqual(decoded["reward_id"], reward_id)
def test_batched_monopoly_response_normalizes_embedded_page_header(self): def test_batched_monopoly_response_normalizes_embedded_page_header(self):
page_7 = [load_v7_row("limited_all_04_v7.csv", row) for row in range(31, 36)] page_7 = [load_v7_row("limited_all_04_v7.csv", row) for row in range(31, 36)]
page_8 = [load_v7_row("limited_all_04_v7.csv", row) for row in range(36, 41)] page_8 = [load_v7_row("limited_all_04_v7.csv", row) for row in range(36, 41)]
@ -626,7 +663,7 @@ class BoundaryExportTests(unittest.TestCase):
self.assertEqual(len(decoded), 10) self.assertEqual(len(decoded), 10)
self.assertEqual(decoded[5]["reward_id"], page_8[0]["reward_id"]) self.assertEqual(decoded[5]["reward_id"], page_8[0]["reward_id"])
self.assertEqual(decoded[5]["dice"], 5) self.assertEqual(decoded[5]["dice"], 4)
self.assertEqual(decoded[5]["result_type"], "dice") self.assertEqual(decoded[5]["result_type"], "dice")
self.assertEqual(decoded[5]["record_hex"], page_8[0]["record_hex"]) self.assertEqual(decoded[5]["record_hex"], page_8[0]["record_hex"])
@ -665,10 +702,7 @@ class BoundaryExportTests(unittest.TestCase):
self.assertEqual(decode_arc_key(bytes.fromhex(row["arc_key_hex"])), "fork_nonos") self.assertEqual(decode_arc_key(bytes.fromhex(row["arc_key_hex"])), "fork_nonos")
_ticks, _unix, decoded = decode_arc_timestamp(bytes.fromhex(row["timestamp_raw_hex"])) _ticks, _unix, decoded = decode_arc_timestamp(bytes.fromhex(row["timestamp_raw_hex"]))
self.assertEqual(decoded, "2026-06-10 23:46:29") self.assertEqual(decoded, "2026-06-10 23:46:29")
self.assertEqual( self.assertEqual(make_arc_uid(row["timestamp_raw_hex"], int(row["timestamp_group_ordinal"])), "4435d9729fa8fd0eaf1b1ad7aa4d2172")
make_arc_uid(row["timestamp_raw_hex"], int(row["timestamp_group_ordinal"]), row["arc_key_hex"]),
row["uid"],
)
def test_arc_response_parser_matches_reference_first_page(self): def test_arc_response_parser_matches_reference_first_page(self):
reference_rows = load_arc_csv("arc_pull_10_all_pages_v2.csv")[:5] reference_rows = load_arc_csv("arc_pull_10_all_pages_v2.csv")[:5]