Anchor scans to page 1; export stable prefixes

Anchor page-run selection to start at page 1 (select_continuous_run_from_page_1) instead of picking the longest run, so newest pages are preserved and the newest timestamp ordinal 0 is always captured. Change boundary logic to always assign stable UIDs and export the captured prefix of an oldest timestamp group even when it may be an unfinished 10-pull, emitting informational warnings (INCOMPLETE_TIMESTAMP_GROUP_EXPORTED / INCOMPLETE_ARC_10_PULL_EXPORTED) instead of dropping rows. Apply the same policy to Arc groups and update arc group annotation and warnings accordingly. Add a new console module for improved CLI output and update cli, live_capture, adapters, and session code to use the new run selection and console helpers. Update docs, mappings, and tests to reflect the new boundary/export behavior and messaging.
This commit is contained in:
Golumpa 2026-06-11 12:22:34 +01:00
parent 0a1a19939b
commit 20b2086299
14 changed files with 402 additions and 200 deletions

View file

@ -19,7 +19,7 @@ from nte_history_exporter.constants import (
GAME_UID_PART,
POOL_META,
)
from nte_history_exporter.decoder.boundary import longest_monotonic_page_run
from nte_history_exporter.decoder.boundary import select_continuous_run_from_page_1
from nte_history_exporter.decoder.run import fmt_packet_time
from nte_history_exporter.mappings import ARC_META
@ -145,24 +145,36 @@ def annotate_arc_groups(rows: list[dict[str, Any]]) -> None:
for index, row in enumerate(rows):
groups[row["timestamp_raw_hex"]].append(index)
for group_index, (timestamp_raw, indexes) in enumerate(groups.items()):
complete = len(indexes) % 10 == 0
group_items = list(groups.items())
for group_index, (timestamp_raw, indexes) in enumerate(group_items):
at_oldest_boundary = group_index == len(group_items) - 1
# Arc pulls are always 10-pulls, so the only group that can be short is
# the oldest one, where the scan stopped mid-10-pull. Its captured prefix
# is ordinal-stable (ordinal 0 is captured, unseen rows append after it),
# so export it and flag it rather than dropping it.
incomplete = at_oldest_boundary and len(indexes) % 10 != 0
skip_reason = (
"arc timestamp group is not a complete 10-pull in this capture; "
"exported rows are stable, scroll further to capture the rest"
if incomplete
else ""
)
for ordinal, index in enumerate(indexes):
row = rows[index]
row["timestamp_group_index"] = group_index
row["timestamp_group_ordinal"] = ordinal
row["timestamp_group_size_seen"] = len(indexes)
row["uid"] = make_arc_uid(timestamp_raw, ordinal, row["reward_key_hex"])
row["uid_status"] = "stable" if complete else "skipped_incomplete_timestamp_group"
row["export_record"] = complete
row["skip_reason"] = "" if complete else "arc timestamp group is not a complete 10-pull in this capture"
row["uid_status"] = "incomplete_stable" if incomplete else "stable"
row["export_record"] = True
row["skip_reason"] = skip_reason
def arc_stability_warnings(rows: list[dict[str, Any]]) -> list[dict[str, Any]]:
warnings = []
seen = set()
for row in rows:
if row.get("export_record") is True:
if row.get("uid_status") != "incomplete_stable":
continue
timestamp_raw = row["timestamp_raw_hex"]
if timestamp_raw in seen:
@ -171,38 +183,18 @@ def arc_stability_warnings(rows: list[dict[str, Any]]) -> list[dict[str, Any]]:
group = [r for r in rows if r["timestamp_raw_hex"] == timestamp_raw]
warnings.append(
{
"code": "INCOMPLETE_ARC_10_PULL_DROPPED",
"code": "INCOMPLETE_ARC_10_PULL_EXPORTED",
"timestamp_raw": timestamp_raw,
"timestamp_decoded": row["timestamp_decoded"],
"records": len(group),
"reason": "arc timestamp group is not a complete 10-pull in this capture",
"reason": (
"arc timestamp group is not a complete 10-pull in this capture; "
"exported rows are stable, scroll further to capture the rest"
),
}
)
return warnings
def select_continuous_arc_run(pairs: list[tuple]) -> tuple[list[tuple], list[dict[str, Any]]]:
warnings: list[dict[str, Any]] = []
if not pairs:
return [], warnings
pairs_by_page = {pair[0]: pair for pair in pairs}
seen_pages = sorted(pairs_by_page)
if 1 in pairs_by_page:
selected_pages = []
page = 1
while page in pairs_by_page:
selected_pages.append(page)
page += 1
if len(selected_pages) < len(seen_pages):
ignored = [page for page in seen_pages if page not in selected_pages]
warnings.append(
{
"code": "PAGE_GAP_DETECTED",
"ignored_pages": ignored,
"reason": f"Using continuous pages 1-{selected_pages[-1]}; ignored later pages {ignored}.",
}
)
return [pairs_by_page[page] for page in selected_pages], warnings
warnings.append({"code": "DID_NOT_START_AT_PAGE_1", "reason": "Arc history scan did not start at page 1."})
return longest_monotonic_page_run(pairs), warnings
return select_continuous_run_from_page_1(pairs)

View file

@ -40,31 +40,50 @@ def longest_monotonic_page_run(pairs: list[tuple]) -> list[tuple]:
return max(runs, key=len) if runs else []
def page_gap_warnings(pairs: list[tuple], best_run: list[tuple]) -> list[dict[str, Any]]:
warnings: list[dict[str, Any]] = []
if len(pairs) < 2:
return warnings
def select_continuous_run_from_page_1(pairs: list[tuple]) -> tuple[list[tuple], list[dict[str, Any]]]:
"""Pick the run of pages starting at page 1 and continuing without gaps.
best_run_ids = {id(pair) for pair in best_run}
ignored_pages = [pair[0] for pair in pairs if id(pair) not in best_run_ids]
previous_page = pairs[0][0]
for pair in pairs[1:]:
page = pair[0]
if page != previous_page + 1:
warning = {
"code": "PAGE_GAP_DETECTED",
"previous_page": previous_page,
"next_page": page,
"ignored_pages": ignored_pages,
"reason": (
f"Page gap detected: saw page {previous_page} then page {page}. "
"Pages outside the longest continuous run were ignored for stable dedupe. "
"Re-scan or scroll more slowly."
),
}
warnings.append(warning)
previous_page = page
return warnings
History always loads page 1 first and is scrolled downward, so the page-1
run holds the newest, contiguous history. Anchoring here (instead of the
longest run anywhere) keeps the newest pages even when a later packet is
lost, and guarantees the newest timestamp group's ordinal 0 is captured so
every exported UID is stable. Pages after the first gap are ignored with a
warning. If page 1 itself was not captured we fall back to the longest run
and warn that the result may be unstable.
"""
warnings: list[dict[str, Any]] = []
if not pairs:
return [], warnings
pairs_by_page = {pair[0]: pair for pair in pairs}
seen_pages = sorted(pairs_by_page)
if 1 in pairs_by_page:
selected_pages: list[int] = []
page = 1
while page in pairs_by_page:
selected_pages.append(page)
page += 1
if len(selected_pages) < len(seen_pages):
ignored = [p for p in seen_pages if p not in selected_pages]
warnings.append(
{
"code": "PAGE_GAP_DETECTED",
"ignored_pages": ignored,
"reason": (
f"Page gap detected after page {selected_pages[-1]}; "
f"ignored later pages {ignored}. Re-scan or scroll more slowly."
),
}
)
return [pairs_by_page[p] for p in selected_pages], warnings
warnings.append(
{
"code": "DID_NOT_START_AT_PAGE_1",
"reason": "Page 1 was not captured; results may be unstable. Re-scan from the top.",
}
)
return longest_monotonic_page_run(pairs), warnings
def is_dice_record(row: dict[str, Any]) -> bool:
@ -81,12 +100,17 @@ def is_dice_record(row: dict[str, Any]) -> bool:
return False
def annotate_groups(
rows: list[dict[str, Any]],
*,
starts_from_page_1: bool = True,
stable_only: bool = True,
) -> tuple[list[dict[str, Any]], list[dict[str, Any]]]:
def annotate_groups(rows: list[dict[str, Any]]) -> tuple[list[dict[str, Any]], list[dict[str, Any]]]:
"""Assign timestamp-group ordinals and stable UIDs to every decoded row.
Pages are anchored at page 1 (see select_continuous_run_from_page_1), so the
newest group's ordinal 0 is always captured and every UID is stable. The only
nuance is the oldest group: if the capture did not reach the true end of
history (final page full) and the dice-only count is not a complete multiple
of 10, it may be an unfinished 10-pull continuing onto an uncaptured page. Its
captured prefix is still ordinal-stable, so it is exported with an
informational warning rather than withheld.
"""
if not rows:
return rows, []
@ -110,28 +134,21 @@ def annotate_groups(
warnings: list[dict[str, Any]] = []
for group_index, group in enumerate(groups):
at_newest_boundary = group_index == 0
at_oldest_boundary = group_index == len(groups) - 1
dice_records_in_group = [row for row in group if is_dice_record(row)]
dice_record_count = len(dice_records_in_group)
# A complete multiple of 10 dice rolls proves the pull set is finished.
# Unseen continuation rows can only be non-dice tails that sort after the
# seen rows, so exported ordinals (and UIDs) stay stable.
dice_record_count = sum(1 for row in group if is_dice_record(row))
dice_complete = dice_record_count > 0 and dice_record_count % 10 == 0
group_status = "stable"
skip_reason = ""
incomplete = at_oldest_boundary and not final_page_is_partial and not dice_complete
skip_reason = (
"oldest timestamp group may continue onto the next uncaptured page; "
"exported rows are stable, scroll further to capture the rest"
if incomplete
else ""
)
if at_newest_boundary and not starts_from_page_1:
group_status = "dropped_boundary_group"
skip_reason = "newest timestamp group may be partial because scan did not start from page 1"
elif at_oldest_boundary and not final_page_is_partial and not dice_complete:
group_status = "dropped_boundary_group"
skip_reason = "oldest timestamp group may continue onto the next uncaptured page"
if group_status != "stable":
if incomplete:
warnings.append(
{
"code": "PARTIAL_TIMESTAMP_GROUP_DROPPED",
"code": "INCOMPLETE_TIMESTAMP_GROUP_EXPORTED",
"timestamp_raw": group[0].get("timestamp_raw_hex", ""),
"timestamp_decoded": group[0].get("timestamp_decoded", ""),
"records": len(group),
@ -145,19 +162,9 @@ def annotate_groups(
row["timestamp_group_ordinal"] = ordinal
row["timestamp_group_size_seen"] = dice_record_count
row["timestamp_group_record_size_seen"] = len(group)
row["timestamp_group_boundary"] = ",".join(
name
for name, yes in [("newest", at_newest_boundary), ("oldest", at_oldest_boundary)]
if yes
)
if group_status == "stable":
row["uid_status"] = "stable"
row["uid"] = make_uid(row, ordinal)
row["export_record"] = True
row["skip_reason"] = ""
else:
row["uid_status"] = group_status
row["uid"] = "" if stable_only else make_uid(row, ordinal)
row["export_record"] = not stable_only
row["skip_reason"] = skip_reason
row["timestamp_group_boundary"] = "oldest" if at_oldest_boundary else ("newest" if group_index == 0 else "")
row["uid_status"] = "incomplete_stable" if incomplete else "stable"
row["uid"] = make_uid(row, ordinal)
row["export_record"] = True
row["skip_reason"] = skip_reason
return rows, warnings