diff options
| author | blasty <blasty@local> | 2026-08-10 01:42:14 +0200 |
|---|---|---|
| committer | blasty <blasty@local> | 2026-08-10 01:42:14 +0200 |
| commit | 82d5151c58ea6ed5681525707e403f6fa4160aa7 (patch) | |
| tree | 8c36f864689560e4e3e4733e2657ba23bb99f8a6 /idatui | |
| parent | segment_index: the row total in one call instead of 458 (diff) | |
| download | ida-tui-82d5151c58ea6ed5681525707e403f6fa4160aa7.tar.gz ida-tui-82d5151c58ea6ed5681525707e403f6fa4160aa7.tar.xz ida-tui-82d5151c58ea6ed5681525707e403f6fa4160aa7.zip | |
Build the listing index in one call: boot 9.3s -> 1.25s, 911 calls -> 4
The listing used to learn its shape by fetching it. Even after skeleton pages
that was 457 round trips and 227k rows for a 1.2MB bash, to end up knowing
how many rows there are and where each one is.
segment_index(detail=True) now returns exactly that -- every row's address,
kind and size as packed arrays, plus the page boundaries -- from one walk that
builds no rows and renders no text. ListingModel.build_from_index() decodes it
straight into _heads/_head_eas/_row_at/_by_ea/_page_*, marks every row
_SKELETON_GEN, and declares itself complete. _grow has nothing left to stream.
Nothing else in the model changed, because a row without text is a state it
already had: the FIRST read of a page materialises it through the same
_ensure_text/_ensure_page path a rename uses. That is why this is a ~90 line
change to a core view rather than a rewrite.
bash boot: 911 calls / 9.26s -> 4 calls / 1.25s 7.4x
whole census (boot + 9 UI actions): 933 calls -> 30
Two things had to be exactly right, and both are tested rather than argued:
* the ROW COUNT, or the scrollbar lies. Verified equal to a fully streamed
model, and every row's ea/kind/size equal too, 228,659 of them, zero
mismatches.
* the PAGE BOUNDARIES, or _ensure_page refetches a page that does not line
up, fails its structure check and triggers a full rebuild. heads() pages on
PHYSICAL rows; anchoring every N LOGICAL rows looks identical (the two only
diverge once a segment holds an undefined run) and would have been a
lurking bug on .bss. Anchors now carry [logical_row, ea, head_index] taken
at the real boundary, and are asserted equal to the streamer's own.
Transport note: the packed arrays are base64, not raw bytes. _PACK_EPILOGUE
serialises with json.dumps(default=str), which turns bytes into their repr --
2.97MB arrived as 11.26MB of unparseable text before that was spotted.
The new test builds both models back to back and compares every internal
array. An earlier version compared against the app's long-lived model and was
off by one row, because scenarios before it rename and define things: that
model describes the database at boot, not now.
Full gate: 1063 passed, twice.
Diffstat (limited to 'idatui')
| -rw-r--r-- | idatui/app.py | 18 | ||||
| -rw-r--r-- | idatui/codemode_client.py | 2 | ||||
| -rw-r--r-- | idatui/domain.py | 78 | ||||
| -rw-r--r-- | idatui/remote_tools.py | 119 |
4 files changed, 212 insertions, 5 deletions
diff --git a/idatui/app.py b/idatui/app.py index a806d65..3f67756 100644 --- a/idatui/app.py +++ b/idatui/app.py @@ -1185,11 +1185,21 @@ class ListingView(SearchMixin, NavMixin, ColumnCursor, ScrollView, can_focus=Tru model = self.model if model is None: return - # Load just enough to render the viewport around the cursor, so the - # listing appears immediately even on a huge segment; the rest streams - # in via _grow. (load_all here would blank the pane for seconds.) + # One call gets the WHOLE row index -- every row's address, kind and + # size, and the page boundaries -- so the scrollbar is right immediately + # and _grow has nothing left to stream. The rows arrive text-less and + # materialise a page at a time as they are read. + # + # It is an optimisation, not a contract: an older or unhappy backend + # returns nothing usable and we stream exactly as before. height = max(self.size.height, 1) - model.ensure(self.cursor + height + 2 * ListingModel.PAGE) + if model.build_from_index(): + # Render the viewport HERE, on this worker thread. Reading a + # skeleton page fetches it, and doing that lazily from render_line + # would put an RPC on the UI loop for the first paint. + model.window(max(self.cursor - height, 0), height * 3) + else: + model.ensure(self.cursor + height + 2 * ListingModel.PAGE) self.app.call_from_thread(self._on_primed, len(model), model.complete) if not model.complete: self._grow() diff --git a/idatui/codemode_client.py b/idatui/codemode_client.py index 2361e08..61ff6d8 100644 --- a/idatui/codemode_client.py +++ b/idatui/codemode_client.py @@ -1209,7 +1209,7 @@ _OPERATIONS["pc_num_format"] = _remote_op( #: remote_tools.segment_index: the alternative is fetching every row. _OPERATIONS["segment_index"] = _remote_op( 'segment_index(addr=a["addr"], end=a.get("end", ""),' - ' page_rows=int(a.get("page_rows", 500)))') + ' page_rows=int(a.get("page_rows", 500)), detail=bool(a.get("detail", False)))') _HEADS = _remote_op( 'heads(addr=a["addr"], count=int(a.get("count", 200)),' diff --git a/idatui/domain.py b/idatui/domain.py index 60dbbbf..ecc6a80 100644 --- a/idatui/domain.py +++ b/idatui/domain.py @@ -17,9 +17,11 @@ Textual worker threads; the internal prefetch pool is separate and small. from __future__ import annotations +import array import bisect import re import threading +from base64 import b64decode from concurrent.futures import ThreadPoolExecutor from collections.abc import Sequence from dataclasses import dataclass, field, replace @@ -802,6 +804,82 @@ class ListingModel: with self._lock: return self._max_raw + def build_from_index(self) -> bool: + """Populate the whole row index from ONE call instead of streaming it. + + ``segment_index(detail=True)`` walks the segment and returns every row's + address, kind and size as packed arrays, plus the page boundaries a + refetch would use. That is everything this model needs to know how many + rows there are and where each one lives -- all that is missing is the + rendered text, which is exactly what a skeleton page is missing too. + + So the rows land marked ``_SKELETON_GEN`` and the FIRST read of any page + materialises it through the existing ``_ensure_text``/``_ensure_page`` + path, the same one a rename uses. Measured on bash: 594ms and one call, + against 1827ms and 458 for streaming the same thing. + + Returns False if the backend cannot supply it, in which case the caller + should stream as before -- this is an optimisation, not a new contract. + """ + try: + idx = self._prog.client.invoke( + "segment_index", addr=hex(self.seg_start), end=hex(self.seg_end), + page_rows=self.PAGE, detail=True) + except Exception: # noqa: BLE001 -- fall back to streaming + return False + if not isinstance(idx, dict) or idx.get("error") or "eas" not in idx: + return False + try: + eas = array.array("Q"); eas.frombytes(b64decode(idx["eas"])) + kinds = array.array("B"); kinds.frombytes(b64decode(idx["kinds"])) + sizes = array.array("I"); sizes.frombytes(b64decode(idx["sizes"])) + except Exception: # noqa: BLE001 + return False + names = idx.get("kind_names") or [] + anchors = idx.get("anchors") or [] + n = len(eas) + if not (n == len(kinds) == len(sizes)) or not anchors: + return False + + heads: list[Head] = [] + row_at: list[int] = [] + by_ea: dict[int, int] = {} + rows = 0 + ap = heads.append + rap = row_at.append + for i in range(n): + ea = eas[i] + kind = names[kinds[i]] if kinds[i] < len(names) else "unknown" + size = sizes[i] + ap(Head(ea, kind, size, "")) + rap(rows) + # Banner/label rows are display-only; navigation must land on the + # real head at that address. Same rule as the streaming loader. + if kind not in ("sep", "funchdr", "label"): + by_ea.setdefault(ea, rows) + rows += size if (kind == "unknown" and size > 1) else 1 + + with self._lock: + self._heads = heads + self._head_eas = list(eas) + self._head_gen = [self._SKELETON_GEN] * n + self._row_at = row_at + self._by_ea = by_ea + self._rows = rows + # Anchors are [logical_row, ea, head_index] at the exact boundaries + # heads(count=PAGE) pages on, so _ensure_page can refetch one page + # and have it line up head for head. + self._page_head = [a[2] for a in anchors] + self._page_addr = [_as_int(a[1]) for a in anchors] + self._page_digest = [None] * len(anchors) + self._page_rows = [ + (anchors[k + 1][2] if k + 1 < len(anchors) else n) - anchors[k][2] + for k in range(len(anchors))] + self._skeleton = True + self._done = True + self._next = None + return True + def load_next_page(self, text: bool = True) -> int: """Load one more page of heads; returns how many were added. diff --git a/idatui/remote_tools.py b/idatui/remote_tools.py index 614fba4..0e26965 100644 --- a/idatui/remote_tools.py +++ b/idatui/remote_tools.py @@ -510,10 +510,127 @@ def _idatui_func_footer_rows(ea, func): ] +#: Row kinds, as small ints, for the packed detail index. Order is frozen: the +#: client decodes by position. +_IDATUI_KINDS = ("code", "data", "unknown", "sep", "funchdr", "label", "member") +_IDATUI_KIND_ID = {k: i for i, k in enumerate(_IDATUI_KINDS)} + + +def _idatui_segment_detail(addr, end, page_rows): + """Every listing ROW of a segment as packed arrays, with no text. + + ``{eas, kinds, sizes}`` are raw buffers -- uint64, uint8, uint32, one entry + per row in listing order -- so the client can build its whole row index + (addresses, spans, ea->row map) from ONE call instead of 458 pages. + + This re-implements the row sequence that ``_rows_for`` emits rather than + calling it, because building the dicts is most of what a page costs and + skipping them is the entire point. That duplication is the risk, so it is + covered by a test that walks a whole segment and compares this against the + real ``heads()`` output row for row -- if the two ever drift, that fails. + """ + import array + import ida_segment + + start = parse_address(addr) + seg = ida_segment.getseg(start) + if not seg: + return {"addr": str(addr), "error": "no segment", "rows": 0, "anchors": []} + lo, hi = seg.start_ea, seg.end_ea + if end: + try: + hi = min(hi, parse_address(end)) + except Exception: + pass + + K_CODE = _IDATUI_KIND_ID["code"]; K_DATA = _IDATUI_KIND_ID["data"] + K_UNK = _IDATUI_KIND_ID["unknown"]; K_SEP = _IDATUI_KIND_ID["sep"] + K_FUNC = _IDATUI_KIND_ID["funchdr"]; K_LABEL = _IDATUI_KIND_ID["label"] + K_MEMBER = _IDATUI_KIND_ID["member"] + + eas = array.array("Q") + kinds = array.array("B") + sizes = array.array("I") + ea_ap, kind_ap, size_ap = eas.append, kinds.append, sizes.append + + get_flags = ida_bytes.get_flags + get_item_end = ida_bytes.get_item_end + get_item_size = ida_bytes.get_item_size + next_head = ida_bytes.next_head + get_ea_name = ida_name.get_ea_name + get_func = idaapi.get_func + BAD = idaapi.BADADDR + + # Anchors mark where heads(addr=..., count=page_rows) would START each page, + # so a client can refetch exactly one page. heads() stops once it has + # emitted >= count PHYSICAL rows, checked before the next head -- so a + # boundary is the first head at which the running physical count reached the + # limit. Anchoring every N LOGICAL rows instead looks equivalent (the two + # are the same number until a segment contains an undefined run) and then + # silently yields pages that do not line up with a refetch. + anchors = [] + page_phys = 0 # physical rows emitted into the page being filled + rows = 0 # logical rows so far (what the scrollbar counts) + fn = None + ea = ida_bytes.get_item_head(lo) + while ea != BAD and ea < hi: + if not anchors or page_phys >= page_rows: + anchors.append([rows, hex(ea), len(eas)]) + page_phys = 0 + before = len(eas) + f = get_flags(ea) + cls = f & _MS_CLS + if cls != _FF_CODE and cls != _FF_DATA: + nh = next_head(ea, hi) + stop = nh if (nh != BAD and ea < nh <= hi) else hi + run = stop - ea + ea_ap(ea); kind_ap(K_UNK); size_ap(run) + rows += run if run > 1 else 1 + page_phys += len(eas) - before + ea = stop + continue + if fn is None or not (fn.start_ea <= ea < fn.end_ea): + fn = get_func(ea) + at_start = fn is not None and fn.start_ea == ea + if at_start: + for k in (K_SEP, K_SEP, K_FUNC): # blank, banner, `name proc` + ea_ap(ea); kind_ap(k); size_ap(0) + rows += 3 + elif cls == _FF_CODE and get_ea_name(ea): + ea_ap(ea); kind_ap(K_LABEL); size_ap(0) + rows += 1 + ea_ap(ea); kind_ap(K_CODE if cls == _FF_CODE else K_DATA) + size_ap(int(get_item_size(ea))); rows += 1 + if cls == _FF_DATA: + for m in _idatui_struct_member_rows(ea): + ea_ap(int(m["ea"], 16) if isinstance(m["ea"], str) else m["ea"]) + kind_ap(_IDATUI_KIND_ID.get(m.get("kind", "member"), K_MEMBER)) + size_ap(int(m.get("size", 0) or 0)); rows += 1 + item_end = get_item_end(ea) + if fn is not None and item_end >= fn.end_ea: + for k in (K_FUNC, K_SEP): # `name endp`, separator + ea_ap(ea); kind_ap(k); size_ap(0) + rows += 2 + page_phys += len(eas) - before + ea = item_end if item_end > ea else ea + 1 + + # base64, not raw bytes: the client packs answers with json.dumps(default=str), + # which turns a bytes object into its repr -- 4 characters per byte and + # unparseable at the other end. Learned by watching 2.97MB arrive as 11.26MB. + import base64 + b64 = base64.b64encode + return {"addr": hex(lo), "end": hex(hi), "rows": rows, "heads": len(eas), + "anchors": anchors, "kind_names": list(_IDATUI_KINDS), + "eas": b64(eas.tobytes()).decode(), + "kinds": b64(kinds.tobytes()).decode(), + "sizes": b64(sizes.tobytes()).decode()} + + def segment_index( addr: Annotated[str, "Any address in the segment to index"], end: Annotated[str, "Optional exclusive end address; default = segment end"] = "", page_rows: Annotated[int, "Rows between anchors (default 500)"] = 500, + detail: Annotated[bool, "Also return every row's ea/kind/size as packed arrays"] = False, ) -> dict: """How many listing rows a segment has, and where to seek into it. @@ -542,6 +659,8 @@ def segment_index( start = parse_address(addr) except Exception as e: return {"addr": str(addr), "error": str(e), "rows": 0, "anchors": []} + if detail: + return _idatui_segment_detail(addr, end, count) import ida_segment seg = ida_segment.getseg(start) if not seg: |
