Files
DeeJanuz--frametop/hands/tools/show_set.py
T
DeeJanuzandClaude Opus 5.5 499035216c Make hand tracking a Frametop component
- Programs: ft-camd (the camera broker), ft-hands (the tracker), and
  ft-handreplay and ft-ringplay for recordings, built by hands/build.sh
  into hands/build/ with one Makefile. The first build fetches ncnn at
  frame-hands' pinned tag and builds it with the same options.
- ft-camd gets its privileges from file capabilities (CAP_SYS_PTRACE,
  CAP_PERFMON, CAP_DAC_READ_SEARCH) that hands/run.sh install sets with
  sudo, and drops them once set up. It still works under sudo. It runs
  on the host, linked statically, as frametop-camd.service. ft-hands
  runs in the dev container as frametop-hands.service. Both start and
  stop with SteamVR.
- Files move to /run/user/UID/frametop/ (cam-ring, hands, gestures),
  not $XDG_RUNTIME_DIR, which a terminal in the Frametop desktop has
  its own of. SIGUSR1 recordings go to ~/.local/share/frametop/hands.
- The calibration is read through /run/host in the container.
- Settings: HANDS_SWAP_SIDES and HANDS_CPUS in frametop.conf.
- install.sh offers hand tracking as an optional last step.
- The container gets jsoncpp-devel, glibc-static, and NumPy and OpenCV
  for the Python tools.
- tools/ring.py reads the ring, and models/NOTICE credits the
  Apache-2.0 models.

Checked: ft-handreplay gives identical summaries and byte-identical
depth dumps to frame-hands' fh-replay on both 2026-09-29 recordings.

Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com>
2026-09-30 09:15:07 -06:00

130 lines
4.8 KiB
Python

"""Draw frame sets from a recording (ft-hands --record) with what the tracker saw.
usage: python tools/show_set.py REC_DIR SET [SET...] [--timeline TL] [--out DIR]
SET is a set index (ft-handreplay's timeline gives them). With --timeline (ft-handreplay
--timeline), each camera shows the tracker's views at that set: the crop for the next
frame, labelled with the hand and presence. Recordings made with ft-camd --with-dark get
a second row: each camera's latest dark frame (<name>_dk), stretched to be visible and
labelled with its mean brightness. Recordings made with ft-camd --with-color get a row of
the color cameras (color_video<N>). Writes OUT/set_<n>.jpg (default /tmp).
"""
import argparse
import os
import struct
import cv2
import numpy as np
HDR = struct.Struct('<8sII')
CAM = struct.Struct('<16sIIQQ')
ORDER = ['slam_left', 'slam_right', 'upper_left', 'upper_right']
def index(path):
"""Byte offset of every set in sets.bin."""
offs, size = [], os.path.getsize(path)
with open(path, 'rb') as f:
off = 0
while off + HDR.size <= size:
f.seek(off)
magic, n, nbytes = HDR.unpack(f.read(HDR.size))
if magic[:7] != b'FHSET01' or off + nbytes > size:
break
offs.append(off)
off += nbytes
return offs
def read_set(path, off):
with open(path, 'rb') as f:
f.seek(off)
_, n, _ = HDR.unpack(f.read(HDR.size))
cams = [CAM.unpack(f.read(CAM.size)) for _ in range(n)]
out = {}
for name, w, h, cap, dq in cams:
px = np.frombuffer(f.read(w * h), np.uint8).reshape(h, w)
out[name.rstrip(b'\0').decode()] = (px, cap)
return out
def views_at(timeline, n):
out = []
for line in open(timeline):
f = line.split()
if len(f) > 1 and f[1] == 'view' and int(f[-1]) == n:
out.append({'hand': int(f[2]), 'cam': f[3], 'presence': float(f[5]),
'c': (float(f[7]), float(f[8])), 'size': float(f[9]), 'rot': float(f[10])})
return out
def dark_tile(frame, shape, name):
"""A dark frame, stretched from its 1st to 99.5th percentile; black if there's none."""
h, w = shape
if frame is None:
return np.zeros((h, w, 3), np.uint8)
px = frame[0]
lo, hi = np.percentile(px, (1, 99.5))
gain = 255 / max(hi - lo, 1)
img = np.clip((px.astype(np.float32) - lo) * gain, 0, 255).astype(np.uint8)
img = cv2.cvtColor(cv2.resize(img, (w, h)), cv2.COLOR_GRAY2BGR)
cv2.putText(img, '%s_dk mean %.1f, x%.0f' % (name, px.mean(), gain), (10, 30), cv2.FONT_HERSHEY_SIMPLEX, 1.0,
(255, 255, 0), 2)
return img
def view_tile(px, name, views):
"""A frame, CLAHE'd, with the tracker's views on it, 512 px high."""
img = cv2.cvtColor(cv2.createCLAHE(2.0, (8, 8)).apply(px), cv2.COLOR_GRAY2BGR)
for v in views:
if v['cam'] != name:
continue
c, s, r = v['c'], v['size'], v['rot']
box = cv2.boxPoints(((c[0], c[1]), (s, s), np.degrees(r)))
col = (0, 255, 0) if v['presence'] >= 0.5 else (0, 0, 255)
cv2.polylines(img, [box.astype(np.int32)], True, col, 2)
cv2.putText(img, 'h%d %.2f' % (v['hand'], v['presence']), (int(c[0] - s / 2), int(c[1] - s / 2) - 6),
cv2.FONT_HERSHEY_SIMPLEX, 0.8, col, 2)
cv2.putText(img, name, (10, 30), cv2.FONT_HERSHEY_SIMPLEX, 1.0, (255, 255, 0), 2)
scale = 512 / img.shape[0]
return cv2.resize(img, (int(img.shape[1] * scale), 512))
def draw(images, views):
"""Rows: the mono cameras; their dark frames, if recorded; the color cameras, if recorded."""
tiles, dark = [], []
for name in ORDER:
if name not in images:
continue
tiles.append(view_tile(images[name][0], name, views))
dark.append(dark_tile(images.get(name + '_dk'), tiles[-1].shape[:2], name))
rows = [np.hstack(tiles)]
if any(k.endswith('_dk') for k in images):
rows.append(np.hstack(dark))
color = sorted(k for k in images if k.startswith('color_'))
if color:
rows.append(np.hstack([view_tile(images[k][0], k, views) for k in color]))
width = max(r.shape[1] for r in rows)
return np.vstack([np.pad(r, ((0, 0), (0, width - r.shape[1]), (0, 0))) for r in rows])
def main():
ap = argparse.ArgumentParser()
ap.add_argument('rec')
ap.add_argument('sets', type=int, nargs='+')
ap.add_argument('--timeline')
ap.add_argument('--out', default='/tmp')
a = ap.parse_args()
path = os.path.join(a.rec, 'sets.bin')
offs = index(path)
for n in a.sets:
images = read_set(path, offs[n])
views = views_at(a.timeline, n) if a.timeline else []
out = os.path.join(a.out, 'set_%05d.jpg' % n)
cv2.imwrite(out, draw(images, views), [cv2.IMWRITE_JPEG_QUALITY, 85])
print(out)
if __name__ == '__main__':
main()