Hand tracking for Frametop's hand cutouts

fh-camd (camd/) borrows XRService's camera buffers and publishes the tracking
cameras' frames to a shared ring. fh-tracker (trackd/) finds hands in them with
MediaPipe's palm and landmark models on ncnn, triangulates them in 3D, and
publishes them for ft-screens. fh-replay replays recordings offline. tracker/ is
the earlier Python version; tools/ and probes/ hold the checks and experiments.

As of this commit: crop contrast defaults to CLAHE for the palm search and plain
crops for the landmarks, --swap-sides works around fh-camd naming the side
cameras backwards after some XRService restarts (tools/check_sides.py detects
it), and --record-only, --with-dark, --cpus and --keep-presence support the
bright-light and CPU-placement tests.

Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com>
This commit is contained in:
DeeJanuzandClaude Opus 5.5 committed 2026-09-29 22:36:48 -06:00
commit 067a03ce38
42 files changed
+6321

No files matched your search

+169
View File
@@ -0,0 +1,169 @@
#include "io.h"
#include <fcntl.h>
#include <sys/mman.h>
#include <sys/stat.h>
#include <time.h>
#include <unistd.h>
#include <algorithm>
#include <cstdlib>
#include <cstring>
uint64_t mono_ns() {
timespec ts;
clock_gettime(CLOCK_MONOTONIC, &ts);
return uint64_t(ts.tv_sec) * 1'000'000'000 + uint64_t(ts.tv_nsec);
}
int64_t raw_minus_mono_ns() {
timespec a, r, b;
clock_gettime(CLOCK_MONOTONIC, &a);
clock_gettime(CLOCK_MONOTONIC_RAW, &r);
clock_gettime(CLOCK_MONOTONIC, &b);
const int64_t ma = int64_t(a.tv_sec) * 1'000'000'000 + a.tv_nsec, mb = int64_t(b.tv_sec) * 1'000'000'000 + b.tv_nsec;
return int64_t(r.tv_sec) * 1'000'000'000 + r.tv_nsec - (ma + mb) / 2;
}
// ------------------------------------------------------------------------------ ring
bool Ring::open(const char *path, std::string &err) {
const int fd = ::open(path, O_RDONLY | O_CLOEXEC);
if (fd < 0) return err = std::string(path) + ": " + std::strerror(errno), false;
struct stat st;
fstat(fd, &st);
len_ = size_t(st.st_size);
void *m = len_ >= sizeof(fh_ring_hdr_t) ? mmap(nullptr, len_, PROT_READ, MAP_SHARED, fd, 0) : MAP_FAILED;
close(fd);
if (m == MAP_FAILED) return err = std::string(path) + ": can't map it", false;
map_ = static_cast<const uint8_t *>(m);
hdr_ = reinterpret_cast<const fh_ring_hdr_t *>(map_);
if (std::memcmp(hdr_->magic, FH_RING_MAGIC, 8) || hdr_->version != FH_RING_VERSION || hdr_->file_bytes > len_)
return err = std::string(path) + " is not an fh-camd ring", false;
return true;
}
bool Ring::alive() const {
const uint64_t hb = __atomic_load_n(&hdr_->heartbeat_ns, __ATOMIC_ACQUIRE);
return hb && mono_ns() - hb < 1'000'000'000;
}
uint64_t Ring::latest(int i) const { return __atomic_load_n(&hdr_->cams[i].latest, __ATOMIC_ACQUIRE); }
bool Ring::read(int i, uint64_t n, std::vector<uint8_t> &out, fh_ring_slot_t *meta) const {
const fh_ring_cam_t &c = hdr_->cams[i];
if (!n || c.slot_offset + c.nslots * c.slot_bytes > len_) return false;
const uint8_t *slot = map_ + c.slot_offset + (n % c.nslots) * c.slot_bytes;
const auto *s = reinterpret_cast<const fh_ring_slot_t *>(slot);
const uint64_t seq = __atomic_load_n(&s->seq, __ATOMIC_ACQUIRE);
if (seq != 2 * n + 2) return false;
std::memcpy(meta, slot, sizeof *meta);
out.resize(size_t(c.width) * c.height);
for (uint32_t y = 0; y < c.height; ++y)
std::memcpy(out.data() + size_t(y) * c.width, slot + sizeof(fh_ring_slot_t) + size_t(y) * c.stride, c.width);
__atomic_thread_fence(__ATOMIC_ACQUIRE);
return __atomic_load_n(&s->seq, __ATOMIC_RELAXED) == seq;
}
bool Ring::meta(int i, uint64_t n, fh_ring_slot_t *meta) const {
const fh_ring_cam_t &c = hdr_->cams[i];
if (!n || c.slot_offset + c.nslots * c.slot_bytes > len_) return false;
const uint8_t *slot = map_ + c.slot_offset + (n % c.nslots) * c.slot_bytes;
const auto *s = reinterpret_cast<const fh_ring_slot_t *>(slot);
const uint64_t seq = __atomic_load_n(&s->seq, __ATOMIC_ACQUIRE);
if (seq != 2 * n + 2) return false;
std::memcpy(meta, slot, sizeof *meta);
__atomic_thread_fence(__ATOMIC_ACQUIRE);
return __atomic_load_n(&s->seq, __ATOMIC_RELAXED) == seq;
}
// ------------------------------------------------------------------------- publisher
namespace {
// The hand's shape to cut out, as capsules (tracker/publish.py has the same model).
// Radii are a real hand's half-widths plus a small margin for tracking noise.
const int kThumb[][2] = {{0, 1}, {1, 2}, {2, 3}, {3, 4}};
const int kFingers[][2] = {{5, 6}, {6, 7}, {7, 8}, {9, 10}, {10, 11}, {11, 12}, {13, 14}, {14, 15}, {15, 16},
{17, 18}, {18, 19}, {19, 20}};
const int kPalm[][2] = {{0, 5}, {0, 9}, {0, 13}, {0, 17}, {5, 9}, {9, 13}, {13, 17}, {1, 5}};
constexpr double kThumbR = 0.0095, kFinger = 0.0085, kPalmR = 0.015, kArm[2] = {0.028, 0.034}, kArmLen = 0.16,
kMargin = 0.004;
// Nothing is cut closer than this in front of the eyes (head frame, -z is forward). A
// point near the eyes' plane lands far across a screen with a huge radius, so one bad
// estimate there tears a hole through it; real hands that close aren't tracked anyway.
constexpr double kNear = 0.12;
// Adds the capsule, clipped to the part at least kNear in front of the eyes.
void put(fh_capsule_t *caps, uint32_t &n, V3 a, V3 b, double ra, double rb) {
if (n >= FH_HANDS_MAX_CAPSULES) return;
const double za = -a[2] - kNear, zb = -b[2] - kNear; // >= 0: far enough in front
if (za < 0 && zb < 0) return;
if (za < 0 || zb < 0) {
const double t = za / (za - zb); // where the segment crosses the near plane
const V3 m = a + (b - a) * t;
const double rm = ra + (rb - ra) * t;
if (za < 0) a = m, ra = rm;
else b = m, rb = rm;
}
fh_capsule_t &c = caps[n++];
for (int k = 0; k < 3; ++k) c.a[k] = float(a[k]), c.b[k] = float(b[k]);
c.ra = float(ra + kMargin), c.rb = float(rb + kMargin);
}
} // namespace
bool Publisher::open(std::string &err) {
const char *run = std::getenv("XDG_RUNTIME_DIR");
const std::string dir = std::string(run ? run : "/run/user/" + std::to_string(getuid())) + "/frame-hands";
mkdir(dir.c_str(), 0700);
chmod(dir.c_str(), 0700);
const std::string path = dir + "/hands";
const int fd = ::open(path.c_str(), O_RDWR | O_CREAT | O_NOFOLLOW | O_CLOEXEC, 0600);
if (fd < 0 || ftruncate(fd, sizeof(fh_hands_t)) < 0) return err = path + ": " + std::strerror(errno), false;
void *m = mmap(nullptr, sizeof(fh_hands_t), PROT_READ | PROT_WRITE, MAP_SHARED, fd, 0);
close(fd);
if (m == MAP_FAILED) return err = path + ": can't map it", false;
out_ = static_cast<fh_hands_t *>(m);
std::memset(out_, 0, sizeof *out_);
std::memcpy(out_->magic, FH_HANDS_MAGIC, 8);
out_->version = FH_HANDS_VERSION;
out_->size = sizeof(fh_hands_t);
return true;
}
void Publisher::write(const std::vector<const Hand *> &in, uint64_t capture_ns) {
std::vector<const Hand *> hands = in;
std::sort(hands.begin(), hands.end(), [](const Hand *a, const Hand *b) { return a->frames > b->frames; });
if (hands.size() > FH_HANDS_MAX_HANDS) hands.resize(FH_HANDS_MAX_HANDS);
__atomic_store_n(&out_->seq, 2 * ++seq_ - 1, __ATOMIC_RELAXED);
__atomic_thread_fence(__ATOMIC_RELEASE);
uint32_t nc = 0;
for (size_t k = 0; k < FH_HANDS_MAX_HANDS; ++k) {
fh_hand_t &o = out_->hands[k];
std::memset(&o, 0, sizeof o);
if (k >= hands.size()) continue;
const Hand &h = *hands[k];
o.id = uint32_t(h.id);
o.flags = (h.right() ? FH_HAND_RIGHT : 0) | (h.nviews >= 2 ? FH_HAND_STEREO : 0);
o.confidence = float(std::min(1.0, h.frames / 5.0));
for (int i = 0; i < 21; ++i)
for (int j = 0; j < 3; ++j) o.pts[i][j] = float(h.smooth[i][j]);
const uint32_t first = nc;
for (auto &b : kThumb) put(out_->capsules, nc, h.smooth[b[0]], h.smooth[b[1]], kThumbR, kThumbR);
for (auto &b : kFingers) put(out_->capsules, nc, h.smooth[b[0]], h.smooth[b[1]], kFinger, kFinger);
for (auto &b : kPalm) put(out_->capsules, nc, h.smooth[b[0]], h.smooth[b[1]], kPalmR, kPalmR);
// the forearm carries on from the hand's own axis (middle knuckle -> wrist); the
// wrist bends, but much less than a guess at where the elbow is gets wrong
const V3 wrist = h.smooth[0], d = wrist - h.smooth[9];
const double n = norm(d);
if (n > 0.02) put(out_->capsules, nc, wrist, wrist + d * (kArmLen / n), kArm[0], kArm[1]);
o.ncapsules = nc - first;
}
for (uint32_t k = nc; k < FH_HANDS_MAX_CAPSULES; ++k) std::memset(&out_->capsules[k], 0, sizeof(fh_capsule_t));
out_->capture_ns = capture_ns;
out_->publish_ns = mono_ns();
out_->nhands = uint32_t(hands.size());
out_->ncapsules = nc;
__atomic_store_n(&out_->seq, 2 * seq_, __ATOMIC_RELEASE);
}