mirror of
https://github.com/DeeJanuz/frametop.git
synced 2026-10-06 10:00:11 +02:00
Hand tracking for Frametop's hand cutouts
fh-camd (camd/) borrows XRService's camera buffers and publishes the tracking cameras' frames to a shared ring. fh-tracker (trackd/) finds hands in them with MediaPipe's palm and landmark models on ncnn, triangulates them in 3D, and publishes them for ft-screens. fh-replay replays recordings offline. tracker/ is the earlier Python version; tools/ and probes/ hold the checks and experiments. As of this commit: crop contrast defaults to CLAHE for the palm search and plain crops for the landmarks, --swap-sides works around fh-camd naming the side cameras backwards after some XRService restarts (tools/check_sides.py detects it), and --record-only, --with-dark, --cpus and --keep-presence support the bright-light and CPU-placement tests. Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com>
This commit is contained in:
commit
067a03ce38
42 files changed
+6321
No files matched your search
+169
@@ -0,0 +1,169 @@
|
||||
#include "io.h"
|
||||
|
||||
#include <fcntl.h>
|
||||
#include <sys/mman.h>
|
||||
#include <sys/stat.h>
|
||||
#include <time.h>
|
||||
#include <unistd.h>
|
||||
|
||||
#include <algorithm>
|
||||
#include <cstdlib>
|
||||
#include <cstring>
|
||||
|
||||
uint64_t mono_ns() {
|
||||
timespec ts;
|
||||
clock_gettime(CLOCK_MONOTONIC, &ts);
|
||||
return uint64_t(ts.tv_sec) * 1'000'000'000 + uint64_t(ts.tv_nsec);
|
||||
}
|
||||
|
||||
int64_t raw_minus_mono_ns() {
|
||||
timespec a, r, b;
|
||||
clock_gettime(CLOCK_MONOTONIC, &a);
|
||||
clock_gettime(CLOCK_MONOTONIC_RAW, &r);
|
||||
clock_gettime(CLOCK_MONOTONIC, &b);
|
||||
const int64_t ma = int64_t(a.tv_sec) * 1'000'000'000 + a.tv_nsec, mb = int64_t(b.tv_sec) * 1'000'000'000 + b.tv_nsec;
|
||||
return int64_t(r.tv_sec) * 1'000'000'000 + r.tv_nsec - (ma + mb) / 2;
|
||||
}
|
||||
|
||||
// ------------------------------------------------------------------------------ ring
|
||||
|
||||
bool Ring::open(const char *path, std::string &err) {
|
||||
const int fd = ::open(path, O_RDONLY | O_CLOEXEC);
|
||||
if (fd < 0) return err = std::string(path) + ": " + std::strerror(errno), false;
|
||||
struct stat st;
|
||||
fstat(fd, &st);
|
||||
len_ = size_t(st.st_size);
|
||||
void *m = len_ >= sizeof(fh_ring_hdr_t) ? mmap(nullptr, len_, PROT_READ, MAP_SHARED, fd, 0) : MAP_FAILED;
|
||||
close(fd);
|
||||
if (m == MAP_FAILED) return err = std::string(path) + ": can't map it", false;
|
||||
map_ = static_cast<const uint8_t *>(m);
|
||||
hdr_ = reinterpret_cast<const fh_ring_hdr_t *>(map_);
|
||||
if (std::memcmp(hdr_->magic, FH_RING_MAGIC, 8) || hdr_->version != FH_RING_VERSION || hdr_->file_bytes > len_)
|
||||
return err = std::string(path) + " is not an fh-camd ring", false;
|
||||
return true;
|
||||
}
|
||||
|
||||
bool Ring::alive() const {
|
||||
const uint64_t hb = __atomic_load_n(&hdr_->heartbeat_ns, __ATOMIC_ACQUIRE);
|
||||
return hb && mono_ns() - hb < 1'000'000'000;
|
||||
}
|
||||
|
||||
uint64_t Ring::latest(int i) const { return __atomic_load_n(&hdr_->cams[i].latest, __ATOMIC_ACQUIRE); }
|
||||
|
||||
bool Ring::read(int i, uint64_t n, std::vector<uint8_t> &out, fh_ring_slot_t *meta) const {
|
||||
const fh_ring_cam_t &c = hdr_->cams[i];
|
||||
if (!n || c.slot_offset + c.nslots * c.slot_bytes > len_) return false;
|
||||
const uint8_t *slot = map_ + c.slot_offset + (n % c.nslots) * c.slot_bytes;
|
||||
const auto *s = reinterpret_cast<const fh_ring_slot_t *>(slot);
|
||||
const uint64_t seq = __atomic_load_n(&s->seq, __ATOMIC_ACQUIRE);
|
||||
if (seq != 2 * n + 2) return false;
|
||||
std::memcpy(meta, slot, sizeof *meta);
|
||||
out.resize(size_t(c.width) * c.height);
|
||||
for (uint32_t y = 0; y < c.height; ++y)
|
||||
std::memcpy(out.data() + size_t(y) * c.width, slot + sizeof(fh_ring_slot_t) + size_t(y) * c.stride, c.width);
|
||||
__atomic_thread_fence(__ATOMIC_ACQUIRE);
|
||||
return __atomic_load_n(&s->seq, __ATOMIC_RELAXED) == seq;
|
||||
}
|
||||
|
||||
bool Ring::meta(int i, uint64_t n, fh_ring_slot_t *meta) const {
|
||||
const fh_ring_cam_t &c = hdr_->cams[i];
|
||||
if (!n || c.slot_offset + c.nslots * c.slot_bytes > len_) return false;
|
||||
const uint8_t *slot = map_ + c.slot_offset + (n % c.nslots) * c.slot_bytes;
|
||||
const auto *s = reinterpret_cast<const fh_ring_slot_t *>(slot);
|
||||
const uint64_t seq = __atomic_load_n(&s->seq, __ATOMIC_ACQUIRE);
|
||||
if (seq != 2 * n + 2) return false;
|
||||
std::memcpy(meta, slot, sizeof *meta);
|
||||
__atomic_thread_fence(__ATOMIC_ACQUIRE);
|
||||
return __atomic_load_n(&s->seq, __ATOMIC_RELAXED) == seq;
|
||||
}
|
||||
|
||||
// ------------------------------------------------------------------------- publisher
|
||||
|
||||
namespace {
|
||||
|
||||
// The hand's shape to cut out, as capsules (tracker/publish.py has the same model).
|
||||
// Radii are a real hand's half-widths plus a small margin for tracking noise.
|
||||
const int kThumb[][2] = {{0, 1}, {1, 2}, {2, 3}, {3, 4}};
|
||||
const int kFingers[][2] = {{5, 6}, {6, 7}, {7, 8}, {9, 10}, {10, 11}, {11, 12}, {13, 14}, {14, 15}, {15, 16},
|
||||
{17, 18}, {18, 19}, {19, 20}};
|
||||
const int kPalm[][2] = {{0, 5}, {0, 9}, {0, 13}, {0, 17}, {5, 9}, {9, 13}, {13, 17}, {1, 5}};
|
||||
constexpr double kThumbR = 0.0095, kFinger = 0.0085, kPalmR = 0.015, kArm[2] = {0.028, 0.034}, kArmLen = 0.16,
|
||||
kMargin = 0.004;
|
||||
// Nothing is cut closer than this in front of the eyes (head frame, -z is forward). A
|
||||
// point near the eyes' plane lands far across a screen with a huge radius, so one bad
|
||||
// estimate there tears a hole through it; real hands that close aren't tracked anyway.
|
||||
constexpr double kNear = 0.12;
|
||||
|
||||
// Adds the capsule, clipped to the part at least kNear in front of the eyes.
|
||||
void put(fh_capsule_t *caps, uint32_t &n, V3 a, V3 b, double ra, double rb) {
|
||||
if (n >= FH_HANDS_MAX_CAPSULES) return;
|
||||
const double za = -a[2] - kNear, zb = -b[2] - kNear; // >= 0: far enough in front
|
||||
if (za < 0 && zb < 0) return;
|
||||
if (za < 0 || zb < 0) {
|
||||
const double t = za / (za - zb); // where the segment crosses the near plane
|
||||
const V3 m = a + (b - a) * t;
|
||||
const double rm = ra + (rb - ra) * t;
|
||||
if (za < 0) a = m, ra = rm;
|
||||
else b = m, rb = rm;
|
||||
}
|
||||
fh_capsule_t &c = caps[n++];
|
||||
for (int k = 0; k < 3; ++k) c.a[k] = float(a[k]), c.b[k] = float(b[k]);
|
||||
c.ra = float(ra + kMargin), c.rb = float(rb + kMargin);
|
||||
}
|
||||
|
||||
} // namespace
|
||||
|
||||
bool Publisher::open(std::string &err) {
|
||||
const char *run = std::getenv("XDG_RUNTIME_DIR");
|
||||
const std::string dir = std::string(run ? run : "/run/user/" + std::to_string(getuid())) + "/frame-hands";
|
||||
mkdir(dir.c_str(), 0700);
|
||||
chmod(dir.c_str(), 0700);
|
||||
const std::string path = dir + "/hands";
|
||||
const int fd = ::open(path.c_str(), O_RDWR | O_CREAT | O_NOFOLLOW | O_CLOEXEC, 0600);
|
||||
if (fd < 0 || ftruncate(fd, sizeof(fh_hands_t)) < 0) return err = path + ": " + std::strerror(errno), false;
|
||||
void *m = mmap(nullptr, sizeof(fh_hands_t), PROT_READ | PROT_WRITE, MAP_SHARED, fd, 0);
|
||||
close(fd);
|
||||
if (m == MAP_FAILED) return err = path + ": can't map it", false;
|
||||
out_ = static_cast<fh_hands_t *>(m);
|
||||
std::memset(out_, 0, sizeof *out_);
|
||||
std::memcpy(out_->magic, FH_HANDS_MAGIC, 8);
|
||||
out_->version = FH_HANDS_VERSION;
|
||||
out_->size = sizeof(fh_hands_t);
|
||||
return true;
|
||||
}
|
||||
|
||||
void Publisher::write(const std::vector<const Hand *> &in, uint64_t capture_ns) {
|
||||
std::vector<const Hand *> hands = in;
|
||||
std::sort(hands.begin(), hands.end(), [](const Hand *a, const Hand *b) { return a->frames > b->frames; });
|
||||
if (hands.size() > FH_HANDS_MAX_HANDS) hands.resize(FH_HANDS_MAX_HANDS);
|
||||
__atomic_store_n(&out_->seq, 2 * ++seq_ - 1, __ATOMIC_RELAXED);
|
||||
__atomic_thread_fence(__ATOMIC_RELEASE);
|
||||
uint32_t nc = 0;
|
||||
for (size_t k = 0; k < FH_HANDS_MAX_HANDS; ++k) {
|
||||
fh_hand_t &o = out_->hands[k];
|
||||
std::memset(&o, 0, sizeof o);
|
||||
if (k >= hands.size()) continue;
|
||||
const Hand &h = *hands[k];
|
||||
o.id = uint32_t(h.id);
|
||||
o.flags = (h.right() ? FH_HAND_RIGHT : 0) | (h.nviews >= 2 ? FH_HAND_STEREO : 0);
|
||||
o.confidence = float(std::min(1.0, h.frames / 5.0));
|
||||
for (int i = 0; i < 21; ++i)
|
||||
for (int j = 0; j < 3; ++j) o.pts[i][j] = float(h.smooth[i][j]);
|
||||
const uint32_t first = nc;
|
||||
for (auto &b : kThumb) put(out_->capsules, nc, h.smooth[b[0]], h.smooth[b[1]], kThumbR, kThumbR);
|
||||
for (auto &b : kFingers) put(out_->capsules, nc, h.smooth[b[0]], h.smooth[b[1]], kFinger, kFinger);
|
||||
for (auto &b : kPalm) put(out_->capsules, nc, h.smooth[b[0]], h.smooth[b[1]], kPalmR, kPalmR);
|
||||
// the forearm carries on from the hand's own axis (middle knuckle -> wrist); the
|
||||
// wrist bends, but much less than a guess at where the elbow is gets wrong
|
||||
const V3 wrist = h.smooth[0], d = wrist - h.smooth[9];
|
||||
const double n = norm(d);
|
||||
if (n > 0.02) put(out_->capsules, nc, wrist, wrist + d * (kArmLen / n), kArm[0], kArm[1]);
|
||||
o.ncapsules = nc - first;
|
||||
}
|
||||
for (uint32_t k = nc; k < FH_HANDS_MAX_CAPSULES; ++k) std::memset(&out_->capsules[k], 0, sizeof(fh_capsule_t));
|
||||
out_->capture_ns = capture_ns;
|
||||
out_->publish_ns = mono_ns();
|
||||
out_->nhands = uint32_t(hands.size());
|
||||
out_->ncapsules = nc;
|
||||
__atomic_store_n(&out_->seq, 2 * seq_, __ATOMIC_RELEASE);
|
||||
}
|
||||
Reference in new issue
Block a user