feat: add plasma-face-unlock

This commit is contained in:
Felitendo committed 2026-09-22 19:39:04 +02:00
commit f671acc93b
105 files changed
+13862

No files matched your search

+285
View File
@@ -0,0 +1,285 @@
// SPDX-License-Identifier: GPL-3.0-or-later
#include "camera.h"
#include <QDir>
#include <QFile>
#include <QFileInfo>
#include <opencv2/imgcodecs.hpp>
#include <opencv2/imgproc.hpp>
#include <fcntl.h>
#include <linux/videodev2.h>
#include <sys/ioctl.h>
#include <unistd.h>
#include <algorithm>
#include <thread>
using namespace std::chrono;
namespace
{
constexpr int Width = 640;
constexpr int Height = 480;
// Pictures stand in for a camera at this rate, each held for a while so the
// checks see a still face the way they would see a person holding still.
constexpr double ImageFrameMs = 1000.0 / 15.0;
constexpr int ImageRepeat = 12;
QString sysName(const QString &device)
{
QFile file(QStringLiteral("/sys/class/video4linux/%1/name").arg(QFileInfo(device).fileName()));
if (!file.open(QIODevice::ReadOnly)) {
return {};
}
return QString::fromUtf8(file.readAll()).trimmed();
}
// Opens the node just long enough to ask what it is. Neither this nor the
// format list below starts streaming, so it does not switch the light on.
bool probe(const QString &path, CameraInfo *info)
{
const int fd = ::open(QFile::encodeName(path).constData(), O_RDONLY | O_NONBLOCK | O_CLOEXEC);
if (fd < 0) {
return false;
}
v4l2_capability cap{};
bool ok = ::ioctl(fd, VIDIOC_QUERYCAP, &cap) == 0;
if (ok) {
const quint32 caps = (cap.capabilities & V4L2_CAP_DEVICE_CAPS) ? cap.device_caps : cap.capabilities;
ok = (caps & V4L2_CAP_VIDEO_CAPTURE) && !(caps & V4L2_CAP_META_CAPTURE);
}
bool colour = false;
bool grey = false;
if (ok) {
for (quint32 i = 0;; ++i) {
v4l2_fmtdesc fmt{};
fmt.index = i;
fmt.type = V4L2_BUF_TYPE_VIDEO_CAPTURE;
if (::ioctl(fd, VIDIOC_ENUM_FMT, &fmt) != 0) {
break;
}
switch (fmt.pixelformat) {
case V4L2_PIX_FMT_GREY:
case V4L2_PIX_FMT_Y10:
case V4L2_PIX_FMT_Y12:
case V4L2_PIX_FMT_Y16:
grey = true;
break;
default:
colour = true;
break;
}
}
ok = colour || grey;
}
::close(fd);
if (ok && info) {
info->path = path;
info->name = sysName(path);
if (info->name.isEmpty()) {
info->name = QString::fromUtf8(reinterpret_cast<const char *>(cap.card));
}
info->infrared = grey && !colour;
}
return ok;
}
} // namespace
Camera::Camera() = default;
Camera::~Camera()
{
close();
}
QList<CameraInfo> Camera::list()
{
QList<CameraInfo> result;
const QDir dev(QStringLiteral("/dev"));
QStringList nodes = dev.entryList({QStringLiteral("video*")}, QDir::System);
std::sort(nodes.begin(), nodes.end(), [](const QString &a, const QString &b) {
return a.mid(5).toInt() < b.mid(5).toInt();
});
for (const QString &node : std::as_const(nodes)) {
CameraInfo info;
if (probe(dev.filePath(node), &info)) {
result.append(info);
}
}
return result;
}
QString Camera::autoPath()
{
const QList<CameraInfo> cameras = list();
for (const CameraInfo &c : cameras) {
if (!c.infrared) {
return c.path;
}
}
return cameras.isEmpty() ? QString() : cameras.first().path;
}
bool Camera::open(const QString &spec, QString *error)
{
close();
m_start = steady_clock::now();
m_next = m_start;
if (spec.startsWith(u"images:")) {
return openImages(spec.mid(7), error);
}
if (spec.startsWith(u"file:")) {
const QString path = spec.mid(5);
if (!m_capture.open(QFile::encodeName(path).toStdString(), cv::CAP_ANY) || !m_capture.isOpened()) {
*error = QStringLiteral("cannot open video file %1").arg(path);
return false;
}
m_paced = true;
m_open = true;
m_description = QFileInfo(path).fileName();
return true;
}
QString path = spec;
if (path.isEmpty() || path == u"auto") {
path = autoPath();
if (path.isEmpty()) {
*error = QStringLiteral("no camera found");
return false;
}
}
CameraInfo info;
if (!probe(path, &info)) {
*error = QStringLiteral("%1 is not a camera").arg(path);
return false;
}
if (!m_capture.open(QFile::encodeName(path).toStdString(), cv::CAP_V4L2) || !m_capture.isOpened()) {
*error = QStringLiteral("cannot open %1 (in use by another program?)").arg(path);
return false;
}
// MJPEG gets a webcam its full frame rate over USB 2; raw YUYV at 640x480
// often tops out at 15 fps. A camera that has no MJPEG ignores the request.
if (!info.infrared) {
m_capture.set(cv::CAP_PROP_FOURCC, cv::VideoWriter::fourcc('M', 'J', 'P', 'G'));
}
m_capture.set(cv::CAP_PROP_FRAME_WIDTH, Width);
m_capture.set(cv::CAP_PROP_FRAME_HEIGHT, Height);
m_capture.set(cv::CAP_PROP_FPS, 30);
// A stale frame in the driver's queue is a picture of whoever sat there a
// moment ago. Keep the queue as short as the driver allows.
m_capture.set(cv::CAP_PROP_BUFFERSIZE, 1);
m_infrared = info.infrared;
m_paced = false;
m_open = true;
m_description = QStringLiteral("%1 (%2)").arg(info.name, path);
return true;
}
bool Camera::openImages(const QString &dir, QString *error)
{
const QDir d(dir);
const QStringList files = d.entryList({QStringLiteral("*.jpg"), QStringLiteral("*.jpeg"), QStringLiteral("*.png")},
QDir::Files, QDir::Name);
for (const QString &f : files) {
cv::Mat img = cv::imread(QFile::encodeName(d.filePath(f)).toStdString(), cv::IMREAD_COLOR);
if (img.empty()) {
continue;
}
const double scale = double(Width) / std::max(img.cols, img.rows);
if (scale < 1.0) {
cv::resize(img, img, {}, scale, scale, cv::INTER_AREA);
}
m_images.append(img);
}
if (m_images.isEmpty()) {
*error = QStringLiteral("no pictures in %1").arg(dir);
return false;
}
m_imageIndex = 0;
m_imageRepeat = 0;
m_paced = true;
m_open = true;
m_description = QStringLiteral("pictures in %1").arg(dir);
return true;
}
void Camera::close()
{
if (m_capture.isOpened()) {
m_capture.release();
}
m_images.clear();
m_open = false;
m_infrared = false;
}
bool Camera::isOpen() const
{
return m_open;
}
void Camera::pace(double frameMs)
{
m_next += duration_cast<steady_clock::duration>(duration<double, std::milli>(frameMs));
const auto now = steady_clock::now();
if (m_next > now) {
std::this_thread::sleep_until(m_next);
} else {
m_next = now;
}
}
bool Camera::readImages(cv::Mat &bgr)
{
pace(ImageFrameMs);
bgr = m_images.at(m_imageIndex).clone();
if (++m_imageRepeat >= ImageRepeat) {
m_imageRepeat = 0;
m_imageIndex = (m_imageIndex + 1) % m_images.size();
}
return true;
}
bool Camera::read(cv::Mat &bgr, double *timestampMs)
{
if (!m_open) {
return false;
}
bool ok;
if (!m_images.isEmpty()) {
ok = readImages(bgr);
} else {
if (m_paced) {
double fps = m_capture.get(cv::CAP_PROP_FPS);
if (!(fps > 1 && fps < 240)) {
fps = 30;
}
pace(1000.0 / fps);
}
ok = m_capture.read(bgr) && !bgr.empty();
}
if (!ok) {
return false;
}
if (bgr.channels() == 1) {
cv::cvtColor(bgr, bgr, cv::COLOR_GRAY2BGR);
}
if (timestampMs) {
*timestampMs = duration<double, std::milli>(steady_clock::now() - m_start).count();
}
return true;
}
+76
View File
@@ -0,0 +1,76 @@
// SPDX-License-Identifier: GPL-3.0-or-later
//
// Where frames come from.
//
// Normally a V4L2 device. For testing without a camera (a virtual machine, a
// build server) a video file or a folder of pictures stands in for one and is
// played back at the pace a camera would deliver it, so that anything timed in
// the liveness checks behaves the way it would with the real thing.
#pragma once
#include <QList>
#include <QString>
#include <opencv2/core.hpp>
#include <opencv2/videoio.hpp>
#include <chrono>
#include <memory>
struct CameraInfo {
QString path;
QString name;
// Infrared cameras (the kind Windows Hello uses) only offer grey formats.
bool infrared = false;
};
class Camera
{
public:
Camera();
~Camera();
// spec is a /dev/video path, "auto", "file:<video>" or "images:<dir>".
bool open(const QString &spec, QString *error);
void close();
bool isOpen() const;
// Blocks until the next frame. The timestamp is in milliseconds on a
// monotonic clock and only means something relative to other frames.
bool read(cv::Mat &bgr, double *timestampMs);
bool isInfrared() const
{
return m_infrared;
}
QString description() const
{
return m_description;
}
// Every V4L2 device that can capture video. Metadata nodes, which every
// UVC camera also exposes, are left out.
static QList<CameraInfo> list();
// What "auto" means on this machine: the first colour camera, else the
// first camera at all.
static QString autoPath();
private:
bool openImages(const QString &dir, QString *error);
bool readImages(cv::Mat &bgr);
void pace(double frameMs);
cv::VideoCapture m_capture;
bool m_open = false;
bool m_infrared = false;
bool m_paced = false;
QString m_description;
QList<cv::Mat> m_images;
qsizetype m_imageIndex = 0;
int m_imageRepeat = 0;
std::chrono::steady_clock::time_point m_start;
std::chrono::steady_clock::time_point m_next;
};
+14
View File
@@ -0,0 +1,14 @@
// SPDX-License-Identifier: GPL-3.0-or-later
//
// OpenCV 5 moved the contour and transform helpers (getPerspectiveTransform,
// approxPolyDP and friends) out of imgproc into a module of their own. The
// distributions this builds on ship 4.x and 5.x, so both are included here.
#pragma once
#include <opencv2/core/version.hpp>
#include <opencv2/imgproc.hpp>
#if CV_VERSION_MAJOR >= 5
#include <opencv2/geometry.hpp>
#endif
+82
View File
@@ -0,0 +1,82 @@
// SPDX-License-Identifier: GPL-3.0-or-later
#include "keyvalue.h"
#include <QFile>
KeyValueFile KeyValueFile::load(const QString &path)
{
KeyValueFile kv;
QFile file(path);
if (!file.open(QIODevice::ReadOnly | QIODevice::Text)) {
return kv;
}
while (!file.atEnd()) {
QString line = QString::fromUtf8(file.readLine()).trimmed();
if (line.isEmpty() || line.startsWith(QLatin1Char('#'))) {
continue;
}
const qsizetype eq = line.indexOf(QLatin1Char('='));
if (eq <= 0) {
continue;
}
const QString key = line.left(eq).trimmed();
QString value = line.mid(eq + 1);
// A comment can follow a value, the same as in the shell reader.
const qsizetype hash = value.indexOf(QLatin1Char('#'));
if (hash >= 0) {
value.truncate(hash);
}
value = value.trimmed();
if (value.size() >= 2 && value.startsWith(QLatin1Char('"')) && value.endsWith(QLatin1Char('"'))) {
value = value.mid(1, value.size() - 2);
}
// The last occurrence wins, which is also what the shell reader does.
kv.m_values.insert(key, value);
}
return kv;
}
QString KeyValueFile::value(const QString &key, const QString &fallback) const
{
const auto it = m_values.constFind(key);
if (it == m_values.cend() || it->isEmpty()) {
return fallback;
}
return *it;
}
bool KeyValueFile::contains(const QString &key) const
{
return m_values.contains(key);
}
bool KeyValueFile::parseBool(const QString &value, bool fallback)
{
const QString v = value.trimmed().toLower();
if (v == u"yes" || v == u"y" || v == u"true" || v == u"1" || v == u"on" || v == u"enabled") {
return true;
}
if (v == u"no" || v == u"n" || v == u"false" || v == u"0" || v == u"off" || v == u"disabled") {
return false;
}
return fallback;
}
bool KeyValueFile::boolean(const QString &key, bool fallback) const
{
return parseBool(value(key), fallback);
}
int KeyValueFile::integer(const QString &key, int fallback, int min, int max) const
{
bool ok = false;
const int v = value(key).toInt(&ok);
if (!ok) {
return fallback;
}
return std::clamp(v, min, max);
}
+27
View File
@@ -0,0 +1,27 @@
// SPDX-License-Identifier: GPL-3.0-or-later
//
// The settings files, read the same way the shell code reads them: one
// Key=Value per line, '#' starts a comment, quotes around a value are dropped.
// Both halves of the program write and read these files, so they have to
// agree on every detail of the format.
#pragma once
#include <QHash>
#include <QString>
class KeyValueFile
{
public:
static KeyValueFile load(const QString &path);
QString value(const QString &key, const QString &fallback = {}) const;
bool boolean(const QString &key, bool fallback) const;
int integer(const QString &key, int fallback, int min, int max) const;
bool contains(const QString &key) const;
static bool parseBool(const QString &value, bool fallback);
private:
QHash<QString, QString> m_values;
};
+535
View File
@@ -0,0 +1,535 @@
// SPDX-License-Identifier: GPL-3.0-or-later
#include "liveness.h"
#include "cvcompat.h"
#include <algorithm>
#include <cmath>
namespace
{
// The eyes are levelled and scaled onto a fixed canvas before anything is
// measured, so the numbers below are in canvas pixels and mean the same at
// any distance and any head tilt. 64 pixels between the eyes.
constexpr int Canvas = 128;
constexpr float CanvasIod = 64.f;
constexpr int EyeY = Canvas / 2;
constexpr int RightEyeX = Canvas / 2 - int(CanvasIod / 2);
constexpr int LeftEyeX = Canvas / 2 + int(CanvasIod / 2);
// A slice through the middle of the eye, narrow enough to stay on the iris
// and tall enough to hold a fully open one (about 0.3 eye distances).
constexpr int BandHalfWidth = 7;
constexpr int BandHalfHeight = 10;
// Skin under the eye, clear of lashes and of the shadow below the lid.
constexpr int SkinTop = EyeY + 22;
constexpr int SkinBottom = EyeY + 32;
constexpr int SkinHalfWidth = 12;
// A row of the slice counts as dark below this share of the skin's
// brightness. Iris and pupil are well below it on every skin tone that
// was checked; a closed lid is skin and sits well above it.
constexpr float DarkShare = 0.7f;
// Glare: the Y floor and the chroma tolerance for "colourless and nearly
// white", in YCrCb.
constexpr int SpecularLuma = 235;
constexpr int SpecularChroma = 10;
constexpr int GlareGrid = 8;
cv::Mat levelledFace(const cv::Mat &bgr, const Face &face)
{
const cv::Point2f mid = face.eyeMid();
const cv::Point2f d = face.points[LeftEye] - face.points[RightEye];
const double roll = std::atan2(d.y, d.x) * 180.0 / M_PI;
const double scale = CanvasIod / std::max(1.f, face.interocular());
cv::Mat m = cv::getRotationMatrix2D(mid, roll, scale);
m.at<double>(0, 2) += Canvas / 2.0 - mid.x;
m.at<double>(1, 2) += Canvas / 2.0 - mid.y;
cv::Mat grey, out;
cv::cvtColor(bgr, grey, cv::COLOR_BGR2GRAY);
cv::warpAffine(grey, out, m, cv::Size(Canvas, Canvas), cv::INTER_LINEAR, cv::BORDER_REPLICATE);
return out;
}
float eyeOpenness(const cv::Mat &canvas, int eyeX, float *skinOut)
{
const cv::Rect skinRect(eyeX - SkinHalfWidth, SkinTop, 2 * SkinHalfWidth, SkinBottom - SkinTop);
const float skin = float(cv::mean(canvas(skinRect))[0]);
*skinOut = skin;
if (skin < 20) {
return -1;
}
const cv::Rect band(eyeX - BandHalfWidth, EyeY - BandHalfHeight, 2 * BandHalfWidth + 1, 2 * BandHalfHeight + 1);
cv::Mat rows;
cv::reduce(canvas(band), rows, 1, cv::REDUCE_AVG, CV_32F);
int dark = 0;
for (int y = 0; y < rows.rows; ++y) {
if (rows.at<float>(y, 0) < DarkShare * skin) {
++dark;
}
}
return float(dark) / float(rows.rows);
}
bool insidePolygon(const std::vector<cv::Point> &poly, cv::Point2f p)
{
return cv::pointPolygonTest(poly, p, false) >= 0;
}
} // namespace
GlareSample measureGlare(const cv::Mat &bgr, const Face &face)
{
GlareSample g;
const float iod = face.interocular();
const cv::Point2f centre = (face.eyeMid() + face.mouthMid()) * 0.5f;
cv::Rect roi(int(centre.x - 1.1f * iod), int(centre.y - 1.3f * iod), int(2.2f * iod), int(2.4f * iod));
roi &= cv::Rect(0, 0, bgr.cols, bgr.rows);
if (roi.width < 16 || roi.height < 16) {
return g;
}
cv::Mat ycc;
cv::cvtColor(bgr(roi), ycc, cv::COLOR_BGR2YCrCb);
// Glasses throw the screen back from right in front of the eyes, and
// that is a big flat highlight too. It is not what is being looked for,
// so the eyes are left out.
cv::Mat mask(roi.size(), CV_8U, cv::Scalar(255));
for (int e : {RightEye, LeftEye}) {
const cv::Point2f p = face.points[e] - cv::Point2f(float(roi.x), float(roi.y));
cv::circle(mask, p, int(0.38f * iod), cv::Scalar(0), cv::FILLED);
}
std::array<int, GlareGrid * GlareGrid> cells{};
int total = 0;
int considered = 0;
for (int y = 0; y < ycc.rows; ++y) {
const cv::Vec3b *row = ycc.ptr<cv::Vec3b>(y);
const uchar *m = mask.ptr<uchar>(y);
const int gy = std::min(GlareGrid - 1, y * GlareGrid / ycc.rows);
for (int x = 0; x < ycc.cols; ++x) {
if (!m[x]) {
continue;
}
++considered;
const cv::Vec3b &p = row[x];
if (p[0] < SpecularLuma || std::abs(p[1] - 128) > SpecularChroma || std::abs(p[2] - 128) > SpecularChroma) {
continue;
}
++total;
const int gx = std::min(GlareGrid - 1, x * GlareGrid / ycc.cols);
++cells[gy * GlareGrid + gx];
}
}
g.valid = considered > 0;
g.fraction = considered ? float(total) / float(considered) : 0.f;
// Too few pixels to say anything about their shape.
g.cluster = total >= 24 ? float(*std::max_element(cells.begin(), cells.end())) / float(total) : 0.f;
return g;
}
EyeSample measureEyes(const cv::Mat &bgr, const Face &face)
{
EyeSample s;
if (face.interocular() < 20.f) {
return s;
}
const cv::Mat canvas = levelledFace(bgr, face);
// "Right" is the person's right eye, which is on the canvas' left.
float skinR = 0, skinL = 0;
s.right = eyeOpenness(canvas, RightEyeX, &skinR);
s.left = eyeOpenness(canvas, LeftEyeX, &skinL);
if (s.right < 0 || s.left < 0) {
return s;
}
s.openness = (s.left + s.right) / 2;
s.skin = (skinL + skinR) / 2;
s.valid = true;
return s;
}
bool detectDevice(const cv::Mat &bgr, const Face &face)
{
// Edges are looked for at half size. The bezel of a phone held up to the
// camera is big, and small detail would only add rectangles that are not
// one.
const double scale = 320.0 / std::max(1, bgr.cols);
cv::Mat small, grey, edges;
cv::resize(bgr, small, cv::Size(), scale, scale, cv::INTER_AREA);
cv::cvtColor(small, grey, cv::COLOR_BGR2GRAY);
cv::GaussianBlur(grey, grey, cv::Size(5, 5), 0);
cv::Canny(grey, edges, 40, 120);
cv::dilate(edges, edges, cv::Mat(), cv::Point(-1, -1), 1);
std::vector<std::vector<cv::Point>> contours;
cv::findContours(edges, contours, cv::RETR_LIST, cv::CHAIN_APPROX_SIMPLE);
const double frameArea = double(small.cols) * small.rows;
const double faceArea = double(face.box.area()) * scale * scale;
std::array<cv::Point2f, 5> points;
for (int i = 0; i < 5; ++i) {
points[i] = face.points[i] * float(scale);
}
for (const auto &c : contours) {
const double area = std::abs(cv::contourArea(c));
// A device held up to take somebody's place fills a good part of the
// picture, and the face fills a good part of the device. A door or a
// monitor far behind somebody's head does neither.
if (area < 0.10 * frameArea || area > 0.95 * frameArea || area < 1.6 * faceArea || faceArea / area < 0.08) {
continue;
}
std::vector<cv::Point> quad;
cv::approxPolyDP(c, quad, 0.03 * cv::arcLength(c, true), true);
if (quad.size() != 4 || !cv::isContourConvex(quad)) {
continue;
}
const cv::RotatedRect r = cv::minAreaRect(quad);
const float shortSide = std::min(r.size.width, r.size.height);
const float longSide = std::max(r.size.width, r.size.height);
if (longSide <= 0 || shortSide / longSide < 0.3f) {
continue;
}
// It has to frame the face: every one of the five points inside.
bool framed = true;
for (const cv::Point2f &p : points) {
framed = framed && insidePolygon(quad, p);
}
if (framed) {
return true;
}
}
return false;
}
LivenessFrame measureFrame(const cv::Mat &bgr, const Face &face, double timestampMs)
{
LivenessFrame f;
f.t = timestampMs;
f.points = face.points;
f.interocular = face.interocular();
f.pose = estimatePose(face);
f.glare = measureGlare(bgr, face);
f.device = detectDevice(bgr, face);
f.eyes = measureEyes(bgr, face);
return f;
}
// ---------------------------------------------------------------------------
// The analyzer
// ---------------------------------------------------------------------------
void LivenessAnalyzer::reset()
{
m_frames.clear();
m_total = 0;
m_glareHits = 0;
m_deviceHits = 0;
m_depthFired = false;
m_blinkFired = false;
m_depth.clear();
m_depthNum = 0;
m_depthDen = 0;
m_depthPairs = 0;
}
void LivenessAnalyzer::add(const LivenessFrame &frame)
{
m_frames.push_back(frame);
while (!m_frames.empty() && frame.t - m_frames.front().t > WindowMs) {
m_frames.pop_front();
}
// Deny cues count over the whole scan, not just the window: a phone that
// was seen once does not stop having been a phone.
++m_total;
if (frame.glare.valid && frame.glare.fraction >= GlareFraction && frame.glare.cluster >= GlareCluster) {
++m_glareHits;
}
if (frame.device) {
++m_deviceHits;
}
addDepth(frame);
// Confirm cues latch once seen, for the rest of this scan.
if (!m_depthFired) {
m_depthFired = depthRange() >= DepthMinRange && m_depthPairs >= 6 && depthRatio() >= DepthMinRatio && depthConsistent();
}
if (!m_blinkFired) {
float strength = 0;
m_blinkFired = blinkSeen(&strength);
}
}
// How far the nose misses the spot a flat face would put it, measured against
// how far the head turned.
//
// For a pair of frames far enough apart in yaw, the four points that lie close
// to one plane (eyes, mouth corners) give the homography between the two
// frames. A point on a flat picture lands exactly where it predicts. The real
// nose tip is about a third of an eye distance in front of that plane, so it
// lands off the prediction, sideways, by about as much as the yaw estimate
// changed: both are the same parallax, seen two ways. The slope of miss
// against yaw change is therefore near 1 for anything with a nose sticking
// out of it, and near 0 for paper or a screen.
//
// Two details decide whether that holds up against a real camera.
//
// The yaw a frame is compared at comes from its neighbours, never from the
// frame itself. Yaw and miss are both read off the same nose, so the nose's
// own jitter would push both the same way in every frame, and the slope of
// noise against itself is 1. That alone made a photo shaken at two pixels of
// jitter pass as a head. The neighbours are 50 ms away, so they see the same
// head position with jitter of their own.
//
// And the slope alone says nothing about depth, only that nose and plane
// disagree consistently. A card curled around a vertical axis has a little
// depth too, and its slope is near 1 as well. What it cannot do is move the
// nose off the midline by much, so the cue also needs the (smoothed) yaw to
// have covered DepthMinRange: about 12 degrees of a real head turning.
//
// A card that is curled hard and turned far enough still gets there. So the
// last check asks whether the face turned as far as its nose says it did.
// Turning narrows the eyes against the eye to mouth distance by 1 - cos(turn),
// whatever the face is made of. A nose as flat as a real one can be (0.3 eye
// distances) that moved DepthMinRange can only have turned so far, and a face
// that narrowed a lot more than that was turned a lot more than that: it has
// less nose than any head. See depthConsistent().
void LivenessAnalyzer::addDepth(const LivenessFrame &frame)
{
constexpr size_t MaxFrames = 400;
if (m_depth.size() >= MaxFrames) {
return;
}
Face face;
face.points = frame.points;
const float emd = float(cv::norm(face.mouthMid() - face.eyeMid()));
const float shape = emd > 1.f ? face.interocular() / emd : 0.f;
m_depth.push_back({frame.points, frame.pose.yaw, 0.f, frame.pose.noseT, shape, 0.f});
// The frame two back now has two neighbours on each side.
const size_t n = m_depth.size();
if (n < 5) {
return;
}
const size_t j = n - 3;
DepthPoint &b = m_depth[j];
b.smoothYaw = (m_depth[j - 2].yaw + m_depth[j - 1].yaw + m_depth[j + 1].yaw + m_depth[j + 2].yaw) / 4.f;
b.smoothShape = (m_depth[j - 2].shape + m_depth[j - 1].shape + b.shape + m_depth[j + 1].shape + m_depth[j + 2].shape) / 5.f;
const cv::Point2f axis = b.points[LeftEye] - b.points[RightEye];
const float axisLen = std::max(1e-3f, float(cv::norm(axis)));
const cv::Point2f unit = axis * (1.f / axisLen);
for (size_t i = 2; i < j; ++i) {
const DepthPoint &a = m_depth[i];
const float dy = b.smoothYaw - a.smoothYaw;
if (std::abs(dy) < DepthMinTurn) {
continue;
}
const cv::Point2f src[4] = {a.points[RightEye], a.points[LeftEye], a.points[LeftMouth], a.points[RightMouth]};
const cv::Point2f dst[4] = {b.points[RightEye], b.points[LeftEye], b.points[LeftMouth], b.points[RightMouth]};
const cv::Mat h = cv::getPerspectiveTransform(src, dst);
if (h.empty()) {
continue;
}
std::vector<cv::Point2f> in{a.points[NoseTip]}, out;
cv::perspectiveTransform(in, out, h);
const cv::Point2f miss = b.points[NoseTip] - out[0];
const float rx = (miss.x * unit.x + miss.y * unit.y) / axisLen;
m_depthNum += double(rx) * dy;
m_depthDen += double(dy) * dy;
++m_depthPairs;
}
}
float LivenessAnalyzer::depthRatio() const
{
return m_depthDen > 0 ? float(m_depthNum / m_depthDen) : 0.f;
}
// The spread of the smoothed yaw, from the 10th to the 90th percentile, so a
// single bad frame cannot stretch it.
float LivenessAnalyzer::depthRange() const
{
if (m_depth.size() < 5) {
return 0;
}
std::vector<float> ys;
for (size_t k = 2; k + 2 < m_depth.size(); ++k) {
ys.push_back(m_depth[k].smoothYaw);
}
if (ys.size() < 3) {
return 0;
}
std::sort(ys.begin(), ys.end());
const auto at = [&](double q) {
return ys[size_t(q * double(ys.size() - 1))];
};
return at(0.9) - at(0.1);
}
bool LivenessAnalyzer::depthConsistent() const
{
// Nodding also changes the eye to mouth distance, so only frames at the
// head's usual pitch are compared. A card has no pitch of its own to read
// (its nose is printed on), so none of its frames are left out.
std::vector<float> noseTs;
for (size_t k = 2; k + 2 < m_depth.size(); ++k) {
noseTs.push_back(m_depth[k].noseT);
}
if (noseTs.size() < 3) {
return false;
}
std::vector<float> sortedT = noseTs;
std::sort(sortedT.begin(), sortedT.end());
const float medianT = sortedT[sortedT.size() / 2];
std::vector<float> shapes;
for (size_t k = 2; k + 2 < m_depth.size(); ++k) {
if (std::abs(m_depth[k].noseT - medianT) < 0.05f) {
shapes.push_back(m_depth[k].smoothShape);
}
}
if (shapes.size() < 3) {
return false;
}
std::sort(shapes.begin(), shapes.end());
const float widest = shapes[size_t(0.95 * double(shapes.size() - 1))];
const float narrowest = shapes[size_t(0.10 * double(shapes.size() - 1))];
if (widest <= 0) {
return false;
}
const float narrowed = 1.f - narrowest / widest;
const float sinTurn = std::min(1.f, depthRange() / DepthFlattestNose);
const float allowed = 1.f - std::sqrt(1.f - sinTurn * sinTurn);
return narrowed <= allowed + DepthShapeSlack;
}
bool LivenessAnalyzer::blinkSeen(float *strength) const
{
*strength = 0;
std::vector<const LivenessFrame *> s;
for (const LivenessFrame &f : m_frames) {
if (f.eyes.valid) {
s.push_back(&f);
}
}
if (s.size() < 6) {
return false;
}
std::vector<float> values;
for (const LivenessFrame *f : s) {
values.push_back(f->eyes.openness);
}
std::vector<float> sorted = values;
std::sort(sorted.begin(), sorted.end());
const float baseline = sorted[size_t(0.7 * double(sorted.size() - 1))];
// An eye this narrow is not one the measure works on (heavy lids, very
// dark eyes on dark skin, a camera that is too soft). Better to abstain
// than to guess.
if (baseline < 0.15f) {
return false;
}
*strength = std::clamp(1.f - sorted.front() / baseline, 0.f, 1.f) / (1.f - BlinkDip);
const size_t n = s.size();
for (size_t i = 1; i + 1 < n; ++i) {
if (values[i] >= BlinkDip * baseline) {
continue;
}
// The run of closed frames around this one.
size_t a = i, b = i;
while (a > 0 && values[a - 1] < BlinkOpen * baseline) {
--a;
}
while (b + 1 < n && values[b + 1] < BlinkOpen * baseline) {
++b;
}
if (a == 0 || b + 1 >= n) {
continue;
}
// Open right before and right after, within the time a blink takes
// (a real one is 100 to 400 ms; slower is somebody closing their
// eyes, or a picture being tilted away and back).
const LivenessFrame *before = s[a - 1];
const LivenessFrame *after = s[b + 1];
if (after->t - before->t > 600) {
continue;
}
// The rest of the face held still. A picture being moved blurs and
// shifts everything at once; a blink only moves the lids.
const cv::Point2f moved = (after->points[NoseTip] - before->points[NoseTip]);
const float iod = std::max(1.f, before->interocular);
if (float(cv::norm(moved)) > 0.15f * iod) {
continue;
}
if (std::abs(after->interocular - before->interocular) > 0.08f * iod) {
continue;
}
bool steadyLight = true;
for (size_t k = a; k <= b; ++k) {
const float ratio = s[k]->eyes.skin / std::max(1.f, before->eyes.skin);
steadyLight = steadyLight && ratio > 0.9f && ratio < 1.1f;
}
if (!steadyLight) {
continue;
}
// Both eyes together. One eye going dark on its own is a shadow or a
// hand, not a blink.
bool both = false;
for (size_t k = a; k <= b && !both; ++k) {
both = s[k]->eyes.left < 0.7f * baseline && s[k]->eyes.right < 0.7f * baseline;
}
if (both) {
return true;
}
}
return false;
}
LivenessReading LivenessAnalyzer::reading() const
{
LivenessReading r;
r.frames = int(m_frames.size());
const int seen = std::max(1, m_total);
const float glareShare = float(m_glareHits) / float(seen);
const float deviceShare = float(m_deviceHits) / float(seen);
r.glare = std::min(glareShare / 0.3f, float(m_glareHits) / 3.f);
r.device = std::min(deviceShare / 0.3f, float(m_deviceHits) / 3.f);
if (m_glareHits >= 3 && glareShare >= 0.3f) {
r.denied = true;
r.deniedBy = QStringLiteral("glare");
} else if (m_deviceHits >= 3 && deviceShare >= 0.3f) {
r.denied = true;
r.deniedBy = QStringLiteral("device");
}
r.depth = m_depthFired ? 1.f
: std::clamp(depthRatio() / DepthMinRatio, 0.f, 1.f) * std::clamp(depthRange() / DepthMinRange, 0.f, 1.f);
float strength = 0;
blinkSeen(&strength);
r.blink = m_blinkFired ? 1.f : std::min(strength, 0.99f);
if (m_depthFired) {
r.confirmed = true;
r.confirmedBy = QStringLiteral("depth");
} else if (m_blinkFired) {
r.confirmed = true;
r.confirmedBy = QStringLiteral("blink");
}
return r;
}
+154
View File
@@ -0,0 +1,154 @@
// SPDX-License-Identifier: GPL-3.0-or-later
//
// Telling a face from a picture of one.
//
// A webcam sees a flat image either way, so there is no single test that
// settles it. What is here follows the model Glance (the macOS face unlock)
// arrived at after trying and dropping a dozen weaker signals: a few cues,
// each one decisive on its own, split into two kinds.
//
// Deny cues are evidence of a fake. Either one fails the scan outright, and
// a match that already happened does not outvote it.
//
// glare a phone screen or a glossy print throws back one large, flat,
// colourless highlight. Skin shines in small scattered spots.
// device the straight edges of a phone or a tablet around the face.
//
// Confirm cues are evidence of a real head. Any one of them is enough, and
// their absence is never held against anybody on its own: a person can sit
// still and not blink for a few seconds.
//
// depth when the head turns, the tip of the nose moves further than a
// flat face would let it. The eyes and the corners of the mouth lie
// close to one plane, so four of them predict exactly where the
// nose of a flat picture has to go. A real nose, about a third of
// an eye distance in front of that plane, misses the prediction by
// about as much as the head turned.
// blink the dark of the eye shrinks to a line and comes back, within
// the fraction of a second a blink takes, while the rest of the face
// holds still.
//
// "Light" uses the deny cues. "Heavy" also needs one confirm cue before it
// lets anybody in.
#pragma once
#include "settings.h"
#include "vision.h"
#include <QString>
#include <deque>
struct GlareSample {
bool valid = false;
// Share of the face that is a colourless highlight.
float fraction = 0;
// Share of all highlight pixels that fall in the fullest of 8x8 cells.
// One slab of glass reflects into one place; skin sparkles everywhere.
float cluster = 0;
};
struct EyeSample {
bool valid = false;
// How much of the eye's height is dark (iris, pupil), per eye and on
// average. Falls towards zero when the lid comes down.
float left = 0;
float right = 0;
float openness = 0;
// Brightness of the skin under the eyes, to tell a blink from a change
// in the light.
float skin = 0;
};
struct LivenessFrame {
double t = 0;
std::array<cv::Point2f, 5> points{};
float interocular = 0;
HeadPose pose;
GlareSample glare;
bool device = false;
EyeSample eyes;
};
GlareSample measureGlare(const cv::Mat &bgr, const Face &face);
EyeSample measureEyes(const cv::Mat &bgr, const Face &face);
bool detectDevice(const cv::Mat &bgr, const Face &face);
LivenessFrame measureFrame(const cv::Mat &bgr, const Face &face, double timestampMs);
struct LivenessReading {
int frames = 0;
bool denied = false;
QString deniedBy;
bool confirmed = false;
QString confirmedBy;
// For the test screen: how close each cue is to firing, 0 to 1 and more.
float glare = 0;
float device = 0;
float depth = 0;
float blink = 0;
};
class LivenessAnalyzer
{
public:
void reset();
void add(const LivenessFrame &frame);
LivenessReading reading() const;
int frameCount() const
{
return int(m_frames.size());
}
// Tuning, in one place. See liveness.cpp for where each number comes
// from.
static constexpr double WindowMs = 4000;
static constexpr float GlareFraction = 0.012f;
static constexpr float GlareCluster = 0.55f;
static constexpr float DepthMinTurn = 0.05f;
static constexpr float DepthMinRange = 0.10f;
static constexpr float DepthMinRatio = 0.65f;
static constexpr float DepthFlattestNose = 0.30f;
static constexpr float DepthShapeSlack = 0.02f;
static constexpr float BlinkDip = 0.55f;
static constexpr float BlinkOpen = 0.8f;
// The depth cue's workings, for the test screen and the tests.
float depthRatio() const;
float depthRange() const;
bool depthConsistent() const;
private:
void addDepth(const LivenessFrame &frame);
bool blinkSeen(float *strength) const;
std::deque<LivenessFrame> m_frames;
int m_total = 0;
// The depth cue looks at the whole scan rather than the window, and is
// worked out a frame at a time so it stays cheap.
struct DepthPoint {
std::array<cv::Point2f, 5> points;
float yaw;
float smoothYaw;
float noseT;
// Eye distance over eye to mouth distance: shrinks as the face turns
// away from the camera, whatever the face is made of.
float shape;
float smoothShape;
};
std::vector<DepthPoint> m_depth;
double m_depthNum = 0;
double m_depthDen = 0;
int m_depthPairs = 0;
int m_glareHits = 0;
int m_deviceHits = 0;
bool m_depthFired = false;
bool m_blinkFired = false;
};
+81
View File
@@ -0,0 +1,81 @@
// SPDX-License-Identifier: GPL-3.0-or-later
#include "settings.h"
#include "keyvalue.h"
double Settings::threshold() const
{
// Cosine similarity between two SFace embeddings. Photos of the same
// person taken years apart land around 0.75, different people below 0.3.
// OpenCV's own recommendation (0.363) is tuned for telling apart photos
// in a data set; this guards a login, so the bar is higher.
switch (strictness) {
case Strictness::Relaxed:
return 0.42;
case Strictness::Normal:
return 0.50;
case Strictness::Strict:
return 0.58;
}
return 0.50;
}
Settings Settings::load(const QString &path)
{
Settings s;
const KeyValueFile kv = KeyValueFile::load(path);
s.camera = kv.value(QStringLiteral("Camera"), s.camera);
const QString liveness = kv.value(QStringLiteral("Liveness")).toLower();
if (liveness == u"off") {
s.liveness = LivenessMode::Off;
} else if (liveness == u"light") {
s.liveness = LivenessMode::Light;
} else if (liveness == u"heavy") {
s.liveness = LivenessMode::Heavy;
}
const QString strictness = kv.value(QStringLiteral("Strictness")).toLower();
if (strictness == u"relaxed") {
s.strictness = Strictness::Relaxed;
} else if (strictness == u"normal") {
s.strictness = Strictness::Normal;
} else if (strictness == u"strict") {
s.strictness = Strictness::Strict;
}
s.attention = kv.boolean(QStringLiteral("Attention"), s.attention);
s.scanSeconds = kv.integer(QStringLiteral("ScanSeconds"), s.scanSeconds, 2, 15);
s.maxFailures = kv.integer(QStringLiteral("MaxFailures"), s.maxFailures, 1, 20);
s.lockoutMinutes = kv.integer(QStringLiteral("LockoutMinutes"), s.lockoutMinutes, 1, 24 * 60);
s.skipLidClosed = kv.boolean(QStringLiteral("SkipLidClosed"), s.skipLidClosed);
s.adapt = kv.boolean(QStringLiteral("Adapt"), s.adapt);
return s;
}
QString Settings::livenessName(LivenessMode mode)
{
switch (mode) {
case LivenessMode::Off:
return QStringLiteral("off");
case LivenessMode::Light:
return QStringLiteral("light");
case LivenessMode::Heavy:
return QStringLiteral("heavy");
}
return {};
}
QString Settings::strictnessName(Strictness strictness)
{
switch (strictness) {
case Strictness::Relaxed:
return QStringLiteral("relaxed");
case Strictness::Normal:
return QStringLiteral("normal");
case Strictness::Strict:
return QStringLiteral("strict");
}
return {};
}
+55
View File
@@ -0,0 +1,55 @@
// SPDX-License-Identifier: GPL-3.0-or-later
//
// The system settings, /etc/plasma-face-unlock/config.
//
// These are the ones that decide how hard it is to get in: which camera, how
// strict the match is, whether a photo is checked for. They belong to root
// for that reason. A setting a user process could change would be a setting
// any program running as that user could lower before asking sudo for help.
#pragma once
#include <QString>
enum class LivenessMode {
Off,
// Deny cues only: glare off a screen, the edge of a phone around the face.
Light,
// Deny cues, and a sign of life on top: a blink, or the nose moving like
// a nose does when the head turns.
Heavy,
};
enum class Strictness {
Relaxed,
Normal,
Strict,
};
struct Settings {
// A /dev/video path, "auto", or for testing "file:<video>" and
// "images:<directory>".
QString camera = QStringLiteral("auto");
LivenessMode liveness = LivenessMode::Heavy;
Strictness strictness = Strictness::Normal;
// Only a face that looks at the screen with its eyes open counts.
bool attention = true;
// How long one scan looks before it gives up.
int scanSeconds = 5;
// Failed scans in a row with a face in view before face unlock stops
// until the password has been used, and how long that lasts at most.
int maxFailures = 5;
int lockoutMinutes = 15;
// A laptop with the lid shut has its camera looking at the keyboard.
bool skipLidClosed = true;
// Take in a little of each confident unlock, so a new haircut or a pair
// of glasses does not need a new setup. Face ID does the same.
bool adapt = true;
double threshold() const;
static Settings load(const QString &path);
static QString livenessName(LivenessMode mode);
static QString strictnessName(Strictness strictness);
};
+206
View File
@@ -0,0 +1,206 @@
// SPDX-License-Identifier: GPL-3.0-or-later
#include "store.h"
#include <QDir>
#include <QFile>
#include <QJsonArray>
#include <QJsonDocument>
#include <QJsonObject>
#include <QRandomGenerator>
#include <QSaveFile>
namespace
{
constexpr int FormatVersion = 1;
QByteArray packEmbedding(const Embedding &e)
{
QByteArray raw(reinterpret_cast<const char *>(e.data()), qsizetype(e.size() * sizeof(float)));
return raw.toBase64();
}
Embedding unpackEmbedding(const QString &b64)
{
const QByteArray raw = QByteArray::fromBase64(b64.toLatin1());
Embedding e;
if (raw.size() != qsizetype(Vision::EmbeddingSize * sizeof(float))) {
return e;
}
e.resize(Vision::EmbeddingSize);
memcpy(e.data(), raw.constData(), size_t(raw.size()));
return e;
}
} // namespace
int Identity::adaptiveCount() const
{
int n = 0;
for (const FaceSample &s : samples) {
n += s.pose == u"adaptive";
}
return n;
}
FaceStore::FaceStore(const QString &stateDir)
: m_dir(QDir(stateDir).filePath(QStringLiteral("users")))
{
}
QString FaceStore::fileFor(uint uid) const
{
return QDir(m_dir).filePath(QStringLiteral("%1.json").arg(uid));
}
QList<Identity> FaceStore::load(uint uid, QString *error) const
{
QList<Identity> faces;
QFile file(fileFor(uid));
if (!file.exists()) {
return faces;
}
if (!file.open(QIODevice::ReadOnly)) {
if (error) {
*error = file.errorString();
}
return faces;
}
QJsonParseError parseError;
const QJsonDocument doc = QJsonDocument::fromJson(file.readAll(), &parseError);
if (!doc.isObject()) {
if (error) {
*error = parseError.errorString();
}
return faces;
}
const QJsonArray list = doc.object().value(u"faces").toArray();
for (const QJsonValue &v : list) {
const QJsonObject o = v.toObject();
Identity id;
id.id = o.value(u"id").toString();
id.name = o.value(u"name").toString();
id.created = qint64(o.value(u"created").toDouble());
id.enabled = o.value(u"enabled").toBool(true);
id.noseT = float(o.value(u"noseT").toDouble(0.55));
id.eyes = float(o.value(u"eyes").toDouble(0));
for (const QJsonValue &sv : o.value(u"samples").toArray()) {
const QJsonObject so = sv.toObject();
FaceSample s;
s.embedding = unpackEmbedding(so.value(u"e").toString());
s.pose = so.value(u"pose").toString();
s.time = qint64(so.value(u"t").toDouble());
if (!s.embedding.empty()) {
id.samples.append(s);
}
}
if (!id.id.isEmpty() && !id.samples.isEmpty()) {
faces.append(id);
}
}
return faces;
}
bool FaceStore::save(uint uid, const QList<Identity> &faces, QString *error) const
{
if (!QDir().mkpath(m_dir)) {
*error = QStringLiteral("cannot create %1").arg(m_dir);
return false;
}
QFile::setPermissions(m_dir, QFileDevice::ReadOwner | QFileDevice::WriteOwner | QFileDevice::ExeOwner);
if (faces.isEmpty()) {
remove(uid);
return true;
}
QJsonArray list;
for (const Identity &id : faces) {
QJsonArray samples;
for (const FaceSample &s : id.samples) {
samples.append(QJsonObject{
{QStringLiteral("e"), QString::fromLatin1(packEmbedding(s.embedding))},
{QStringLiteral("pose"), s.pose},
{QStringLiteral("t"), double(s.time)},
});
}
list.append(QJsonObject{
{QStringLiteral("id"), id.id},
{QStringLiteral("name"), id.name},
{QStringLiteral("created"), double(id.created)},
{QStringLiteral("enabled"), id.enabled},
{QStringLiteral("noseT"), double(id.noseT)},
{QStringLiteral("eyes"), double(id.eyes)},
{QStringLiteral("samples"), samples},
});
}
const QJsonObject root{
{QStringLiteral("version"), FormatVersion},
{QStringLiteral("faces"), list},
};
QSaveFile file(fileFor(uid));
if (!file.open(QIODevice::WriteOnly)) {
*error = file.errorString();
return false;
}
file.setPermissions(QFileDevice::ReadOwner | QFileDevice::WriteOwner);
file.write(QJsonDocument(root).toJson(QJsonDocument::Compact));
if (!file.commit()) {
*error = file.errorString();
return false;
}
return true;
}
bool FaceStore::remove(uint uid) const
{
return QFile::remove(fileFor(uid));
}
QString FaceStore::newId()
{
return QString::number(QRandomGenerator::system()->generate64() & 0xffffffffffffULL, 16);
}
FaceMatch FaceStore::bestMatch(const QList<Identity> &faces, const Embedding &e)
{
FaceMatch best;
for (int i = 0; i < faces.size(); ++i) {
const Identity &id = faces.at(i);
if (!id.enabled) {
continue;
}
for (int k = 0; k < id.samples.size(); ++k) {
const float s = Vision::similarity(id.samples.at(k).embedding, e);
if (s > best.score) {
best = {s, i, k};
}
}
}
return best;
}
bool FaceStore::adapt(Identity &identity, const Embedding &e, qint64 now)
{
// Something the samples already cover adds nothing but weight.
float closest = -1;
for (const FaceSample &s : std::as_const(identity.samples)) {
closest = std::max(closest, Vision::similarity(s.embedding, e));
}
if (closest >= 0.9f) {
return false;
}
if (identity.adaptiveCount() >= MaxAdaptive) {
for (qsizetype i = 0; i < identity.samples.size(); ++i) {
if (identity.samples.at(i).pose == u"adaptive") {
identity.samples.removeAt(i);
break;
}
}
}
identity.samples.append(FaceSample{e, QStringLiteral("adaptive"), now});
return true;
}
+72
View File
@@ -0,0 +1,72 @@
// SPDX-License-Identifier: GPL-3.0-or-later
//
// The face data.
//
// One file per user under /var/lib/plasma-face-unlock/users, named after the
// numeric user id so a rename cannot hand one person's faces to another. Only
// root can read or write the directory. There are no pictures in it: every
// sample is the 128 numbers the recognizer made of one frame, and the frame
// itself was never written anywhere.
#pragma once
#include "vision.h"
#include <QList>
#include <QString>
struct FaceSample {
Embedding embedding;
// Which way the head pointed when it was taken: "center", one of the
// eight directions ("up", "up-right", ...), or "adaptive" for one taken
// in from a successful unlock.
QString pose;
qint64 time = 0;
};
struct Identity {
QString id;
QString name;
qint64 created = 0;
bool enabled = true;
// Where this person's nose sits for a level head (see HeadPose), and how
// open their eyes measure when they look at the camera. Both are what
// the attention check compares against.
float noseT = 0.55f;
float eyes = 0;
QList<FaceSample> samples;
int adaptiveCount() const;
};
struct FaceMatch {
float score = -1;
int identity = -1;
int sample = -1;
};
class FaceStore
{
public:
explicit FaceStore(const QString &stateDir);
QList<Identity> load(uint uid, QString *error = nullptr) const;
bool save(uint uid, const QList<Identity> &faces, QString *error) const;
bool remove(uint uid) const;
static QString newId();
// The best sample over every enabled identity.
static FaceMatch bestMatch(const QList<Identity> &faces, const Embedding &e);
// Adds what a confident unlock saw, if it adds anything, and keeps at most
// MaxAdaptive of those per identity (the oldest go first). The samples
// from the setup are never touched.
static bool adapt(Identity &identity, const Embedding &e, qint64 now);
static constexpr int MaxAdaptive = 12;
private:
QString fileFor(uint uid) const;
QString m_dir;
};
+219
View File
@@ -0,0 +1,219 @@
// SPDX-License-Identifier: GPL-3.0-or-later
#include "vision.h"
#include <QDir>
#include <QFile>
#include <QFileInfo>
#include <opencv2/core/utils/logger.hpp>
#include <opencv2/imgproc.hpp>
#include <algorithm>
#include <cmath>
#include <numeric>
namespace
{
const char DetectorFile[] = "face_detection_yunet_2023mar.onnx";
const char RecognizerFile[] = "face_recognition_sface_2021dec.onnx";
// A face whose eyes are closer together than this is too far away for the
// recognizer to see detail in, and the liveness checks drown in noise.
constexpr float MinInterocular = 28.f;
cv::Point2f rotateAround(cv::Point2f p, cv::Point2f centre, float angle)
{
const float c = std::cos(angle);
const float s = std::sin(angle);
const cv::Point2f d = p - centre;
return {centre.x + d.x * c - d.y * s, centre.y + d.x * s + d.y * c};
}
} // namespace
float Face::interocular() const
{
return float(cv::norm(points[LeftEye] - points[RightEye]));
}
cv::Point2f Face::eyeMid() const
{
return (points[RightEye] + points[LeftEye]) * 0.5f;
}
cv::Point2f Face::mouthMid() const
{
return (points[RightMouth] + points[LeftMouth]) * 0.5f;
}
HeadPose estimatePose(const Face &face)
{
HeadPose pose;
const cv::Point2f eyes = face.points[LeftEye] - face.points[RightEye];
pose.roll = std::atan2(eyes.y, eyes.x);
// Level the face first, so a tilted head does not read as a turned one.
const cv::Point2f centre = face.eyeMid();
std::array<cv::Point2f, 5> p;
for (int i = 0; i < 5; ++i) {
p[i] = rotateAround(face.points[i], centre, -pose.roll);
}
const cv::Point2f eyeMid = (p[RightEye] + p[LeftEye]) * 0.5f;
const cv::Point2f mouthMid = (p[RightMouth] + p[LeftMouth]) * 0.5f;
const float iod = std::max(1.f, float(cv::norm(p[LeftEye] - p[RightEye])));
const float span = mouthMid.y - eyeMid.y;
if (span < 1.f) {
return pose;
}
const cv::Point2f nose = p[NoseTip];
const float t = (nose.y - eyeMid.y) / span;
const float midlineX = eyeMid.x + (mouthMid.x - eyeMid.x) * t;
// The eyes are reported person-right first, which is image-left on an
// unmirrored frame. A nose moving image-right is a head turning to the
// person's left.
pose.yaw = (nose.x - midlineX) / iod;
pose.noseT = t;
return pose;
}
float yawDegrees(float yaw)
{
return float(std::asin(std::clamp(yaw / 0.5f, -1.f, 1.f)) * 180.0 / M_PI);
}
FaceQuality assessQuality(const cv::Mat &bgr, const Face &face)
{
FaceQuality q;
q.tooSmall = face.interocular() < MinInterocular;
// The middle of the face, without hair and background at the edges.
const float iod = face.interocular();
const cv::Point2f c = (face.eyeMid() + face.mouthMid()) * 0.5f;
cv::Rect roi(cv::Point(int(c.x - iod), int(c.y - iod)), cv::Size(int(2 * iod), int(2 * iod)));
roi &= cv::Rect(0, 0, bgr.cols, bgr.rows);
if (roi.width < 8 || roi.height < 8) {
q.tooSmall = true;
return q;
}
cv::Mat grey;
cv::cvtColor(bgr(roi), grey, cv::COLOR_BGR2GRAY);
q.brightness = float(cv::mean(grey)[0]);
// Sharpness at a fixed scale, so a face far away and one up close are
// judged the same.
cv::Mat small, lap;
cv::resize(grey, small, cv::Size(96, 96), 0, 0, cv::INTER_AREA);
cv::Laplacian(small, lap, CV_32F);
cv::Scalar mean, stddev;
cv::meanStdDev(lap, mean, stddev);
q.sharpness = float(stddev[0] * stddev[0]);
q.tooDark = q.brightness < 40;
q.tooBright = q.brightness > 230;
q.blurry = q.sharpness < 12;
return q;
}
bool Vision::load(const QString &modelDir, QString *error)
{
// The new DNN engine in OpenCV 5 prints a warning for every network it
// loads about targets it does not support yet. Nothing here asks for one.
cv::utils::logging::setLogLevel(cv::utils::logging::LOG_LEVEL_ERROR);
const QDir dir(modelDir);
const QString detector = dir.filePath(QLatin1String(DetectorFile));
const QString recognizer = dir.filePath(QLatin1String(RecognizerFile));
for (const QString &f : {detector, recognizer}) {
if (!QFileInfo::exists(f)) {
*error = QStringLiteral("model missing: %1").arg(f);
return false;
}
}
try {
m_inputSize = cv::Size(640, 480);
m_detector = cv::FaceDetectorYN::create(QFile::encodeName(detector).toStdString(), "", m_inputSize, 0.75f, 0.3f, 20);
m_recognizer = cv::FaceRecognizerSF::create(QFile::encodeName(recognizer).toStdString(), "");
} catch (const cv::Exception &e) {
*error = QString::fromStdString(e.what());
m_detector.reset();
m_recognizer.reset();
return false;
}
return true;
}
std::vector<Face> Vision::detect(const cv::Mat &bgr)
{
std::vector<Face> result;
if (!m_detector || bgr.empty()) {
return result;
}
if (bgr.size() != m_inputSize) {
m_inputSize = bgr.size();
m_detector->setInputSize(m_inputSize);
}
cv::Mat rows;
try {
m_detector->detect(bgr, rows);
} catch (const cv::Exception &) {
return result;
}
for (int i = 0; i < rows.rows; ++i) {
const float *r = rows.ptr<float>(i);
Face f;
f.box = cv::Rect2f(r[0], r[1], r[2], r[3]);
for (int k = 0; k < 5; ++k) {
f.points[k] = cv::Point2f(r[4 + 2 * k], r[5 + 2 * k]);
}
f.score = r[14];
f.row = rows.row(i).clone();
result.push_back(std::move(f));
}
std::sort(result.begin(), result.end(), [](const Face &a, const Face &b) {
return a.box.area() > b.box.area();
});
return result;
}
Embedding Vision::embed(const cv::Mat &bgr, const Face &face)
{
Embedding out;
if (!m_recognizer) {
return out;
}
try {
cv::Mat aligned, feature;
m_recognizer->alignCrop(bgr, face.row, aligned);
m_recognizer->feature(aligned, feature);
feature = feature.reshape(1, 1);
if (feature.cols != EmbeddingSize) {
return out;
}
const double norm = cv::norm(feature);
if (norm <= 0) {
return out;
}
out.resize(EmbeddingSize);
for (int i = 0; i < EmbeddingSize; ++i) {
out[i] = float(feature.at<float>(0, i) / norm);
}
} catch (const cv::Exception &) {
out.clear();
}
return out;
}
float Vision::similarity(const Embedding &a, const Embedding &b)
{
if (a.size() != b.size() || a.empty()) {
return -1.f;
}
return std::inner_product(a.begin(), a.end(), b.begin(), 0.f);
}
+108
View File
@@ -0,0 +1,108 @@
// SPDX-License-Identifier: GPL-3.0-or-later
//
// Finding a face and turning it into numbers.
//
// Two small networks from the OpenCV model zoo do the work. YuNet finds faces
// and five points on each (the eyes, the tip of the nose, the corners of the
// mouth). SFace turns an aligned crop of one face into 128 numbers, and two
// crops of the same person give numbers that point the same way. Both run on
// the CPU through OpenCV's own DNN module, in a few milliseconds each.
//
// Everything else here is plain geometry on those five points.
#pragma once
#include <QString>
#include <opencv2/core.hpp>
#include <opencv2/objdetect/face.hpp>
#include <array>
#include <vector>
// The five points, in the order YuNet reports them. "Right" and "left" are
// the person's own, so on an unmirrored picture the right eye is on the left.
enum Landmark {
RightEye = 0,
LeftEye = 1,
NoseTip = 2,
RightMouth = 3,
LeftMouth = 4,
};
struct Face {
cv::Rect2f box;
std::array<cv::Point2f, 5> points;
float score = 0;
// YuNet's own row, which the recognizer wants back for its alignment.
cv::Mat row;
float interocular() const;
cv::Point2f eyeMid() const;
cv::Point2f mouthMid() const;
};
// Where the head points, read off the five points alone.
//
// yaw is the nose's sideways offset from the line through the middle of the
// eyes and the middle of the mouth, in interocular distances. A nose sits
// about half an interocular distance in front of the face, so this is close to
// 0.5 * sin(head yaw). Positive when the head turns to the person's left.
//
// noseT is how far down the nose tip sits between the eye line (0) and the
// mouth line (1). It changes with pitch, but where it sits for a level head is
// different for every face, so it only means something next to the same
// person's own resting value.
struct HeadPose {
float roll = 0;
float yaw = 0;
float noseT = 0;
};
HeadPose estimatePose(const Face &face);
// Degrees, for people. The estimate is rough by nature.
float yawDegrees(float yaw);
struct FaceQuality {
float brightness = 0;
float sharpness = 0;
bool tooSmall = false;
bool tooDark = false;
bool tooBright = false;
bool blurry = false;
bool ok() const
{
return !tooSmall && !tooDark && !tooBright && !blurry;
}
};
FaceQuality assessQuality(const cv::Mat &bgr, const Face &face);
using Embedding = std::vector<float>;
class Vision
{
public:
bool load(const QString &modelDir, QString *error);
bool isLoaded() const
{
return m_detector && m_recognizer;
}
// Faces sorted by size, largest first.
std::vector<Face> detect(const cv::Mat &bgr);
// Unit length, so the similarity of two is their dot product.
Embedding embed(const cv::Mat &bgr, const Face &face);
static float similarity(const Embedding &a, const Embedding &b);
static constexpr int EmbeddingSize = 128;
private:
cv::Ptr<cv::FaceDetectorYN> m_detector;
cv::Ptr<cv::FaceRecognizerSF> m_recognizer;
cv::Size m_inputSize;
};