feat: add plasma-face-unlock
This commit is contained in:
commit
f671acc93b
105 files changed
+13862
No files matched your search
@@ -0,0 +1,285 @@
|
||||
// SPDX-License-Identifier: GPL-3.0-or-later
|
||||
|
||||
#include "camera.h"
|
||||
|
||||
#include <QDir>
|
||||
#include <QFile>
|
||||
#include <QFileInfo>
|
||||
|
||||
#include <opencv2/imgcodecs.hpp>
|
||||
#include <opencv2/imgproc.hpp>
|
||||
|
||||
#include <fcntl.h>
|
||||
#include <linux/videodev2.h>
|
||||
#include <sys/ioctl.h>
|
||||
#include <unistd.h>
|
||||
|
||||
#include <algorithm>
|
||||
#include <thread>
|
||||
|
||||
using namespace std::chrono;
|
||||
|
||||
namespace
|
||||
{
|
||||
constexpr int Width = 640;
|
||||
constexpr int Height = 480;
|
||||
// Pictures stand in for a camera at this rate, each held for a while so the
|
||||
// checks see a still face the way they would see a person holding still.
|
||||
constexpr double ImageFrameMs = 1000.0 / 15.0;
|
||||
constexpr int ImageRepeat = 12;
|
||||
|
||||
QString sysName(const QString &device)
|
||||
{
|
||||
QFile file(QStringLiteral("/sys/class/video4linux/%1/name").arg(QFileInfo(device).fileName()));
|
||||
if (!file.open(QIODevice::ReadOnly)) {
|
||||
return {};
|
||||
}
|
||||
return QString::fromUtf8(file.readAll()).trimmed();
|
||||
}
|
||||
|
||||
// Opens the node just long enough to ask what it is. Neither this nor the
|
||||
// format list below starts streaming, so it does not switch the light on.
|
||||
bool probe(const QString &path, CameraInfo *info)
|
||||
{
|
||||
const int fd = ::open(QFile::encodeName(path).constData(), O_RDONLY | O_NONBLOCK | O_CLOEXEC);
|
||||
if (fd < 0) {
|
||||
return false;
|
||||
}
|
||||
|
||||
v4l2_capability cap{};
|
||||
bool ok = ::ioctl(fd, VIDIOC_QUERYCAP, &cap) == 0;
|
||||
if (ok) {
|
||||
const quint32 caps = (cap.capabilities & V4L2_CAP_DEVICE_CAPS) ? cap.device_caps : cap.capabilities;
|
||||
ok = (caps & V4L2_CAP_VIDEO_CAPTURE) && !(caps & V4L2_CAP_META_CAPTURE);
|
||||
}
|
||||
|
||||
bool colour = false;
|
||||
bool grey = false;
|
||||
if (ok) {
|
||||
for (quint32 i = 0;; ++i) {
|
||||
v4l2_fmtdesc fmt{};
|
||||
fmt.index = i;
|
||||
fmt.type = V4L2_BUF_TYPE_VIDEO_CAPTURE;
|
||||
if (::ioctl(fd, VIDIOC_ENUM_FMT, &fmt) != 0) {
|
||||
break;
|
||||
}
|
||||
switch (fmt.pixelformat) {
|
||||
case V4L2_PIX_FMT_GREY:
|
||||
case V4L2_PIX_FMT_Y10:
|
||||
case V4L2_PIX_FMT_Y12:
|
||||
case V4L2_PIX_FMT_Y16:
|
||||
grey = true;
|
||||
break;
|
||||
default:
|
||||
colour = true;
|
||||
break;
|
||||
}
|
||||
}
|
||||
ok = colour || grey;
|
||||
}
|
||||
::close(fd);
|
||||
|
||||
if (ok && info) {
|
||||
info->path = path;
|
||||
info->name = sysName(path);
|
||||
if (info->name.isEmpty()) {
|
||||
info->name = QString::fromUtf8(reinterpret_cast<const char *>(cap.card));
|
||||
}
|
||||
info->infrared = grey && !colour;
|
||||
}
|
||||
return ok;
|
||||
}
|
||||
} // namespace
|
||||
|
||||
Camera::Camera() = default;
|
||||
|
||||
Camera::~Camera()
|
||||
{
|
||||
close();
|
||||
}
|
||||
|
||||
QList<CameraInfo> Camera::list()
|
||||
{
|
||||
QList<CameraInfo> result;
|
||||
const QDir dev(QStringLiteral("/dev"));
|
||||
QStringList nodes = dev.entryList({QStringLiteral("video*")}, QDir::System);
|
||||
std::sort(nodes.begin(), nodes.end(), [](const QString &a, const QString &b) {
|
||||
return a.mid(5).toInt() < b.mid(5).toInt();
|
||||
});
|
||||
for (const QString &node : std::as_const(nodes)) {
|
||||
CameraInfo info;
|
||||
if (probe(dev.filePath(node), &info)) {
|
||||
result.append(info);
|
||||
}
|
||||
}
|
||||
return result;
|
||||
}
|
||||
|
||||
QString Camera::autoPath()
|
||||
{
|
||||
const QList<CameraInfo> cameras = list();
|
||||
for (const CameraInfo &c : cameras) {
|
||||
if (!c.infrared) {
|
||||
return c.path;
|
||||
}
|
||||
}
|
||||
return cameras.isEmpty() ? QString() : cameras.first().path;
|
||||
}
|
||||
|
||||
bool Camera::open(const QString &spec, QString *error)
|
||||
{
|
||||
close();
|
||||
m_start = steady_clock::now();
|
||||
m_next = m_start;
|
||||
|
||||
if (spec.startsWith(u"images:")) {
|
||||
return openImages(spec.mid(7), error);
|
||||
}
|
||||
|
||||
if (spec.startsWith(u"file:")) {
|
||||
const QString path = spec.mid(5);
|
||||
if (!m_capture.open(QFile::encodeName(path).toStdString(), cv::CAP_ANY) || !m_capture.isOpened()) {
|
||||
*error = QStringLiteral("cannot open video file %1").arg(path);
|
||||
return false;
|
||||
}
|
||||
m_paced = true;
|
||||
m_open = true;
|
||||
m_description = QFileInfo(path).fileName();
|
||||
return true;
|
||||
}
|
||||
|
||||
QString path = spec;
|
||||
if (path.isEmpty() || path == u"auto") {
|
||||
path = autoPath();
|
||||
if (path.isEmpty()) {
|
||||
*error = QStringLiteral("no camera found");
|
||||
return false;
|
||||
}
|
||||
}
|
||||
|
||||
CameraInfo info;
|
||||
if (!probe(path, &info)) {
|
||||
*error = QStringLiteral("%1 is not a camera").arg(path);
|
||||
return false;
|
||||
}
|
||||
|
||||
if (!m_capture.open(QFile::encodeName(path).toStdString(), cv::CAP_V4L2) || !m_capture.isOpened()) {
|
||||
*error = QStringLiteral("cannot open %1 (in use by another program?)").arg(path);
|
||||
return false;
|
||||
}
|
||||
|
||||
// MJPEG gets a webcam its full frame rate over USB 2; raw YUYV at 640x480
|
||||
// often tops out at 15 fps. A camera that has no MJPEG ignores the request.
|
||||
if (!info.infrared) {
|
||||
m_capture.set(cv::CAP_PROP_FOURCC, cv::VideoWriter::fourcc('M', 'J', 'P', 'G'));
|
||||
}
|
||||
m_capture.set(cv::CAP_PROP_FRAME_WIDTH, Width);
|
||||
m_capture.set(cv::CAP_PROP_FRAME_HEIGHT, Height);
|
||||
m_capture.set(cv::CAP_PROP_FPS, 30);
|
||||
// A stale frame in the driver's queue is a picture of whoever sat there a
|
||||
// moment ago. Keep the queue as short as the driver allows.
|
||||
m_capture.set(cv::CAP_PROP_BUFFERSIZE, 1);
|
||||
|
||||
m_infrared = info.infrared;
|
||||
m_paced = false;
|
||||
m_open = true;
|
||||
m_description = QStringLiteral("%1 (%2)").arg(info.name, path);
|
||||
return true;
|
||||
}
|
||||
|
||||
bool Camera::openImages(const QString &dir, QString *error)
|
||||
{
|
||||
const QDir d(dir);
|
||||
const QStringList files = d.entryList({QStringLiteral("*.jpg"), QStringLiteral("*.jpeg"), QStringLiteral("*.png")},
|
||||
QDir::Files, QDir::Name);
|
||||
for (const QString &f : files) {
|
||||
cv::Mat img = cv::imread(QFile::encodeName(d.filePath(f)).toStdString(), cv::IMREAD_COLOR);
|
||||
if (img.empty()) {
|
||||
continue;
|
||||
}
|
||||
const double scale = double(Width) / std::max(img.cols, img.rows);
|
||||
if (scale < 1.0) {
|
||||
cv::resize(img, img, {}, scale, scale, cv::INTER_AREA);
|
||||
}
|
||||
m_images.append(img);
|
||||
}
|
||||
if (m_images.isEmpty()) {
|
||||
*error = QStringLiteral("no pictures in %1").arg(dir);
|
||||
return false;
|
||||
}
|
||||
m_imageIndex = 0;
|
||||
m_imageRepeat = 0;
|
||||
m_paced = true;
|
||||
m_open = true;
|
||||
m_description = QStringLiteral("pictures in %1").arg(dir);
|
||||
return true;
|
||||
}
|
||||
|
||||
void Camera::close()
|
||||
{
|
||||
if (m_capture.isOpened()) {
|
||||
m_capture.release();
|
||||
}
|
||||
m_images.clear();
|
||||
m_open = false;
|
||||
m_infrared = false;
|
||||
}
|
||||
|
||||
bool Camera::isOpen() const
|
||||
{
|
||||
return m_open;
|
||||
}
|
||||
|
||||
void Camera::pace(double frameMs)
|
||||
{
|
||||
m_next += duration_cast<steady_clock::duration>(duration<double, std::milli>(frameMs));
|
||||
const auto now = steady_clock::now();
|
||||
if (m_next > now) {
|
||||
std::this_thread::sleep_until(m_next);
|
||||
} else {
|
||||
m_next = now;
|
||||
}
|
||||
}
|
||||
|
||||
bool Camera::readImages(cv::Mat &bgr)
|
||||
{
|
||||
pace(ImageFrameMs);
|
||||
bgr = m_images.at(m_imageIndex).clone();
|
||||
if (++m_imageRepeat >= ImageRepeat) {
|
||||
m_imageRepeat = 0;
|
||||
m_imageIndex = (m_imageIndex + 1) % m_images.size();
|
||||
}
|
||||
return true;
|
||||
}
|
||||
|
||||
bool Camera::read(cv::Mat &bgr, double *timestampMs)
|
||||
{
|
||||
if (!m_open) {
|
||||
return false;
|
||||
}
|
||||
|
||||
bool ok;
|
||||
if (!m_images.isEmpty()) {
|
||||
ok = readImages(bgr);
|
||||
} else {
|
||||
if (m_paced) {
|
||||
double fps = m_capture.get(cv::CAP_PROP_FPS);
|
||||
if (!(fps > 1 && fps < 240)) {
|
||||
fps = 30;
|
||||
}
|
||||
pace(1000.0 / fps);
|
||||
}
|
||||
ok = m_capture.read(bgr) && !bgr.empty();
|
||||
}
|
||||
if (!ok) {
|
||||
return false;
|
||||
}
|
||||
|
||||
if (bgr.channels() == 1) {
|
||||
cv::cvtColor(bgr, bgr, cv::COLOR_GRAY2BGR);
|
||||
}
|
||||
if (timestampMs) {
|
||||
*timestampMs = duration<double, std::milli>(steady_clock::now() - m_start).count();
|
||||
}
|
||||
return true;
|
||||
}
|
||||
@@ -0,0 +1,76 @@
|
||||
// SPDX-License-Identifier: GPL-3.0-or-later
|
||||
//
|
||||
// Where frames come from.
|
||||
//
|
||||
// Normally a V4L2 device. For testing without a camera (a virtual machine, a
|
||||
// build server) a video file or a folder of pictures stands in for one and is
|
||||
// played back at the pace a camera would deliver it, so that anything timed in
|
||||
// the liveness checks behaves the way it would with the real thing.
|
||||
|
||||
#pragma once
|
||||
|
||||
#include <QList>
|
||||
#include <QString>
|
||||
|
||||
#include <opencv2/core.hpp>
|
||||
#include <opencv2/videoio.hpp>
|
||||
|
||||
#include <chrono>
|
||||
#include <memory>
|
||||
|
||||
struct CameraInfo {
|
||||
QString path;
|
||||
QString name;
|
||||
// Infrared cameras (the kind Windows Hello uses) only offer grey formats.
|
||||
bool infrared = false;
|
||||
};
|
||||
|
||||
class Camera
|
||||
{
|
||||
public:
|
||||
Camera();
|
||||
~Camera();
|
||||
|
||||
// spec is a /dev/video path, "auto", "file:<video>" or "images:<dir>".
|
||||
bool open(const QString &spec, QString *error);
|
||||
void close();
|
||||
bool isOpen() const;
|
||||
|
||||
// Blocks until the next frame. The timestamp is in milliseconds on a
|
||||
// monotonic clock and only means something relative to other frames.
|
||||
bool read(cv::Mat &bgr, double *timestampMs);
|
||||
|
||||
bool isInfrared() const
|
||||
{
|
||||
return m_infrared;
|
||||
}
|
||||
QString description() const
|
||||
{
|
||||
return m_description;
|
||||
}
|
||||
|
||||
// Every V4L2 device that can capture video. Metadata nodes, which every
|
||||
// UVC camera also exposes, are left out.
|
||||
static QList<CameraInfo> list();
|
||||
// What "auto" means on this machine: the first colour camera, else the
|
||||
// first camera at all.
|
||||
static QString autoPath();
|
||||
|
||||
private:
|
||||
bool openImages(const QString &dir, QString *error);
|
||||
bool readImages(cv::Mat &bgr);
|
||||
void pace(double frameMs);
|
||||
|
||||
cv::VideoCapture m_capture;
|
||||
bool m_open = false;
|
||||
bool m_infrared = false;
|
||||
bool m_paced = false;
|
||||
QString m_description;
|
||||
|
||||
QList<cv::Mat> m_images;
|
||||
qsizetype m_imageIndex = 0;
|
||||
int m_imageRepeat = 0;
|
||||
|
||||
std::chrono::steady_clock::time_point m_start;
|
||||
std::chrono::steady_clock::time_point m_next;
|
||||
};
|
||||
@@ -0,0 +1,14 @@
|
||||
// SPDX-License-Identifier: GPL-3.0-or-later
|
||||
//
|
||||
// OpenCV 5 moved the contour and transform helpers (getPerspectiveTransform,
|
||||
// approxPolyDP and friends) out of imgproc into a module of their own. The
|
||||
// distributions this builds on ship 4.x and 5.x, so both are included here.
|
||||
|
||||
#pragma once
|
||||
|
||||
#include <opencv2/core/version.hpp>
|
||||
#include <opencv2/imgproc.hpp>
|
||||
|
||||
#if CV_VERSION_MAJOR >= 5
|
||||
#include <opencv2/geometry.hpp>
|
||||
#endif
|
||||
@@ -0,0 +1,82 @@
|
||||
// SPDX-License-Identifier: GPL-3.0-or-later
|
||||
|
||||
#include "keyvalue.h"
|
||||
|
||||
#include <QFile>
|
||||
|
||||
KeyValueFile KeyValueFile::load(const QString &path)
|
||||
{
|
||||
KeyValueFile kv;
|
||||
QFile file(path);
|
||||
if (!file.open(QIODevice::ReadOnly | QIODevice::Text)) {
|
||||
return kv;
|
||||
}
|
||||
|
||||
while (!file.atEnd()) {
|
||||
QString line = QString::fromUtf8(file.readLine()).trimmed();
|
||||
if (line.isEmpty() || line.startsWith(QLatin1Char('#'))) {
|
||||
continue;
|
||||
}
|
||||
const qsizetype eq = line.indexOf(QLatin1Char('='));
|
||||
if (eq <= 0) {
|
||||
continue;
|
||||
}
|
||||
const QString key = line.left(eq).trimmed();
|
||||
QString value = line.mid(eq + 1);
|
||||
|
||||
// A comment can follow a value, the same as in the shell reader.
|
||||
const qsizetype hash = value.indexOf(QLatin1Char('#'));
|
||||
if (hash >= 0) {
|
||||
value.truncate(hash);
|
||||
}
|
||||
value = value.trimmed();
|
||||
if (value.size() >= 2 && value.startsWith(QLatin1Char('"')) && value.endsWith(QLatin1Char('"'))) {
|
||||
value = value.mid(1, value.size() - 2);
|
||||
}
|
||||
|
||||
// The last occurrence wins, which is also what the shell reader does.
|
||||
kv.m_values.insert(key, value);
|
||||
}
|
||||
return kv;
|
||||
}
|
||||
|
||||
QString KeyValueFile::value(const QString &key, const QString &fallback) const
|
||||
{
|
||||
const auto it = m_values.constFind(key);
|
||||
if (it == m_values.cend() || it->isEmpty()) {
|
||||
return fallback;
|
||||
}
|
||||
return *it;
|
||||
}
|
||||
|
||||
bool KeyValueFile::contains(const QString &key) const
|
||||
{
|
||||
return m_values.contains(key);
|
||||
}
|
||||
|
||||
bool KeyValueFile::parseBool(const QString &value, bool fallback)
|
||||
{
|
||||
const QString v = value.trimmed().toLower();
|
||||
if (v == u"yes" || v == u"y" || v == u"true" || v == u"1" || v == u"on" || v == u"enabled") {
|
||||
return true;
|
||||
}
|
||||
if (v == u"no" || v == u"n" || v == u"false" || v == u"0" || v == u"off" || v == u"disabled") {
|
||||
return false;
|
||||
}
|
||||
return fallback;
|
||||
}
|
||||
|
||||
bool KeyValueFile::boolean(const QString &key, bool fallback) const
|
||||
{
|
||||
return parseBool(value(key), fallback);
|
||||
}
|
||||
|
||||
int KeyValueFile::integer(const QString &key, int fallback, int min, int max) const
|
||||
{
|
||||
bool ok = false;
|
||||
const int v = value(key).toInt(&ok);
|
||||
if (!ok) {
|
||||
return fallback;
|
||||
}
|
||||
return std::clamp(v, min, max);
|
||||
}
|
||||
@@ -0,0 +1,27 @@
|
||||
// SPDX-License-Identifier: GPL-3.0-or-later
|
||||
//
|
||||
// The settings files, read the same way the shell code reads them: one
|
||||
// Key=Value per line, '#' starts a comment, quotes around a value are dropped.
|
||||
// Both halves of the program write and read these files, so they have to
|
||||
// agree on every detail of the format.
|
||||
|
||||
#pragma once
|
||||
|
||||
#include <QHash>
|
||||
#include <QString>
|
||||
|
||||
class KeyValueFile
|
||||
{
|
||||
public:
|
||||
static KeyValueFile load(const QString &path);
|
||||
|
||||
QString value(const QString &key, const QString &fallback = {}) const;
|
||||
bool boolean(const QString &key, bool fallback) const;
|
||||
int integer(const QString &key, int fallback, int min, int max) const;
|
||||
bool contains(const QString &key) const;
|
||||
|
||||
static bool parseBool(const QString &value, bool fallback);
|
||||
|
||||
private:
|
||||
QHash<QString, QString> m_values;
|
||||
};
|
||||
@@ -0,0 +1,535 @@
|
||||
// SPDX-License-Identifier: GPL-3.0-or-later
|
||||
|
||||
#include "liveness.h"
|
||||
|
||||
#include "cvcompat.h"
|
||||
|
||||
#include <algorithm>
|
||||
#include <cmath>
|
||||
|
||||
namespace
|
||||
{
|
||||
// The eyes are levelled and scaled onto a fixed canvas before anything is
|
||||
// measured, so the numbers below are in canvas pixels and mean the same at
|
||||
// any distance and any head tilt. 64 pixels between the eyes.
|
||||
constexpr int Canvas = 128;
|
||||
constexpr float CanvasIod = 64.f;
|
||||
constexpr int EyeY = Canvas / 2;
|
||||
constexpr int RightEyeX = Canvas / 2 - int(CanvasIod / 2);
|
||||
constexpr int LeftEyeX = Canvas / 2 + int(CanvasIod / 2);
|
||||
|
||||
// A slice through the middle of the eye, narrow enough to stay on the iris
|
||||
// and tall enough to hold a fully open one (about 0.3 eye distances).
|
||||
constexpr int BandHalfWidth = 7;
|
||||
constexpr int BandHalfHeight = 10;
|
||||
// Skin under the eye, clear of lashes and of the shadow below the lid.
|
||||
constexpr int SkinTop = EyeY + 22;
|
||||
constexpr int SkinBottom = EyeY + 32;
|
||||
constexpr int SkinHalfWidth = 12;
|
||||
// A row of the slice counts as dark below this share of the skin's
|
||||
// brightness. Iris and pupil are well below it on every skin tone that
|
||||
// was checked; a closed lid is skin and sits well above it.
|
||||
constexpr float DarkShare = 0.7f;
|
||||
|
||||
// Glare: the Y floor and the chroma tolerance for "colourless and nearly
|
||||
// white", in YCrCb.
|
||||
constexpr int SpecularLuma = 235;
|
||||
constexpr int SpecularChroma = 10;
|
||||
constexpr int GlareGrid = 8;
|
||||
|
||||
cv::Mat levelledFace(const cv::Mat &bgr, const Face &face)
|
||||
{
|
||||
const cv::Point2f mid = face.eyeMid();
|
||||
const cv::Point2f d = face.points[LeftEye] - face.points[RightEye];
|
||||
const double roll = std::atan2(d.y, d.x) * 180.0 / M_PI;
|
||||
const double scale = CanvasIod / std::max(1.f, face.interocular());
|
||||
|
||||
cv::Mat m = cv::getRotationMatrix2D(mid, roll, scale);
|
||||
m.at<double>(0, 2) += Canvas / 2.0 - mid.x;
|
||||
m.at<double>(1, 2) += Canvas / 2.0 - mid.y;
|
||||
|
||||
cv::Mat grey, out;
|
||||
cv::cvtColor(bgr, grey, cv::COLOR_BGR2GRAY);
|
||||
cv::warpAffine(grey, out, m, cv::Size(Canvas, Canvas), cv::INTER_LINEAR, cv::BORDER_REPLICATE);
|
||||
return out;
|
||||
}
|
||||
|
||||
float eyeOpenness(const cv::Mat &canvas, int eyeX, float *skinOut)
|
||||
{
|
||||
const cv::Rect skinRect(eyeX - SkinHalfWidth, SkinTop, 2 * SkinHalfWidth, SkinBottom - SkinTop);
|
||||
const float skin = float(cv::mean(canvas(skinRect))[0]);
|
||||
*skinOut = skin;
|
||||
if (skin < 20) {
|
||||
return -1;
|
||||
}
|
||||
|
||||
const cv::Rect band(eyeX - BandHalfWidth, EyeY - BandHalfHeight, 2 * BandHalfWidth + 1, 2 * BandHalfHeight + 1);
|
||||
cv::Mat rows;
|
||||
cv::reduce(canvas(band), rows, 1, cv::REDUCE_AVG, CV_32F);
|
||||
|
||||
int dark = 0;
|
||||
for (int y = 0; y < rows.rows; ++y) {
|
||||
if (rows.at<float>(y, 0) < DarkShare * skin) {
|
||||
++dark;
|
||||
}
|
||||
}
|
||||
return float(dark) / float(rows.rows);
|
||||
}
|
||||
|
||||
bool insidePolygon(const std::vector<cv::Point> &poly, cv::Point2f p)
|
||||
{
|
||||
return cv::pointPolygonTest(poly, p, false) >= 0;
|
||||
}
|
||||
} // namespace
|
||||
|
||||
GlareSample measureGlare(const cv::Mat &bgr, const Face &face)
|
||||
{
|
||||
GlareSample g;
|
||||
const float iod = face.interocular();
|
||||
const cv::Point2f centre = (face.eyeMid() + face.mouthMid()) * 0.5f;
|
||||
cv::Rect roi(int(centre.x - 1.1f * iod), int(centre.y - 1.3f * iod), int(2.2f * iod), int(2.4f * iod));
|
||||
roi &= cv::Rect(0, 0, bgr.cols, bgr.rows);
|
||||
if (roi.width < 16 || roi.height < 16) {
|
||||
return g;
|
||||
}
|
||||
|
||||
cv::Mat ycc;
|
||||
cv::cvtColor(bgr(roi), ycc, cv::COLOR_BGR2YCrCb);
|
||||
|
||||
// Glasses throw the screen back from right in front of the eyes, and
|
||||
// that is a big flat highlight too. It is not what is being looked for,
|
||||
// so the eyes are left out.
|
||||
cv::Mat mask(roi.size(), CV_8U, cv::Scalar(255));
|
||||
for (int e : {RightEye, LeftEye}) {
|
||||
const cv::Point2f p = face.points[e] - cv::Point2f(float(roi.x), float(roi.y));
|
||||
cv::circle(mask, p, int(0.38f * iod), cv::Scalar(0), cv::FILLED);
|
||||
}
|
||||
|
||||
std::array<int, GlareGrid * GlareGrid> cells{};
|
||||
int total = 0;
|
||||
int considered = 0;
|
||||
for (int y = 0; y < ycc.rows; ++y) {
|
||||
const cv::Vec3b *row = ycc.ptr<cv::Vec3b>(y);
|
||||
const uchar *m = mask.ptr<uchar>(y);
|
||||
const int gy = std::min(GlareGrid - 1, y * GlareGrid / ycc.rows);
|
||||
for (int x = 0; x < ycc.cols; ++x) {
|
||||
if (!m[x]) {
|
||||
continue;
|
||||
}
|
||||
++considered;
|
||||
const cv::Vec3b &p = row[x];
|
||||
if (p[0] < SpecularLuma || std::abs(p[1] - 128) > SpecularChroma || std::abs(p[2] - 128) > SpecularChroma) {
|
||||
continue;
|
||||
}
|
||||
++total;
|
||||
const int gx = std::min(GlareGrid - 1, x * GlareGrid / ycc.cols);
|
||||
++cells[gy * GlareGrid + gx];
|
||||
}
|
||||
}
|
||||
|
||||
g.valid = considered > 0;
|
||||
g.fraction = considered ? float(total) / float(considered) : 0.f;
|
||||
// Too few pixels to say anything about their shape.
|
||||
g.cluster = total >= 24 ? float(*std::max_element(cells.begin(), cells.end())) / float(total) : 0.f;
|
||||
return g;
|
||||
}
|
||||
|
||||
EyeSample measureEyes(const cv::Mat &bgr, const Face &face)
|
||||
{
|
||||
EyeSample s;
|
||||
if (face.interocular() < 20.f) {
|
||||
return s;
|
||||
}
|
||||
const cv::Mat canvas = levelledFace(bgr, face);
|
||||
|
||||
// "Right" is the person's right eye, which is on the canvas' left.
|
||||
float skinR = 0, skinL = 0;
|
||||
s.right = eyeOpenness(canvas, RightEyeX, &skinR);
|
||||
s.left = eyeOpenness(canvas, LeftEyeX, &skinL);
|
||||
if (s.right < 0 || s.left < 0) {
|
||||
return s;
|
||||
}
|
||||
s.openness = (s.left + s.right) / 2;
|
||||
s.skin = (skinL + skinR) / 2;
|
||||
s.valid = true;
|
||||
return s;
|
||||
}
|
||||
|
||||
bool detectDevice(const cv::Mat &bgr, const Face &face)
|
||||
{
|
||||
// Edges are looked for at half size. The bezel of a phone held up to the
|
||||
// camera is big, and small detail would only add rectangles that are not
|
||||
// one.
|
||||
const double scale = 320.0 / std::max(1, bgr.cols);
|
||||
cv::Mat small, grey, edges;
|
||||
cv::resize(bgr, small, cv::Size(), scale, scale, cv::INTER_AREA);
|
||||
cv::cvtColor(small, grey, cv::COLOR_BGR2GRAY);
|
||||
cv::GaussianBlur(grey, grey, cv::Size(5, 5), 0);
|
||||
cv::Canny(grey, edges, 40, 120);
|
||||
cv::dilate(edges, edges, cv::Mat(), cv::Point(-1, -1), 1);
|
||||
|
||||
std::vector<std::vector<cv::Point>> contours;
|
||||
cv::findContours(edges, contours, cv::RETR_LIST, cv::CHAIN_APPROX_SIMPLE);
|
||||
|
||||
const double frameArea = double(small.cols) * small.rows;
|
||||
const double faceArea = double(face.box.area()) * scale * scale;
|
||||
std::array<cv::Point2f, 5> points;
|
||||
for (int i = 0; i < 5; ++i) {
|
||||
points[i] = face.points[i] * float(scale);
|
||||
}
|
||||
|
||||
for (const auto &c : contours) {
|
||||
const double area = std::abs(cv::contourArea(c));
|
||||
// A device held up to take somebody's place fills a good part of the
|
||||
// picture, and the face fills a good part of the device. A door or a
|
||||
// monitor far behind somebody's head does neither.
|
||||
if (area < 0.10 * frameArea || area > 0.95 * frameArea || area < 1.6 * faceArea || faceArea / area < 0.08) {
|
||||
continue;
|
||||
}
|
||||
std::vector<cv::Point> quad;
|
||||
cv::approxPolyDP(c, quad, 0.03 * cv::arcLength(c, true), true);
|
||||
if (quad.size() != 4 || !cv::isContourConvex(quad)) {
|
||||
continue;
|
||||
}
|
||||
const cv::RotatedRect r = cv::minAreaRect(quad);
|
||||
const float shortSide = std::min(r.size.width, r.size.height);
|
||||
const float longSide = std::max(r.size.width, r.size.height);
|
||||
if (longSide <= 0 || shortSide / longSide < 0.3f) {
|
||||
continue;
|
||||
}
|
||||
// It has to frame the face: every one of the five points inside.
|
||||
bool framed = true;
|
||||
for (const cv::Point2f &p : points) {
|
||||
framed = framed && insidePolygon(quad, p);
|
||||
}
|
||||
if (framed) {
|
||||
return true;
|
||||
}
|
||||
}
|
||||
return false;
|
||||
}
|
||||
|
||||
LivenessFrame measureFrame(const cv::Mat &bgr, const Face &face, double timestampMs)
|
||||
{
|
||||
LivenessFrame f;
|
||||
f.t = timestampMs;
|
||||
f.points = face.points;
|
||||
f.interocular = face.interocular();
|
||||
f.pose = estimatePose(face);
|
||||
f.glare = measureGlare(bgr, face);
|
||||
f.device = detectDevice(bgr, face);
|
||||
f.eyes = measureEyes(bgr, face);
|
||||
return f;
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// The analyzer
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
void LivenessAnalyzer::reset()
|
||||
{
|
||||
m_frames.clear();
|
||||
m_total = 0;
|
||||
m_glareHits = 0;
|
||||
m_deviceHits = 0;
|
||||
m_depthFired = false;
|
||||
m_blinkFired = false;
|
||||
m_depth.clear();
|
||||
m_depthNum = 0;
|
||||
m_depthDen = 0;
|
||||
m_depthPairs = 0;
|
||||
}
|
||||
|
||||
void LivenessAnalyzer::add(const LivenessFrame &frame)
|
||||
{
|
||||
m_frames.push_back(frame);
|
||||
while (!m_frames.empty() && frame.t - m_frames.front().t > WindowMs) {
|
||||
m_frames.pop_front();
|
||||
}
|
||||
|
||||
// Deny cues count over the whole scan, not just the window: a phone that
|
||||
// was seen once does not stop having been a phone.
|
||||
++m_total;
|
||||
if (frame.glare.valid && frame.glare.fraction >= GlareFraction && frame.glare.cluster >= GlareCluster) {
|
||||
++m_glareHits;
|
||||
}
|
||||
if (frame.device) {
|
||||
++m_deviceHits;
|
||||
}
|
||||
|
||||
addDepth(frame);
|
||||
|
||||
// Confirm cues latch once seen, for the rest of this scan.
|
||||
if (!m_depthFired) {
|
||||
m_depthFired = depthRange() >= DepthMinRange && m_depthPairs >= 6 && depthRatio() >= DepthMinRatio && depthConsistent();
|
||||
}
|
||||
if (!m_blinkFired) {
|
||||
float strength = 0;
|
||||
m_blinkFired = blinkSeen(&strength);
|
||||
}
|
||||
}
|
||||
|
||||
// How far the nose misses the spot a flat face would put it, measured against
|
||||
// how far the head turned.
|
||||
//
|
||||
// For a pair of frames far enough apart in yaw, the four points that lie close
|
||||
// to one plane (eyes, mouth corners) give the homography between the two
|
||||
// frames. A point on a flat picture lands exactly where it predicts. The real
|
||||
// nose tip is about a third of an eye distance in front of that plane, so it
|
||||
// lands off the prediction, sideways, by about as much as the yaw estimate
|
||||
// changed: both are the same parallax, seen two ways. The slope of miss
|
||||
// against yaw change is therefore near 1 for anything with a nose sticking
|
||||
// out of it, and near 0 for paper or a screen.
|
||||
//
|
||||
// Two details decide whether that holds up against a real camera.
|
||||
//
|
||||
// The yaw a frame is compared at comes from its neighbours, never from the
|
||||
// frame itself. Yaw and miss are both read off the same nose, so the nose's
|
||||
// own jitter would push both the same way in every frame, and the slope of
|
||||
// noise against itself is 1. That alone made a photo shaken at two pixels of
|
||||
// jitter pass as a head. The neighbours are 50 ms away, so they see the same
|
||||
// head position with jitter of their own.
|
||||
//
|
||||
// And the slope alone says nothing about depth, only that nose and plane
|
||||
// disagree consistently. A card curled around a vertical axis has a little
|
||||
// depth too, and its slope is near 1 as well. What it cannot do is move the
|
||||
// nose off the midline by much, so the cue also needs the (smoothed) yaw to
|
||||
// have covered DepthMinRange: about 12 degrees of a real head turning.
|
||||
//
|
||||
// A card that is curled hard and turned far enough still gets there. So the
|
||||
// last check asks whether the face turned as far as its nose says it did.
|
||||
// Turning narrows the eyes against the eye to mouth distance by 1 - cos(turn),
|
||||
// whatever the face is made of. A nose as flat as a real one can be (0.3 eye
|
||||
// distances) that moved DepthMinRange can only have turned so far, and a face
|
||||
// that narrowed a lot more than that was turned a lot more than that: it has
|
||||
// less nose than any head. See depthConsistent().
|
||||
void LivenessAnalyzer::addDepth(const LivenessFrame &frame)
|
||||
{
|
||||
constexpr size_t MaxFrames = 400;
|
||||
if (m_depth.size() >= MaxFrames) {
|
||||
return;
|
||||
}
|
||||
Face face;
|
||||
face.points = frame.points;
|
||||
const float emd = float(cv::norm(face.mouthMid() - face.eyeMid()));
|
||||
const float shape = emd > 1.f ? face.interocular() / emd : 0.f;
|
||||
m_depth.push_back({frame.points, frame.pose.yaw, 0.f, frame.pose.noseT, shape, 0.f});
|
||||
|
||||
// The frame two back now has two neighbours on each side.
|
||||
const size_t n = m_depth.size();
|
||||
if (n < 5) {
|
||||
return;
|
||||
}
|
||||
const size_t j = n - 3;
|
||||
DepthPoint &b = m_depth[j];
|
||||
b.smoothYaw = (m_depth[j - 2].yaw + m_depth[j - 1].yaw + m_depth[j + 1].yaw + m_depth[j + 2].yaw) / 4.f;
|
||||
b.smoothShape = (m_depth[j - 2].shape + m_depth[j - 1].shape + b.shape + m_depth[j + 1].shape + m_depth[j + 2].shape) / 5.f;
|
||||
|
||||
const cv::Point2f axis = b.points[LeftEye] - b.points[RightEye];
|
||||
const float axisLen = std::max(1e-3f, float(cv::norm(axis)));
|
||||
const cv::Point2f unit = axis * (1.f / axisLen);
|
||||
|
||||
for (size_t i = 2; i < j; ++i) {
|
||||
const DepthPoint &a = m_depth[i];
|
||||
const float dy = b.smoothYaw - a.smoothYaw;
|
||||
if (std::abs(dy) < DepthMinTurn) {
|
||||
continue;
|
||||
}
|
||||
const cv::Point2f src[4] = {a.points[RightEye], a.points[LeftEye], a.points[LeftMouth], a.points[RightMouth]};
|
||||
const cv::Point2f dst[4] = {b.points[RightEye], b.points[LeftEye], b.points[LeftMouth], b.points[RightMouth]};
|
||||
const cv::Mat h = cv::getPerspectiveTransform(src, dst);
|
||||
if (h.empty()) {
|
||||
continue;
|
||||
}
|
||||
std::vector<cv::Point2f> in{a.points[NoseTip]}, out;
|
||||
cv::perspectiveTransform(in, out, h);
|
||||
const cv::Point2f miss = b.points[NoseTip] - out[0];
|
||||
const float rx = (miss.x * unit.x + miss.y * unit.y) / axisLen;
|
||||
m_depthNum += double(rx) * dy;
|
||||
m_depthDen += double(dy) * dy;
|
||||
++m_depthPairs;
|
||||
}
|
||||
}
|
||||
|
||||
float LivenessAnalyzer::depthRatio() const
|
||||
{
|
||||
return m_depthDen > 0 ? float(m_depthNum / m_depthDen) : 0.f;
|
||||
}
|
||||
|
||||
// The spread of the smoothed yaw, from the 10th to the 90th percentile, so a
|
||||
// single bad frame cannot stretch it.
|
||||
float LivenessAnalyzer::depthRange() const
|
||||
{
|
||||
if (m_depth.size() < 5) {
|
||||
return 0;
|
||||
}
|
||||
std::vector<float> ys;
|
||||
for (size_t k = 2; k + 2 < m_depth.size(); ++k) {
|
||||
ys.push_back(m_depth[k].smoothYaw);
|
||||
}
|
||||
if (ys.size() < 3) {
|
||||
return 0;
|
||||
}
|
||||
std::sort(ys.begin(), ys.end());
|
||||
const auto at = [&](double q) {
|
||||
return ys[size_t(q * double(ys.size() - 1))];
|
||||
};
|
||||
return at(0.9) - at(0.1);
|
||||
}
|
||||
|
||||
bool LivenessAnalyzer::depthConsistent() const
|
||||
{
|
||||
// Nodding also changes the eye to mouth distance, so only frames at the
|
||||
// head's usual pitch are compared. A card has no pitch of its own to read
|
||||
// (its nose is printed on), so none of its frames are left out.
|
||||
std::vector<float> noseTs;
|
||||
for (size_t k = 2; k + 2 < m_depth.size(); ++k) {
|
||||
noseTs.push_back(m_depth[k].noseT);
|
||||
}
|
||||
if (noseTs.size() < 3) {
|
||||
return false;
|
||||
}
|
||||
std::vector<float> sortedT = noseTs;
|
||||
std::sort(sortedT.begin(), sortedT.end());
|
||||
const float medianT = sortedT[sortedT.size() / 2];
|
||||
|
||||
std::vector<float> shapes;
|
||||
for (size_t k = 2; k + 2 < m_depth.size(); ++k) {
|
||||
if (std::abs(m_depth[k].noseT - medianT) < 0.05f) {
|
||||
shapes.push_back(m_depth[k].smoothShape);
|
||||
}
|
||||
}
|
||||
if (shapes.size() < 3) {
|
||||
return false;
|
||||
}
|
||||
std::sort(shapes.begin(), shapes.end());
|
||||
const float widest = shapes[size_t(0.95 * double(shapes.size() - 1))];
|
||||
const float narrowest = shapes[size_t(0.10 * double(shapes.size() - 1))];
|
||||
if (widest <= 0) {
|
||||
return false;
|
||||
}
|
||||
const float narrowed = 1.f - narrowest / widest;
|
||||
|
||||
const float sinTurn = std::min(1.f, depthRange() / DepthFlattestNose);
|
||||
const float allowed = 1.f - std::sqrt(1.f - sinTurn * sinTurn);
|
||||
return narrowed <= allowed + DepthShapeSlack;
|
||||
}
|
||||
|
||||
bool LivenessAnalyzer::blinkSeen(float *strength) const
|
||||
{
|
||||
*strength = 0;
|
||||
std::vector<const LivenessFrame *> s;
|
||||
for (const LivenessFrame &f : m_frames) {
|
||||
if (f.eyes.valid) {
|
||||
s.push_back(&f);
|
||||
}
|
||||
}
|
||||
if (s.size() < 6) {
|
||||
return false;
|
||||
}
|
||||
|
||||
std::vector<float> values;
|
||||
for (const LivenessFrame *f : s) {
|
||||
values.push_back(f->eyes.openness);
|
||||
}
|
||||
std::vector<float> sorted = values;
|
||||
std::sort(sorted.begin(), sorted.end());
|
||||
const float baseline = sorted[size_t(0.7 * double(sorted.size() - 1))];
|
||||
// An eye this narrow is not one the measure works on (heavy lids, very
|
||||
// dark eyes on dark skin, a camera that is too soft). Better to abstain
|
||||
// than to guess.
|
||||
if (baseline < 0.15f) {
|
||||
return false;
|
||||
}
|
||||
*strength = std::clamp(1.f - sorted.front() / baseline, 0.f, 1.f) / (1.f - BlinkDip);
|
||||
|
||||
const size_t n = s.size();
|
||||
for (size_t i = 1; i + 1 < n; ++i) {
|
||||
if (values[i] >= BlinkDip * baseline) {
|
||||
continue;
|
||||
}
|
||||
// The run of closed frames around this one.
|
||||
size_t a = i, b = i;
|
||||
while (a > 0 && values[a - 1] < BlinkOpen * baseline) {
|
||||
--a;
|
||||
}
|
||||
while (b + 1 < n && values[b + 1] < BlinkOpen * baseline) {
|
||||
++b;
|
||||
}
|
||||
if (a == 0 || b + 1 >= n) {
|
||||
continue;
|
||||
}
|
||||
// Open right before and right after, within the time a blink takes
|
||||
// (a real one is 100 to 400 ms; slower is somebody closing their
|
||||
// eyes, or a picture being tilted away and back).
|
||||
const LivenessFrame *before = s[a - 1];
|
||||
const LivenessFrame *after = s[b + 1];
|
||||
if (after->t - before->t > 600) {
|
||||
continue;
|
||||
}
|
||||
// The rest of the face held still. A picture being moved blurs and
|
||||
// shifts everything at once; a blink only moves the lids.
|
||||
const cv::Point2f moved = (after->points[NoseTip] - before->points[NoseTip]);
|
||||
const float iod = std::max(1.f, before->interocular);
|
||||
if (float(cv::norm(moved)) > 0.15f * iod) {
|
||||
continue;
|
||||
}
|
||||
if (std::abs(after->interocular - before->interocular) > 0.08f * iod) {
|
||||
continue;
|
||||
}
|
||||
bool steadyLight = true;
|
||||
for (size_t k = a; k <= b; ++k) {
|
||||
const float ratio = s[k]->eyes.skin / std::max(1.f, before->eyes.skin);
|
||||
steadyLight = steadyLight && ratio > 0.9f && ratio < 1.1f;
|
||||
}
|
||||
if (!steadyLight) {
|
||||
continue;
|
||||
}
|
||||
// Both eyes together. One eye going dark on its own is a shadow or a
|
||||
// hand, not a blink.
|
||||
bool both = false;
|
||||
for (size_t k = a; k <= b && !both; ++k) {
|
||||
both = s[k]->eyes.left < 0.7f * baseline && s[k]->eyes.right < 0.7f * baseline;
|
||||
}
|
||||
if (both) {
|
||||
return true;
|
||||
}
|
||||
}
|
||||
return false;
|
||||
}
|
||||
|
||||
LivenessReading LivenessAnalyzer::reading() const
|
||||
{
|
||||
LivenessReading r;
|
||||
r.frames = int(m_frames.size());
|
||||
|
||||
const int seen = std::max(1, m_total);
|
||||
const float glareShare = float(m_glareHits) / float(seen);
|
||||
const float deviceShare = float(m_deviceHits) / float(seen);
|
||||
r.glare = std::min(glareShare / 0.3f, float(m_glareHits) / 3.f);
|
||||
r.device = std::min(deviceShare / 0.3f, float(m_deviceHits) / 3.f);
|
||||
|
||||
if (m_glareHits >= 3 && glareShare >= 0.3f) {
|
||||
r.denied = true;
|
||||
r.deniedBy = QStringLiteral("glare");
|
||||
} else if (m_deviceHits >= 3 && deviceShare >= 0.3f) {
|
||||
r.denied = true;
|
||||
r.deniedBy = QStringLiteral("device");
|
||||
}
|
||||
|
||||
r.depth = m_depthFired ? 1.f
|
||||
: std::clamp(depthRatio() / DepthMinRatio, 0.f, 1.f) * std::clamp(depthRange() / DepthMinRange, 0.f, 1.f);
|
||||
|
||||
float strength = 0;
|
||||
blinkSeen(&strength);
|
||||
r.blink = m_blinkFired ? 1.f : std::min(strength, 0.99f);
|
||||
|
||||
if (m_depthFired) {
|
||||
r.confirmed = true;
|
||||
r.confirmedBy = QStringLiteral("depth");
|
||||
} else if (m_blinkFired) {
|
||||
r.confirmed = true;
|
||||
r.confirmedBy = QStringLiteral("blink");
|
||||
}
|
||||
return r;
|
||||
}
|
||||
@@ -0,0 +1,154 @@
|
||||
// SPDX-License-Identifier: GPL-3.0-or-later
|
||||
//
|
||||
// Telling a face from a picture of one.
|
||||
//
|
||||
// A webcam sees a flat image either way, so there is no single test that
|
||||
// settles it. What is here follows the model Glance (the macOS face unlock)
|
||||
// arrived at after trying and dropping a dozen weaker signals: a few cues,
|
||||
// each one decisive on its own, split into two kinds.
|
||||
//
|
||||
// Deny cues are evidence of a fake. Either one fails the scan outright, and
|
||||
// a match that already happened does not outvote it.
|
||||
//
|
||||
// glare a phone screen or a glossy print throws back one large, flat,
|
||||
// colourless highlight. Skin shines in small scattered spots.
|
||||
// device the straight edges of a phone or a tablet around the face.
|
||||
//
|
||||
// Confirm cues are evidence of a real head. Any one of them is enough, and
|
||||
// their absence is never held against anybody on its own: a person can sit
|
||||
// still and not blink for a few seconds.
|
||||
//
|
||||
// depth when the head turns, the tip of the nose moves further than a
|
||||
// flat face would let it. The eyes and the corners of the mouth lie
|
||||
// close to one plane, so four of them predict exactly where the
|
||||
// nose of a flat picture has to go. A real nose, about a third of
|
||||
// an eye distance in front of that plane, misses the prediction by
|
||||
// about as much as the head turned.
|
||||
// blink the dark of the eye shrinks to a line and comes back, within
|
||||
// the fraction of a second a blink takes, while the rest of the face
|
||||
// holds still.
|
||||
//
|
||||
// "Light" uses the deny cues. "Heavy" also needs one confirm cue before it
|
||||
// lets anybody in.
|
||||
|
||||
#pragma once
|
||||
|
||||
#include "settings.h"
|
||||
#include "vision.h"
|
||||
|
||||
#include <QString>
|
||||
|
||||
#include <deque>
|
||||
|
||||
struct GlareSample {
|
||||
bool valid = false;
|
||||
// Share of the face that is a colourless highlight.
|
||||
float fraction = 0;
|
||||
// Share of all highlight pixels that fall in the fullest of 8x8 cells.
|
||||
// One slab of glass reflects into one place; skin sparkles everywhere.
|
||||
float cluster = 0;
|
||||
};
|
||||
|
||||
struct EyeSample {
|
||||
bool valid = false;
|
||||
// How much of the eye's height is dark (iris, pupil), per eye and on
|
||||
// average. Falls towards zero when the lid comes down.
|
||||
float left = 0;
|
||||
float right = 0;
|
||||
float openness = 0;
|
||||
// Brightness of the skin under the eyes, to tell a blink from a change
|
||||
// in the light.
|
||||
float skin = 0;
|
||||
};
|
||||
|
||||
struct LivenessFrame {
|
||||
double t = 0;
|
||||
std::array<cv::Point2f, 5> points{};
|
||||
float interocular = 0;
|
||||
HeadPose pose;
|
||||
GlareSample glare;
|
||||
bool device = false;
|
||||
EyeSample eyes;
|
||||
};
|
||||
|
||||
GlareSample measureGlare(const cv::Mat &bgr, const Face &face);
|
||||
EyeSample measureEyes(const cv::Mat &bgr, const Face &face);
|
||||
bool detectDevice(const cv::Mat &bgr, const Face &face);
|
||||
|
||||
LivenessFrame measureFrame(const cv::Mat &bgr, const Face &face, double timestampMs);
|
||||
|
||||
struct LivenessReading {
|
||||
int frames = 0;
|
||||
|
||||
bool denied = false;
|
||||
QString deniedBy;
|
||||
|
||||
bool confirmed = false;
|
||||
QString confirmedBy;
|
||||
|
||||
// For the test screen: how close each cue is to firing, 0 to 1 and more.
|
||||
float glare = 0;
|
||||
float device = 0;
|
||||
float depth = 0;
|
||||
float blink = 0;
|
||||
};
|
||||
|
||||
class LivenessAnalyzer
|
||||
{
|
||||
public:
|
||||
void reset();
|
||||
void add(const LivenessFrame &frame);
|
||||
LivenessReading reading() const;
|
||||
|
||||
int frameCount() const
|
||||
{
|
||||
return int(m_frames.size());
|
||||
}
|
||||
|
||||
// Tuning, in one place. See liveness.cpp for where each number comes
|
||||
// from.
|
||||
static constexpr double WindowMs = 4000;
|
||||
static constexpr float GlareFraction = 0.012f;
|
||||
static constexpr float GlareCluster = 0.55f;
|
||||
static constexpr float DepthMinTurn = 0.05f;
|
||||
static constexpr float DepthMinRange = 0.10f;
|
||||
static constexpr float DepthMinRatio = 0.65f;
|
||||
static constexpr float DepthFlattestNose = 0.30f;
|
||||
static constexpr float DepthShapeSlack = 0.02f;
|
||||
static constexpr float BlinkDip = 0.55f;
|
||||
static constexpr float BlinkOpen = 0.8f;
|
||||
|
||||
// The depth cue's workings, for the test screen and the tests.
|
||||
float depthRatio() const;
|
||||
float depthRange() const;
|
||||
bool depthConsistent() const;
|
||||
|
||||
private:
|
||||
void addDepth(const LivenessFrame &frame);
|
||||
bool blinkSeen(float *strength) const;
|
||||
|
||||
std::deque<LivenessFrame> m_frames;
|
||||
int m_total = 0;
|
||||
|
||||
// The depth cue looks at the whole scan rather than the window, and is
|
||||
// worked out a frame at a time so it stays cheap.
|
||||
struct DepthPoint {
|
||||
std::array<cv::Point2f, 5> points;
|
||||
float yaw;
|
||||
float smoothYaw;
|
||||
float noseT;
|
||||
// Eye distance over eye to mouth distance: shrinks as the face turns
|
||||
// away from the camera, whatever the face is made of.
|
||||
float shape;
|
||||
float smoothShape;
|
||||
};
|
||||
std::vector<DepthPoint> m_depth;
|
||||
double m_depthNum = 0;
|
||||
double m_depthDen = 0;
|
||||
int m_depthPairs = 0;
|
||||
|
||||
int m_glareHits = 0;
|
||||
int m_deviceHits = 0;
|
||||
bool m_depthFired = false;
|
||||
bool m_blinkFired = false;
|
||||
};
|
||||
@@ -0,0 +1,81 @@
|
||||
// SPDX-License-Identifier: GPL-3.0-or-later
|
||||
|
||||
#include "settings.h"
|
||||
#include "keyvalue.h"
|
||||
|
||||
double Settings::threshold() const
|
||||
{
|
||||
// Cosine similarity between two SFace embeddings. Photos of the same
|
||||
// person taken years apart land around 0.75, different people below 0.3.
|
||||
// OpenCV's own recommendation (0.363) is tuned for telling apart photos
|
||||
// in a data set; this guards a login, so the bar is higher.
|
||||
switch (strictness) {
|
||||
case Strictness::Relaxed:
|
||||
return 0.42;
|
||||
case Strictness::Normal:
|
||||
return 0.50;
|
||||
case Strictness::Strict:
|
||||
return 0.58;
|
||||
}
|
||||
return 0.50;
|
||||
}
|
||||
|
||||
Settings Settings::load(const QString &path)
|
||||
{
|
||||
Settings s;
|
||||
const KeyValueFile kv = KeyValueFile::load(path);
|
||||
|
||||
s.camera = kv.value(QStringLiteral("Camera"), s.camera);
|
||||
|
||||
const QString liveness = kv.value(QStringLiteral("Liveness")).toLower();
|
||||
if (liveness == u"off") {
|
||||
s.liveness = LivenessMode::Off;
|
||||
} else if (liveness == u"light") {
|
||||
s.liveness = LivenessMode::Light;
|
||||
} else if (liveness == u"heavy") {
|
||||
s.liveness = LivenessMode::Heavy;
|
||||
}
|
||||
|
||||
const QString strictness = kv.value(QStringLiteral("Strictness")).toLower();
|
||||
if (strictness == u"relaxed") {
|
||||
s.strictness = Strictness::Relaxed;
|
||||
} else if (strictness == u"normal") {
|
||||
s.strictness = Strictness::Normal;
|
||||
} else if (strictness == u"strict") {
|
||||
s.strictness = Strictness::Strict;
|
||||
}
|
||||
|
||||
s.attention = kv.boolean(QStringLiteral("Attention"), s.attention);
|
||||
s.scanSeconds = kv.integer(QStringLiteral("ScanSeconds"), s.scanSeconds, 2, 15);
|
||||
s.maxFailures = kv.integer(QStringLiteral("MaxFailures"), s.maxFailures, 1, 20);
|
||||
s.lockoutMinutes = kv.integer(QStringLiteral("LockoutMinutes"), s.lockoutMinutes, 1, 24 * 60);
|
||||
s.skipLidClosed = kv.boolean(QStringLiteral("SkipLidClosed"), s.skipLidClosed);
|
||||
s.adapt = kv.boolean(QStringLiteral("Adapt"), s.adapt);
|
||||
return s;
|
||||
}
|
||||
|
||||
QString Settings::livenessName(LivenessMode mode)
|
||||
{
|
||||
switch (mode) {
|
||||
case LivenessMode::Off:
|
||||
return QStringLiteral("off");
|
||||
case LivenessMode::Light:
|
||||
return QStringLiteral("light");
|
||||
case LivenessMode::Heavy:
|
||||
return QStringLiteral("heavy");
|
||||
}
|
||||
return {};
|
||||
}
|
||||
|
||||
QString Settings::strictnessName(Strictness strictness)
|
||||
{
|
||||
switch (strictness) {
|
||||
case Strictness::Relaxed:
|
||||
return QStringLiteral("relaxed");
|
||||
case Strictness::Normal:
|
||||
return QStringLiteral("normal");
|
||||
case Strictness::Strict:
|
||||
return QStringLiteral("strict");
|
||||
}
|
||||
return {};
|
||||
}
|
||||
@@ -0,0 +1,55 @@
|
||||
// SPDX-License-Identifier: GPL-3.0-or-later
|
||||
//
|
||||
// The system settings, /etc/plasma-face-unlock/config.
|
||||
//
|
||||
// These are the ones that decide how hard it is to get in: which camera, how
|
||||
// strict the match is, whether a photo is checked for. They belong to root
|
||||
// for that reason. A setting a user process could change would be a setting
|
||||
// any program running as that user could lower before asking sudo for help.
|
||||
|
||||
#pragma once
|
||||
|
||||
#include <QString>
|
||||
|
||||
enum class LivenessMode {
|
||||
Off,
|
||||
// Deny cues only: glare off a screen, the edge of a phone around the face.
|
||||
Light,
|
||||
// Deny cues, and a sign of life on top: a blink, or the nose moving like
|
||||
// a nose does when the head turns.
|
||||
Heavy,
|
||||
};
|
||||
|
||||
enum class Strictness {
|
||||
Relaxed,
|
||||
Normal,
|
||||
Strict,
|
||||
};
|
||||
|
||||
struct Settings {
|
||||
// A /dev/video path, "auto", or for testing "file:<video>" and
|
||||
// "images:<directory>".
|
||||
QString camera = QStringLiteral("auto");
|
||||
LivenessMode liveness = LivenessMode::Heavy;
|
||||
Strictness strictness = Strictness::Normal;
|
||||
// Only a face that looks at the screen with its eyes open counts.
|
||||
bool attention = true;
|
||||
// How long one scan looks before it gives up.
|
||||
int scanSeconds = 5;
|
||||
// Failed scans in a row with a face in view before face unlock stops
|
||||
// until the password has been used, and how long that lasts at most.
|
||||
int maxFailures = 5;
|
||||
int lockoutMinutes = 15;
|
||||
// A laptop with the lid shut has its camera looking at the keyboard.
|
||||
bool skipLidClosed = true;
|
||||
// Take in a little of each confident unlock, so a new haircut or a pair
|
||||
// of glasses does not need a new setup. Face ID does the same.
|
||||
bool adapt = true;
|
||||
|
||||
double threshold() const;
|
||||
|
||||
static Settings load(const QString &path);
|
||||
|
||||
static QString livenessName(LivenessMode mode);
|
||||
static QString strictnessName(Strictness strictness);
|
||||
};
|
||||
@@ -0,0 +1,206 @@
|
||||
// SPDX-License-Identifier: GPL-3.0-or-later
|
||||
|
||||
#include "store.h"
|
||||
|
||||
#include <QDir>
|
||||
#include <QFile>
|
||||
#include <QJsonArray>
|
||||
#include <QJsonDocument>
|
||||
#include <QJsonObject>
|
||||
#include <QRandomGenerator>
|
||||
#include <QSaveFile>
|
||||
|
||||
namespace
|
||||
{
|
||||
constexpr int FormatVersion = 1;
|
||||
|
||||
QByteArray packEmbedding(const Embedding &e)
|
||||
{
|
||||
QByteArray raw(reinterpret_cast<const char *>(e.data()), qsizetype(e.size() * sizeof(float)));
|
||||
return raw.toBase64();
|
||||
}
|
||||
|
||||
Embedding unpackEmbedding(const QString &b64)
|
||||
{
|
||||
const QByteArray raw = QByteArray::fromBase64(b64.toLatin1());
|
||||
Embedding e;
|
||||
if (raw.size() != qsizetype(Vision::EmbeddingSize * sizeof(float))) {
|
||||
return e;
|
||||
}
|
||||
e.resize(Vision::EmbeddingSize);
|
||||
memcpy(e.data(), raw.constData(), size_t(raw.size()));
|
||||
return e;
|
||||
}
|
||||
} // namespace
|
||||
|
||||
int Identity::adaptiveCount() const
|
||||
{
|
||||
int n = 0;
|
||||
for (const FaceSample &s : samples) {
|
||||
n += s.pose == u"adaptive";
|
||||
}
|
||||
return n;
|
||||
}
|
||||
|
||||
FaceStore::FaceStore(const QString &stateDir)
|
||||
: m_dir(QDir(stateDir).filePath(QStringLiteral("users")))
|
||||
{
|
||||
}
|
||||
|
||||
QString FaceStore::fileFor(uint uid) const
|
||||
{
|
||||
return QDir(m_dir).filePath(QStringLiteral("%1.json").arg(uid));
|
||||
}
|
||||
|
||||
QList<Identity> FaceStore::load(uint uid, QString *error) const
|
||||
{
|
||||
QList<Identity> faces;
|
||||
QFile file(fileFor(uid));
|
||||
if (!file.exists()) {
|
||||
return faces;
|
||||
}
|
||||
if (!file.open(QIODevice::ReadOnly)) {
|
||||
if (error) {
|
||||
*error = file.errorString();
|
||||
}
|
||||
return faces;
|
||||
}
|
||||
|
||||
QJsonParseError parseError;
|
||||
const QJsonDocument doc = QJsonDocument::fromJson(file.readAll(), &parseError);
|
||||
if (!doc.isObject()) {
|
||||
if (error) {
|
||||
*error = parseError.errorString();
|
||||
}
|
||||
return faces;
|
||||
}
|
||||
|
||||
const QJsonArray list = doc.object().value(u"faces").toArray();
|
||||
for (const QJsonValue &v : list) {
|
||||
const QJsonObject o = v.toObject();
|
||||
Identity id;
|
||||
id.id = o.value(u"id").toString();
|
||||
id.name = o.value(u"name").toString();
|
||||
id.created = qint64(o.value(u"created").toDouble());
|
||||
id.enabled = o.value(u"enabled").toBool(true);
|
||||
id.noseT = float(o.value(u"noseT").toDouble(0.55));
|
||||
id.eyes = float(o.value(u"eyes").toDouble(0));
|
||||
for (const QJsonValue &sv : o.value(u"samples").toArray()) {
|
||||
const QJsonObject so = sv.toObject();
|
||||
FaceSample s;
|
||||
s.embedding = unpackEmbedding(so.value(u"e").toString());
|
||||
s.pose = so.value(u"pose").toString();
|
||||
s.time = qint64(so.value(u"t").toDouble());
|
||||
if (!s.embedding.empty()) {
|
||||
id.samples.append(s);
|
||||
}
|
||||
}
|
||||
if (!id.id.isEmpty() && !id.samples.isEmpty()) {
|
||||
faces.append(id);
|
||||
}
|
||||
}
|
||||
return faces;
|
||||
}
|
||||
|
||||
bool FaceStore::save(uint uid, const QList<Identity> &faces, QString *error) const
|
||||
{
|
||||
if (!QDir().mkpath(m_dir)) {
|
||||
*error = QStringLiteral("cannot create %1").arg(m_dir);
|
||||
return false;
|
||||
}
|
||||
QFile::setPermissions(m_dir, QFileDevice::ReadOwner | QFileDevice::WriteOwner | QFileDevice::ExeOwner);
|
||||
|
||||
if (faces.isEmpty()) {
|
||||
remove(uid);
|
||||
return true;
|
||||
}
|
||||
|
||||
QJsonArray list;
|
||||
for (const Identity &id : faces) {
|
||||
QJsonArray samples;
|
||||
for (const FaceSample &s : id.samples) {
|
||||
samples.append(QJsonObject{
|
||||
{QStringLiteral("e"), QString::fromLatin1(packEmbedding(s.embedding))},
|
||||
{QStringLiteral("pose"), s.pose},
|
||||
{QStringLiteral("t"), double(s.time)},
|
||||
});
|
||||
}
|
||||
list.append(QJsonObject{
|
||||
{QStringLiteral("id"), id.id},
|
||||
{QStringLiteral("name"), id.name},
|
||||
{QStringLiteral("created"), double(id.created)},
|
||||
{QStringLiteral("enabled"), id.enabled},
|
||||
{QStringLiteral("noseT"), double(id.noseT)},
|
||||
{QStringLiteral("eyes"), double(id.eyes)},
|
||||
{QStringLiteral("samples"), samples},
|
||||
});
|
||||
}
|
||||
const QJsonObject root{
|
||||
{QStringLiteral("version"), FormatVersion},
|
||||
{QStringLiteral("faces"), list},
|
||||
};
|
||||
|
||||
QSaveFile file(fileFor(uid));
|
||||
if (!file.open(QIODevice::WriteOnly)) {
|
||||
*error = file.errorString();
|
||||
return false;
|
||||
}
|
||||
file.setPermissions(QFileDevice::ReadOwner | QFileDevice::WriteOwner);
|
||||
file.write(QJsonDocument(root).toJson(QJsonDocument::Compact));
|
||||
if (!file.commit()) {
|
||||
*error = file.errorString();
|
||||
return false;
|
||||
}
|
||||
return true;
|
||||
}
|
||||
|
||||
bool FaceStore::remove(uint uid) const
|
||||
{
|
||||
return QFile::remove(fileFor(uid));
|
||||
}
|
||||
|
||||
QString FaceStore::newId()
|
||||
{
|
||||
return QString::number(QRandomGenerator::system()->generate64() & 0xffffffffffffULL, 16);
|
||||
}
|
||||
|
||||
FaceMatch FaceStore::bestMatch(const QList<Identity> &faces, const Embedding &e)
|
||||
{
|
||||
FaceMatch best;
|
||||
for (int i = 0; i < faces.size(); ++i) {
|
||||
const Identity &id = faces.at(i);
|
||||
if (!id.enabled) {
|
||||
continue;
|
||||
}
|
||||
for (int k = 0; k < id.samples.size(); ++k) {
|
||||
const float s = Vision::similarity(id.samples.at(k).embedding, e);
|
||||
if (s > best.score) {
|
||||
best = {s, i, k};
|
||||
}
|
||||
}
|
||||
}
|
||||
return best;
|
||||
}
|
||||
|
||||
bool FaceStore::adapt(Identity &identity, const Embedding &e, qint64 now)
|
||||
{
|
||||
// Something the samples already cover adds nothing but weight.
|
||||
float closest = -1;
|
||||
for (const FaceSample &s : std::as_const(identity.samples)) {
|
||||
closest = std::max(closest, Vision::similarity(s.embedding, e));
|
||||
}
|
||||
if (closest >= 0.9f) {
|
||||
return false;
|
||||
}
|
||||
|
||||
if (identity.adaptiveCount() >= MaxAdaptive) {
|
||||
for (qsizetype i = 0; i < identity.samples.size(); ++i) {
|
||||
if (identity.samples.at(i).pose == u"adaptive") {
|
||||
identity.samples.removeAt(i);
|
||||
break;
|
||||
}
|
||||
}
|
||||
}
|
||||
identity.samples.append(FaceSample{e, QStringLiteral("adaptive"), now});
|
||||
return true;
|
||||
}
|
||||
@@ -0,0 +1,72 @@
|
||||
// SPDX-License-Identifier: GPL-3.0-or-later
|
||||
//
|
||||
// The face data.
|
||||
//
|
||||
// One file per user under /var/lib/plasma-face-unlock/users, named after the
|
||||
// numeric user id so a rename cannot hand one person's faces to another. Only
|
||||
// root can read or write the directory. There are no pictures in it: every
|
||||
// sample is the 128 numbers the recognizer made of one frame, and the frame
|
||||
// itself was never written anywhere.
|
||||
|
||||
#pragma once
|
||||
|
||||
#include "vision.h"
|
||||
|
||||
#include <QList>
|
||||
#include <QString>
|
||||
|
||||
struct FaceSample {
|
||||
Embedding embedding;
|
||||
// Which way the head pointed when it was taken: "center", one of the
|
||||
// eight directions ("up", "up-right", ...), or "adaptive" for one taken
|
||||
// in from a successful unlock.
|
||||
QString pose;
|
||||
qint64 time = 0;
|
||||
};
|
||||
|
||||
struct Identity {
|
||||
QString id;
|
||||
QString name;
|
||||
qint64 created = 0;
|
||||
bool enabled = true;
|
||||
// Where this person's nose sits for a level head (see HeadPose), and how
|
||||
// open their eyes measure when they look at the camera. Both are what
|
||||
// the attention check compares against.
|
||||
float noseT = 0.55f;
|
||||
float eyes = 0;
|
||||
QList<FaceSample> samples;
|
||||
|
||||
int adaptiveCount() const;
|
||||
};
|
||||
|
||||
struct FaceMatch {
|
||||
float score = -1;
|
||||
int identity = -1;
|
||||
int sample = -1;
|
||||
};
|
||||
|
||||
class FaceStore
|
||||
{
|
||||
public:
|
||||
explicit FaceStore(const QString &stateDir);
|
||||
|
||||
QList<Identity> load(uint uid, QString *error = nullptr) const;
|
||||
bool save(uint uid, const QList<Identity> &faces, QString *error) const;
|
||||
bool remove(uint uid) const;
|
||||
|
||||
static QString newId();
|
||||
|
||||
// The best sample over every enabled identity.
|
||||
static FaceMatch bestMatch(const QList<Identity> &faces, const Embedding &e);
|
||||
|
||||
// Adds what a confident unlock saw, if it adds anything, and keeps at most
|
||||
// MaxAdaptive of those per identity (the oldest go first). The samples
|
||||
// from the setup are never touched.
|
||||
static bool adapt(Identity &identity, const Embedding &e, qint64 now);
|
||||
|
||||
static constexpr int MaxAdaptive = 12;
|
||||
|
||||
private:
|
||||
QString fileFor(uint uid) const;
|
||||
QString m_dir;
|
||||
};
|
||||
@@ -0,0 +1,219 @@
|
||||
// SPDX-License-Identifier: GPL-3.0-or-later
|
||||
|
||||
#include "vision.h"
|
||||
|
||||
#include <QDir>
|
||||
#include <QFile>
|
||||
#include <QFileInfo>
|
||||
|
||||
#include <opencv2/core/utils/logger.hpp>
|
||||
#include <opencv2/imgproc.hpp>
|
||||
|
||||
#include <algorithm>
|
||||
#include <cmath>
|
||||
#include <numeric>
|
||||
|
||||
namespace
|
||||
{
|
||||
const char DetectorFile[] = "face_detection_yunet_2023mar.onnx";
|
||||
const char RecognizerFile[] = "face_recognition_sface_2021dec.onnx";
|
||||
|
||||
// A face whose eyes are closer together than this is too far away for the
|
||||
// recognizer to see detail in, and the liveness checks drown in noise.
|
||||
constexpr float MinInterocular = 28.f;
|
||||
|
||||
cv::Point2f rotateAround(cv::Point2f p, cv::Point2f centre, float angle)
|
||||
{
|
||||
const float c = std::cos(angle);
|
||||
const float s = std::sin(angle);
|
||||
const cv::Point2f d = p - centre;
|
||||
return {centre.x + d.x * c - d.y * s, centre.y + d.x * s + d.y * c};
|
||||
}
|
||||
} // namespace
|
||||
|
||||
float Face::interocular() const
|
||||
{
|
||||
return float(cv::norm(points[LeftEye] - points[RightEye]));
|
||||
}
|
||||
|
||||
cv::Point2f Face::eyeMid() const
|
||||
{
|
||||
return (points[RightEye] + points[LeftEye]) * 0.5f;
|
||||
}
|
||||
|
||||
cv::Point2f Face::mouthMid() const
|
||||
{
|
||||
return (points[RightMouth] + points[LeftMouth]) * 0.5f;
|
||||
}
|
||||
|
||||
HeadPose estimatePose(const Face &face)
|
||||
{
|
||||
HeadPose pose;
|
||||
const cv::Point2f eyes = face.points[LeftEye] - face.points[RightEye];
|
||||
pose.roll = std::atan2(eyes.y, eyes.x);
|
||||
|
||||
// Level the face first, so a tilted head does not read as a turned one.
|
||||
const cv::Point2f centre = face.eyeMid();
|
||||
std::array<cv::Point2f, 5> p;
|
||||
for (int i = 0; i < 5; ++i) {
|
||||
p[i] = rotateAround(face.points[i], centre, -pose.roll);
|
||||
}
|
||||
|
||||
const cv::Point2f eyeMid = (p[RightEye] + p[LeftEye]) * 0.5f;
|
||||
const cv::Point2f mouthMid = (p[RightMouth] + p[LeftMouth]) * 0.5f;
|
||||
const float iod = std::max(1.f, float(cv::norm(p[LeftEye] - p[RightEye])));
|
||||
const float span = mouthMid.y - eyeMid.y;
|
||||
if (span < 1.f) {
|
||||
return pose;
|
||||
}
|
||||
|
||||
const cv::Point2f nose = p[NoseTip];
|
||||
const float t = (nose.y - eyeMid.y) / span;
|
||||
const float midlineX = eyeMid.x + (mouthMid.x - eyeMid.x) * t;
|
||||
|
||||
// The eyes are reported person-right first, which is image-left on an
|
||||
// unmirrored frame. A nose moving image-right is a head turning to the
|
||||
// person's left.
|
||||
pose.yaw = (nose.x - midlineX) / iod;
|
||||
pose.noseT = t;
|
||||
return pose;
|
||||
}
|
||||
|
||||
float yawDegrees(float yaw)
|
||||
{
|
||||
return float(std::asin(std::clamp(yaw / 0.5f, -1.f, 1.f)) * 180.0 / M_PI);
|
||||
}
|
||||
|
||||
FaceQuality assessQuality(const cv::Mat &bgr, const Face &face)
|
||||
{
|
||||
FaceQuality q;
|
||||
q.tooSmall = face.interocular() < MinInterocular;
|
||||
|
||||
// The middle of the face, without hair and background at the edges.
|
||||
const float iod = face.interocular();
|
||||
const cv::Point2f c = (face.eyeMid() + face.mouthMid()) * 0.5f;
|
||||
cv::Rect roi(cv::Point(int(c.x - iod), int(c.y - iod)), cv::Size(int(2 * iod), int(2 * iod)));
|
||||
roi &= cv::Rect(0, 0, bgr.cols, bgr.rows);
|
||||
if (roi.width < 8 || roi.height < 8) {
|
||||
q.tooSmall = true;
|
||||
return q;
|
||||
}
|
||||
|
||||
cv::Mat grey;
|
||||
cv::cvtColor(bgr(roi), grey, cv::COLOR_BGR2GRAY);
|
||||
q.brightness = float(cv::mean(grey)[0]);
|
||||
|
||||
// Sharpness at a fixed scale, so a face far away and one up close are
|
||||
// judged the same.
|
||||
cv::Mat small, lap;
|
||||
cv::resize(grey, small, cv::Size(96, 96), 0, 0, cv::INTER_AREA);
|
||||
cv::Laplacian(small, lap, CV_32F);
|
||||
cv::Scalar mean, stddev;
|
||||
cv::meanStdDev(lap, mean, stddev);
|
||||
q.sharpness = float(stddev[0] * stddev[0]);
|
||||
|
||||
q.tooDark = q.brightness < 40;
|
||||
q.tooBright = q.brightness > 230;
|
||||
q.blurry = q.sharpness < 12;
|
||||
return q;
|
||||
}
|
||||
|
||||
bool Vision::load(const QString &modelDir, QString *error)
|
||||
{
|
||||
// The new DNN engine in OpenCV 5 prints a warning for every network it
|
||||
// loads about targets it does not support yet. Nothing here asks for one.
|
||||
cv::utils::logging::setLogLevel(cv::utils::logging::LOG_LEVEL_ERROR);
|
||||
|
||||
const QDir dir(modelDir);
|
||||
const QString detector = dir.filePath(QLatin1String(DetectorFile));
|
||||
const QString recognizer = dir.filePath(QLatin1String(RecognizerFile));
|
||||
for (const QString &f : {detector, recognizer}) {
|
||||
if (!QFileInfo::exists(f)) {
|
||||
*error = QStringLiteral("model missing: %1").arg(f);
|
||||
return false;
|
||||
}
|
||||
}
|
||||
|
||||
try {
|
||||
m_inputSize = cv::Size(640, 480);
|
||||
m_detector = cv::FaceDetectorYN::create(QFile::encodeName(detector).toStdString(), "", m_inputSize, 0.75f, 0.3f, 20);
|
||||
m_recognizer = cv::FaceRecognizerSF::create(QFile::encodeName(recognizer).toStdString(), "");
|
||||
} catch (const cv::Exception &e) {
|
||||
*error = QString::fromStdString(e.what());
|
||||
m_detector.reset();
|
||||
m_recognizer.reset();
|
||||
return false;
|
||||
}
|
||||
return true;
|
||||
}
|
||||
|
||||
std::vector<Face> Vision::detect(const cv::Mat &bgr)
|
||||
{
|
||||
std::vector<Face> result;
|
||||
if (!m_detector || bgr.empty()) {
|
||||
return result;
|
||||
}
|
||||
if (bgr.size() != m_inputSize) {
|
||||
m_inputSize = bgr.size();
|
||||
m_detector->setInputSize(m_inputSize);
|
||||
}
|
||||
|
||||
cv::Mat rows;
|
||||
try {
|
||||
m_detector->detect(bgr, rows);
|
||||
} catch (const cv::Exception &) {
|
||||
return result;
|
||||
}
|
||||
|
||||
for (int i = 0; i < rows.rows; ++i) {
|
||||
const float *r = rows.ptr<float>(i);
|
||||
Face f;
|
||||
f.box = cv::Rect2f(r[0], r[1], r[2], r[3]);
|
||||
for (int k = 0; k < 5; ++k) {
|
||||
f.points[k] = cv::Point2f(r[4 + 2 * k], r[5 + 2 * k]);
|
||||
}
|
||||
f.score = r[14];
|
||||
f.row = rows.row(i).clone();
|
||||
result.push_back(std::move(f));
|
||||
}
|
||||
std::sort(result.begin(), result.end(), [](const Face &a, const Face &b) {
|
||||
return a.box.area() > b.box.area();
|
||||
});
|
||||
return result;
|
||||
}
|
||||
|
||||
Embedding Vision::embed(const cv::Mat &bgr, const Face &face)
|
||||
{
|
||||
Embedding out;
|
||||
if (!m_recognizer) {
|
||||
return out;
|
||||
}
|
||||
try {
|
||||
cv::Mat aligned, feature;
|
||||
m_recognizer->alignCrop(bgr, face.row, aligned);
|
||||
m_recognizer->feature(aligned, feature);
|
||||
feature = feature.reshape(1, 1);
|
||||
if (feature.cols != EmbeddingSize) {
|
||||
return out;
|
||||
}
|
||||
const double norm = cv::norm(feature);
|
||||
if (norm <= 0) {
|
||||
return out;
|
||||
}
|
||||
out.resize(EmbeddingSize);
|
||||
for (int i = 0; i < EmbeddingSize; ++i) {
|
||||
out[i] = float(feature.at<float>(0, i) / norm);
|
||||
}
|
||||
} catch (const cv::Exception &) {
|
||||
out.clear();
|
||||
}
|
||||
return out;
|
||||
}
|
||||
|
||||
float Vision::similarity(const Embedding &a, const Embedding &b)
|
||||
{
|
||||
if (a.size() != b.size() || a.empty()) {
|
||||
return -1.f;
|
||||
}
|
||||
return std::inner_product(a.begin(), a.end(), b.begin(), 0.f);
|
||||
}
|
||||
@@ -0,0 +1,108 @@
|
||||
// SPDX-License-Identifier: GPL-3.0-or-later
|
||||
//
|
||||
// Finding a face and turning it into numbers.
|
||||
//
|
||||
// Two small networks from the OpenCV model zoo do the work. YuNet finds faces
|
||||
// and five points on each (the eyes, the tip of the nose, the corners of the
|
||||
// mouth). SFace turns an aligned crop of one face into 128 numbers, and two
|
||||
// crops of the same person give numbers that point the same way. Both run on
|
||||
// the CPU through OpenCV's own DNN module, in a few milliseconds each.
|
||||
//
|
||||
// Everything else here is plain geometry on those five points.
|
||||
|
||||
#pragma once
|
||||
|
||||
#include <QString>
|
||||
|
||||
#include <opencv2/core.hpp>
|
||||
#include <opencv2/objdetect/face.hpp>
|
||||
|
||||
#include <array>
|
||||
#include <vector>
|
||||
|
||||
// The five points, in the order YuNet reports them. "Right" and "left" are
|
||||
// the person's own, so on an unmirrored picture the right eye is on the left.
|
||||
enum Landmark {
|
||||
RightEye = 0,
|
||||
LeftEye = 1,
|
||||
NoseTip = 2,
|
||||
RightMouth = 3,
|
||||
LeftMouth = 4,
|
||||
};
|
||||
|
||||
struct Face {
|
||||
cv::Rect2f box;
|
||||
std::array<cv::Point2f, 5> points;
|
||||
float score = 0;
|
||||
// YuNet's own row, which the recognizer wants back for its alignment.
|
||||
cv::Mat row;
|
||||
|
||||
float interocular() const;
|
||||
cv::Point2f eyeMid() const;
|
||||
cv::Point2f mouthMid() const;
|
||||
};
|
||||
|
||||
// Where the head points, read off the five points alone.
|
||||
//
|
||||
// yaw is the nose's sideways offset from the line through the middle of the
|
||||
// eyes and the middle of the mouth, in interocular distances. A nose sits
|
||||
// about half an interocular distance in front of the face, so this is close to
|
||||
// 0.5 * sin(head yaw). Positive when the head turns to the person's left.
|
||||
//
|
||||
// noseT is how far down the nose tip sits between the eye line (0) and the
|
||||
// mouth line (1). It changes with pitch, but where it sits for a level head is
|
||||
// different for every face, so it only means something next to the same
|
||||
// person's own resting value.
|
||||
struct HeadPose {
|
||||
float roll = 0;
|
||||
float yaw = 0;
|
||||
float noseT = 0;
|
||||
};
|
||||
|
||||
HeadPose estimatePose(const Face &face);
|
||||
|
||||
// Degrees, for people. The estimate is rough by nature.
|
||||
float yawDegrees(float yaw);
|
||||
|
||||
struct FaceQuality {
|
||||
float brightness = 0;
|
||||
float sharpness = 0;
|
||||
bool tooSmall = false;
|
||||
bool tooDark = false;
|
||||
bool tooBright = false;
|
||||
bool blurry = false;
|
||||
|
||||
bool ok() const
|
||||
{
|
||||
return !tooSmall && !tooDark && !tooBright && !blurry;
|
||||
}
|
||||
};
|
||||
|
||||
FaceQuality assessQuality(const cv::Mat &bgr, const Face &face);
|
||||
|
||||
using Embedding = std::vector<float>;
|
||||
|
||||
class Vision
|
||||
{
|
||||
public:
|
||||
bool load(const QString &modelDir, QString *error);
|
||||
bool isLoaded() const
|
||||
{
|
||||
return m_detector && m_recognizer;
|
||||
}
|
||||
|
||||
// Faces sorted by size, largest first.
|
||||
std::vector<Face> detect(const cv::Mat &bgr);
|
||||
|
||||
// Unit length, so the similarity of two is their dot product.
|
||||
Embedding embed(const cv::Mat &bgr, const Face &face);
|
||||
|
||||
static float similarity(const Embedding &a, const Embedding &b);
|
||||
|
||||
static constexpr int EmbeddingSize = 128;
|
||||
|
||||
private:
|
||||
cv::Ptr<cv::FaceDetectorYN> m_detector;
|
||||
cv::Ptr<cv::FaceRecognizerSF> m_recognizer;
|
||||
cv::Size m_inputSize;
|
||||
};
|
||||
Reference in new issue
Block a user