Compare commits
124
Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
b62713bb8d | ||
|
|
ef56eb33ed | ||
|
|
b64f6690f6 | ||
|
|
bf7c84bc15 | ||
|
|
b07e69b5ed | ||
|
|
ce8284df6c | ||
|
|
966241070e | ||
|
|
b25258330c | ||
|
|
db4135e481 | ||
|
|
72f8ddd5e7 | ||
|
|
88e756bc3a | ||
|
|
f77d171063 | ||
|
|
79db6c537b | ||
|
|
1a23243cc1 | ||
|
|
ca4bf75c1b | ||
|
|
8a596c44ac | ||
|
|
c50d5eb109 | ||
|
|
098abb102a | ||
|
|
c1702a00bb | ||
|
|
a24ed3b60a | ||
|
|
f791c78d17 | ||
|
|
461f5fab3c | ||
|
|
4f25927d18 | ||
|
|
fb785b8635 | ||
|
|
50a03df30b | ||
|
|
9976d52e29 | ||
|
|
58d2804278 | ||
|
|
5f6548bfdd | ||
|
|
daa484efbc | ||
|
|
1aedc77791 | ||
|
|
a19d9eec89 | ||
|
|
5c1c4c6248 | ||
|
|
2d8728793c | ||
|
|
9429f1de00 | ||
|
|
76c170db9d | ||
|
|
d7bebd454d | ||
|
|
282ad5fc13 | ||
|
|
133d41df36 | ||
|
|
c4a6e4d117 | ||
|
|
f59a9de4dc | ||
|
|
083f22d6fb | ||
|
|
c43abef8ab | ||
|
|
c9f8028502 | ||
|
|
cd7a0f4163 | ||
|
|
30d4cef0e7 | ||
|
|
6c1eeea038 | ||
|
|
ea9a6f0c58 | ||
|
|
1c1838af05 | ||
|
|
78c9490b1e | ||
|
|
ce2742fc80 | ||
|
|
81deaa69c4 | ||
|
|
d9754c46a6 | ||
|
|
4481279f1c | ||
|
|
20abf69d31 | ||
|
|
69ef48239a | ||
|
|
00f00d6d75 | ||
|
|
68d3ad795e | ||
|
|
d486747c11 | ||
|
|
e69df1ae8d | ||
|
|
031d028ba4 | ||
|
|
6d7afc3c06 | ||
|
|
3d5a1692ab | ||
|
|
1de77316f0 | ||
|
|
26c7c6a897 | ||
|
|
4ba1896eb1 | ||
|
|
b0e53ef966 | ||
|
|
dd57709522 | ||
|
|
3450c31345 | ||
|
|
29d3a5f779 | ||
|
|
f1f112fc38 | ||
|
|
96be849976 | ||
|
|
a5c8927dbc | ||
|
|
ad09eeeefb | ||
|
|
7e06a3b1f6 | ||
|
|
e147206e82 | ||
|
|
ca6484c356 | ||
|
|
faf3d1e056 | ||
|
|
ffcfde0b5a | ||
|
|
f1f05db790 | ||
|
|
e1076a04b2 | ||
|
|
8fc8d94bbc | ||
|
|
0c600a63fa | ||
|
|
f78925b316 | ||
|
|
6ee7ba1b7e | ||
|
|
a992caf810 | ||
|
|
daa486f7e7 | ||
|
|
9c29fb2bea | ||
|
|
abd5811420 | ||
|
|
80051539d5 | ||
|
|
64550ebbd0 | ||
|
|
eea96e8674 | ||
|
|
4c5583e725 | ||
|
|
441ad0b18e | ||
|
|
792ff9d1ee | ||
|
|
af49e94d97 | ||
|
|
37c00b609c | ||
|
|
94a3a762b0 | ||
|
|
9a7d678532 | ||
|
|
823d42d528 | ||
|
|
de514115dd | ||
|
|
0f8816f659 | ||
|
|
b59eebf1e7 | ||
|
|
f4061f542c | ||
|
|
e99f84bd01 | ||
|
|
9653219c53 | ||
|
|
f44382fb09 | ||
|
|
dd73e0ad74 | ||
|
|
7f045c626a | ||
|
|
e99d26de89 | ||
|
|
7f22006e97 | ||
|
|
e35fc3e6d6 | ||
|
|
5f52dad5f1 | ||
|
|
15064d6fd5 | ||
|
|
e0060c9e6e | ||
|
|
a3a36cd5d7 | ||
|
|
8afaee7d21 | ||
|
|
c8280de9c3 | ||
|
|
f8b9df6438 | ||
|
|
92d14fbd60 | ||
|
|
01f6b99631 | ||
|
|
8d5e4ee052 | ||
|
|
9e0aab6b6a | ||
|
|
3eb5a454fd | ||
|
|
dc49bf4d14 |
No files matched your search
+44
-1
@@ -7,7 +7,12 @@
|
||||
# instance resolves short `uses:` against itself, never GitHub, so nothing
|
||||
# unreviewed can be pulled in.
|
||||
#
|
||||
# Not ported, as on GitLab: publish.yml and release.yml still need doing.
|
||||
# BUILD_ON: when the Actions variable BUILD_ON is 'github' (org or repo),
|
||||
# fork-checks and build skip here and the `github` job below waits for the
|
||||
# same work done by .github/workflows/ci.yml on the GitHub mirror, passing or
|
||||
# failing with it -- so this run still carries the answer pull requests and
|
||||
# merges look at. Unset, everything builds here as before. If GitHub is
|
||||
# unavailable, unset BUILD_ON and nothing else has to change.
|
||||
name: ci
|
||||
|
||||
on:
|
||||
@@ -25,6 +30,7 @@ jobs:
|
||||
# without the AGPL 5(a) notice. Seconds, and needs no toolchain. The notice
|
||||
# check diffs against the upstream snapshot branch, hence the full fetch.
|
||||
fork-checks:
|
||||
if: ${{ vars.BUILD_ON != 'github' }}
|
||||
runs-on: light
|
||||
container:
|
||||
image: python:3.13-slim@sha256:8d9d0b8bcf6506481eae4907c18f5e3e7902e629f5f6d684f9e7c32e85e3ddf0 # 3.13-slim
|
||||
@@ -52,6 +58,7 @@ jobs:
|
||||
run: python3 -m unittest discover -s tools/fork/tests
|
||||
|
||||
build:
|
||||
if: ${{ vars.BUILD_ON != 'github' }}
|
||||
# Either runner (host1 or host2): the build needs no docker socket.
|
||||
runs-on: light
|
||||
container:
|
||||
@@ -101,3 +108,39 @@ jobs:
|
||||
used=$(du -s --block-size=1G /cache/target 2>/dev/null | cut -f1)
|
||||
echo "target dir: ${used:-0} GB"
|
||||
if [ "${used:-0}" -gt 60 ]; then rm -rf /cache/target && echo "over 60 GB: target dir cleared"; fi
|
||||
|
||||
# BUILD_ON=github: the GitHub mirror builds this commit and posts the result
|
||||
# back as the commit status "github/ci (branch)". This waits for that status
|
||||
# and takes its answer. The mirror pushes on every commit, so a missing
|
||||
# status means GitHub has not got the push or is not running: after the
|
||||
# timeout this fails, which is the cue to unset BUILD_ON.
|
||||
github:
|
||||
if: ${{ vars.BUILD_ON == 'github' }}
|
||||
# Its own runner label with plenty of slots: this job only polls, but holds a slot
|
||||
# for as long as the GitHub build takes, and must not starve the build runners.
|
||||
runs-on: wait
|
||||
timeout-minutes: 150
|
||||
container:
|
||||
image: python:3.13-slim@sha256:8d9d0b8bcf6506481eae4907c18f5e3e7902e629f5f6d684f9e7c32e85e3ddf0 # 3.13-slim
|
||||
steps:
|
||||
- env:
|
||||
TOKEN: ${{ secrets.GITHUB_TOKEN }}
|
||||
SHA: ${{ github.event.pull_request.head.sha || github.sha }}
|
||||
CONTEXT: github/ci (branch)
|
||||
run: |
|
||||
python3 - <<'EOF'
|
||||
import json, os, time, urllib.request
|
||||
url = (f"{os.environ['CI_SERVER_INTERNAL']}/api/v1/repos/{os.environ['GITHUB_REPOSITORY']}"
|
||||
f"/commits/{os.environ['SHA']}/statuses?limit=50")
|
||||
req = urllib.request.Request(url, headers={"Authorization": f"token {os.environ['TOKEN']}"})
|
||||
ctx, last = os.environ["CONTEXT"], None
|
||||
print(f"waiting for '{ctx}' on {os.environ['SHA']}", flush=True)
|
||||
while True:
|
||||
mine = [s for s in json.load(urllib.request.urlopen(req)) if s["context"] == ctx]
|
||||
state = max(mine, key=lambda s: s["id"]) if mine else None
|
||||
if state and state["status"] != last:
|
||||
last = state["status"]; print(f"{ctx}: {last} {state.get('target_url', '')}", flush=True)
|
||||
if last == "success": raise SystemExit(0)
|
||||
if last in ("failure", "error"): raise SystemExit(1)
|
||||
time.sleep(20)
|
||||
EOF
|
||||
@@ -42,6 +42,14 @@
|
||||
#
|
||||
# The push logs in with PACKAGE_TOKEN (jcoffey-dev, write:package): the job's
|
||||
# own token is refused by the container registry.
|
||||
#
|
||||
# BUILD_ON: when the Actions variable BUILD_ON is 'github' (org or repo), every
|
||||
# job here but the announcement skips, and the tag is published by
|
||||
# .github/workflows/ci.yml on the GitHub mirror instead -- same guards, same
|
||||
# tags, the same Release and binaries, created here through the API. The
|
||||
# `github` job waits for that run's commit status, "github/ci (tag)", and the
|
||||
# announcement follows it as it follows the binaries here. Unset, everything
|
||||
# runs here as before.
|
||||
name: publish
|
||||
|
||||
on:
|
||||
@@ -50,6 +58,7 @@ on:
|
||||
|
||||
jobs:
|
||||
version:
|
||||
if: ${{ vars.BUILD_ON != 'github' }}
|
||||
runs-on: light
|
||||
container:
|
||||
image: python:3.13-slim@sha256:8d9d0b8bcf6506481eae4907c18f5e3e7902e629f5f6d684f9e7c32e85e3ddf0 # 3.13-slim
|
||||
@@ -88,6 +97,7 @@ jobs:
|
||||
echo "version $V"
|
||||
|
||||
publish-amd64:
|
||||
if: ${{ vars.BUILD_ON != 'github' }}
|
||||
needs: [version]
|
||||
runs-on: docker
|
||||
container:
|
||||
@@ -128,6 +138,7 @@ jobs:
|
||||
run: docker logout "$REGISTRY" || true
|
||||
|
||||
publish-arm64:
|
||||
if: ${{ vars.BUILD_ON != 'github' }}
|
||||
needs: [version, publish-amd64]
|
||||
runs-on: docker
|
||||
container:
|
||||
@@ -162,11 +173,46 @@ jobs:
|
||||
- if: always()
|
||||
run: docker logout "$REGISTRY" || true
|
||||
|
||||
# BUILD_ON=github: waits for the GitHub mirror's run for this tag, which
|
||||
# posts its result back as the commit status "github/ci (tag)", and takes
|
||||
# its answer. Fails after the timeout if no answer comes.
|
||||
github:
|
||||
if: ${{ vars.BUILD_ON == 'github' }}
|
||||
# Its own runner label with plenty of slots: this job only polls, but holds a slot
|
||||
# for as long as the GitHub build takes, and must not starve the build runners.
|
||||
runs-on: wait
|
||||
timeout-minutes: 240
|
||||
container:
|
||||
image: python:3.13-slim@sha256:8d9d0b8bcf6506481eae4907c18f5e3e7902e629f5f6d684f9e7c32e85e3ddf0 # 3.13-slim
|
||||
steps:
|
||||
- env:
|
||||
TOKEN: ${{ secrets.GITHUB_TOKEN }}
|
||||
SHA: ${{ github.sha }}
|
||||
CONTEXT: github/ci (tag)
|
||||
run: |
|
||||
python3 - <<'EOF'
|
||||
import json, os, time, urllib.request
|
||||
url = (f"{os.environ['CI_SERVER_INTERNAL']}/api/v1/repos/{os.environ['GITHUB_REPOSITORY']}"
|
||||
f"/commits/{os.environ['SHA']}/statuses?limit=50")
|
||||
req = urllib.request.Request(url, headers={"Authorization": f"token {os.environ['TOKEN']}"})
|
||||
ctx, last = os.environ["CONTEXT"], None
|
||||
print(f"waiting for '{ctx}' on {os.environ['SHA']}", flush=True)
|
||||
while True:
|
||||
mine = [s for s in json.load(urllib.request.urlopen(req)) if s["context"] == ctx]
|
||||
state = max(mine, key=lambda s: s["id"]) if mine else None
|
||||
if state and state["status"] != last:
|
||||
last = state["status"]; print(f"{ctx}: {last} {state.get('target_url', '')}", flush=True)
|
||||
if last == "success": raise SystemExit(0)
|
||||
if last in ("failure", "error"): raise SystemExit(1)
|
||||
time.sleep(20)
|
||||
EOF
|
||||
|
||||
# The weekly release creates its Release (and so the tag) first; a tag
|
||||
# pushed by hand has none. Either way the tag ends up with exactly one
|
||||
# Release, created once the amd64 image exists so its pull instructions
|
||||
# work; arm64 and the binaries follow.
|
||||
release:
|
||||
if: ${{ vars.BUILD_ON != 'github' }}
|
||||
needs: [version, publish-amd64]
|
||||
runs-on: light
|
||||
container:
|
||||
@@ -216,6 +262,7 @@ jobs:
|
||||
# `docker create` does not start anything, so pulling an arm64 image on an
|
||||
# amd64 runner and copying a file out of it needs no emulation.
|
||||
binaries:
|
||||
if: ${{ vars.BUILD_ON != 'github' }}
|
||||
needs: [version, publish-arm64, release]
|
||||
runs-on: docker
|
||||
container:
|
||||
@@ -289,8 +336,14 @@ jobs:
|
||||
# The release above is made with the job's own token, and Gitea starts no
|
||||
# workflow for events the Actions bot causes -- announce.yml's
|
||||
# 'on: release' never fires for it -- so announce it from here.
|
||||
#
|
||||
# With BUILD_ON=github the release and binaries come from the GitHub run,
|
||||
# so the announcement waits for the `github` job instead. The Release that
|
||||
# run creates for a hand-pushed tag is made with a user token, so
|
||||
# announce.yml fires for it too; discourse-release keeps one topic per tag.
|
||||
announce:
|
||||
needs: [release, binaries]
|
||||
needs: [release, binaries, github]
|
||||
if: ${{ always() && ((needs.release.result == 'success' && needs.binaries.result == 'success') || needs.github.result == 'success') }}
|
||||
runs-on: light
|
||||
steps:
|
||||
- uses: coffey-labs/actions/discourse-release@e9293996e2efa770839121fa8f8da93083f216be
|
||||
|
||||
@@ -5,11 +5,6 @@
|
||||
|
||||
version: 2
|
||||
updates:
|
||||
- package-ecosystem: "cargo" # See documentation for possible values
|
||||
directory: "/" # Location of package manifests
|
||||
schedule:
|
||||
interval: "weekly"
|
||||
|
||||
# Enable version updates for GitHub Actions
|
||||
- package-ecosystem: "github-actions"
|
||||
# Workflow files stored in the default location of `.github/workflows`
|
||||
|
||||
@@ -12,7 +12,7 @@ jobs:
|
||||
runs-on: ubuntu-latest
|
||||
steps:
|
||||
- name: Close issues from non-allowed authors
|
||||
uses: actions/github-script@v7
|
||||
uses: actions/github-script@v9
|
||||
with:
|
||||
script: |
|
||||
// Users allowed to open issues directly. All other authors will have
|
||||
|
||||
@@ -18,7 +18,7 @@ jobs:
|
||||
sparse-checkout-cone-mode: false
|
||||
|
||||
- name: Close PRs from non-allowed authors
|
||||
uses: actions/github-script@v7
|
||||
uses: actions/github-script@v9
|
||||
with:
|
||||
script: |
|
||||
const fs = require('fs');
|
||||
|
||||
@@ -12,7 +12,7 @@ jobs:
|
||||
runs-on: ubuntu-latest
|
||||
steps:
|
||||
- name: Post support portal redirect
|
||||
uses: actions/github-script@v7
|
||||
uses: actions/github-script@v9
|
||||
with:
|
||||
script: |
|
||||
const discussion = context.payload.discussion;
|
||||
|
||||
@@ -73,6 +73,6 @@ jobs:
|
||||
# Upload the results to GitHub's code scanning dashboard (optional).
|
||||
# Commenting out will disable upload of results to your repo's Code Scanning dashboard
|
||||
- name: "Upload to code-scanning"
|
||||
uses: github/codeql-action/[email protected]7.4
|
||||
uses: github/codeql-action/[email protected]8.2
|
||||
with:
|
||||
sarif_file: results.sarif
|
||||
@@ -36,6 +36,6 @@ jobs:
|
||||
severity: 'CRITICAL,HIGH'
|
||||
|
||||
- name: Upload Trivy scan results to GitHub Security tab
|
||||
uses: github/codeql-action/[email protected]7.4
|
||||
uses: github/codeql-action/[email protected]8.2
|
||||
with:
|
||||
sarif_file: 'trivy-results.sarif'
|
||||
@@ -1,42 +0,0 @@
|
||||
version: 2
|
||||
updates:
|
||||
# Cargo. One entry: the workspace has a single lockfile at the root, and
|
||||
# ~30 manifests that upstream bumps on every release -- pointing entries at
|
||||
# individual crates would find manifests with no lockfile beside them.
|
||||
#
|
||||
# Minor and patch arrive as one pull request a week. Majors are left out of
|
||||
# the group on purpose: they are migrations rather than bumps, and each one
|
||||
# deserves its own pull request and its own CI run.
|
||||
- package-ecosystem: cargo
|
||||
directory: "/"
|
||||
schedule:
|
||||
interval: weekly
|
||||
day: tuesday
|
||||
time: "09:00"
|
||||
timezone: Etc/UTC
|
||||
open-pull-requests-limit: 5
|
||||
groups:
|
||||
minor-and-patch:
|
||||
update-types:
|
||||
- minor
|
||||
- patch
|
||||
- package-ecosystem: github-actions
|
||||
directory: "/"
|
||||
schedule:
|
||||
interval: weekly
|
||||
day: tuesday
|
||||
time: "09:00"
|
||||
timezone: Etc/UTC
|
||||
groups:
|
||||
actions:
|
||||
patterns:
|
||||
- "*"
|
||||
# The Dockerfiles pin their base images, so this is what keeps a published
|
||||
# image off a stale base between releases.
|
||||
- package-ecosystem: docker
|
||||
directory: "/"
|
||||
schedule:
|
||||
interval: weekly
|
||||
day: tuesday
|
||||
time: "09:00"
|
||||
timezone: Etc/UTC
|
||||
+469
-38
@@ -1,51 +1,482 @@
|
||||
# What CI can check without a mail server's worth of infrastructure.
|
||||
# CI and publishing on GitHub, for the repository Gitea mirrors here.
|
||||
#
|
||||
# The build, and that every test target compiles. It deliberately does not
|
||||
# *run* the test suites: the unit tests only build with the integration crate
|
||||
# in the graph, because that is what switches on the `test_mode` features they
|
||||
# rely on (docs/spec/SPEC.md 2.2b), and the integration suites need a `STORE`,
|
||||
# fixed ports, and in most cases a container apiece (docs/spec/
|
||||
# container-tests.md). Running them here would mean either a green tick that
|
||||
# skipped everything, or a red one that means "the runner has no Redis".
|
||||
# Gitea (git.coffeylabs.org) is where this project lives: pull requests,
|
||||
# issues, releases and the container registry are all there, and it pushes
|
||||
# every branch and tag to this GitHub copy as it changes. GitHub's hosted
|
||||
# runners are faster than the self-hosted ones -- and have native arm64 -- so
|
||||
# the building happens here, and the answer goes back to Gitea as a commit
|
||||
# status that Gitea's own ci.yml / publish.yml wait on.
|
||||
#
|
||||
# So this catches what it can honestly catch -- code that does not compile,
|
||||
# including test code -- and the suites are run by hand, one at a time, as
|
||||
# that page describes. If that changes, it changes because someone made the
|
||||
# suites runnable unattended, not because CI started ignoring failures.
|
||||
name: CI
|
||||
# One switch decides which side builds: the Actions variable BUILD_ON, set on
|
||||
# both forges. BUILD_ON=github runs every job below and turns Gitea's heavy
|
||||
# jobs into a wait for this one; anything else leaves Gitea building exactly
|
||||
# as before and every job here skips. If GitHub is ever unavailable, unset it
|
||||
# on Gitea and nothing else has to change.
|
||||
#
|
||||
# Needs, as organization settings rather than anything in this file:
|
||||
# variables BUILD_ON=github, REGISTRY (the Gitea container registry),
|
||||
# GITEA_URL (the Gitea base URL)
|
||||
# secret GITEA_TOKEN -- jcoffey-dev, write:repository + write:package:
|
||||
# commit statuses, the release and its assets, the registry push
|
||||
#
|
||||
# There is no pull_request trigger: pull requests happen on Gitea, and their
|
||||
# branch arrives here as an ordinary push. Branch pushes get what Gitea's
|
||||
# ci.yml checks; v* tags get what its publish.yml does. Schedules (the weekly
|
||||
# release, the upstream watch) and the release announcement stay on Gitea.
|
||||
#
|
||||
# Every `uses:` is pinned to a full commit SHA with the release in the
|
||||
# trailing comment. A tag is a mutable pointer; do not "simplify" a pin back
|
||||
# to one. Only GitHub's own actions and the three docker/* ones are used.
|
||||
name: ci
|
||||
|
||||
on:
|
||||
push:
|
||||
branches: [main]
|
||||
pull_request:
|
||||
# Lets CI be run by hand against any ref, including one that predates a CI
|
||||
# change, without pushing an empty commit to move it.
|
||||
branches: ['**']
|
||||
tags: ['**']
|
||||
workflow_dispatch:
|
||||
|
||||
# A second push to a branch cancels the run still going for the first: the
|
||||
# older run's answer is about code nobody is looking at any more.
|
||||
# A newer push to a branch cancels the run for the older one, whose answer is
|
||||
# about code nobody is looking at any more. A tag run is never cancelled: it
|
||||
# publishes.
|
||||
concurrency:
|
||||
group: ci-${{ github.ref }}
|
||||
cancel-in-progress: true
|
||||
cancel-in-progress: ${{ github.ref_type == 'branch' }}
|
||||
|
||||
permissions:
|
||||
contents: read
|
||||
|
||||
env:
|
||||
GITEA_URL: ${{ vars.GITEA_URL }}
|
||||
# The Gitea status this run answers for. Gitea waits on the one matching
|
||||
# its own event: "(branch)" from ci.yml, "(tag)" from publish.yml.
|
||||
STATUS_CONTEXT: github/ci (${{ github.ref_type }})
|
||||
|
||||
jobs:
|
||||
build:
|
||||
# Tells Gitea a run has started, so a pull request shows it as pending
|
||||
# rather than missing while the build is still going.
|
||||
start:
|
||||
if: ${{ vars.BUILD_ON == 'github' }}
|
||||
runs-on: ubuntu-latest
|
||||
steps:
|
||||
- env:
|
||||
GITEA_TOKEN: ${{ secrets.GITEA_TOKEN }}
|
||||
run: |
|
||||
jq -n --arg c "$STATUS_CONTEXT" \
|
||||
--arg u "$GITHUB_SERVER_URL/$GITHUB_REPOSITORY/actions/runs/$GITHUB_RUN_ID" \
|
||||
'{state:"pending", context:$c, target_url:$u, description:"GitHub Actions"}' |
|
||||
curl -fsS -o /dev/null -X POST -H "Authorization: token $GITEA_TOKEN" \
|
||||
-H 'Content-Type: application/json' --data @- \
|
||||
"$GITEA_URL/api/v1/repos/$GITHUB_REPOSITORY/statuses/$GITHUB_SHA"
|
||||
|
||||
# ----------------------------------------------------------- branches ------
|
||||
# What an upstream merge can bring in or leave behind without a conflict:
|
||||
# the upstream name in a new string literal, and a changed upstream file
|
||||
# without the AGPL 5(a) notice. Seconds, and needs no toolchain. The notice
|
||||
# check diffs against the upstream snapshot in the history, hence the full
|
||||
# fetch.
|
||||
fork-checks:
|
||||
if: ${{ vars.BUILD_ON == 'github' && github.ref_type == 'branch' }}
|
||||
runs-on: ubuntu-latest
|
||||
steps:
|
||||
# Every `uses:` here is pinned to a full commit SHA, with the release it
|
||||
# belongs to in the trailing comment. A tag is a mutable pointer, so
|
||||
# trusting `@v7` is trusting every future version of that action,
|
||||
# including one pushed by whoever compromises the account. Dependabot
|
||||
# updates both halves together -- do not "simplify" a pin back to a tag.
|
||||
- uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1
|
||||
- uses: Swatinem/rust-cache@6323deb102c322ba6fcbdcafc7e3dddab59af2b6 # v2.9.2
|
||||
- name: System dependencies
|
||||
# foundationdb and the search backends are off by default, but the
|
||||
# default feature set still links against the system's C libraries.
|
||||
run: sudo apt-get update && sudo apt-get install -y --no-install-recommends clang
|
||||
- name: Build the server
|
||||
run: cargo build -p inbuxa --locked
|
||||
- name: Compile every test target
|
||||
# `--no-run` is the point: it builds the unit tests and the integration
|
||||
# crate together, which is the combination that resolves the test
|
||||
# features, and stops short of running anything that wants a store.
|
||||
run: cargo test --workspace --locked --no-run
|
||||
with:
|
||||
fetch-depth: 0
|
||||
- run: python3 tools/fork/name-check.py
|
||||
- if: always()
|
||||
run: python3 tools/fork/notice-check.py
|
||||
# Cargo can patch a dependency to a directory in this repository, and
|
||||
# the image builds from a context .dockerignore prunes to almost
|
||||
# nothing. CI never sees the difference; a release does.
|
||||
- if: always()
|
||||
run: python3 tools/fork/context-check.py
|
||||
# The personal-data catalog must classify every object and field the
|
||||
# schema has, and name nothing that is gone.
|
||||
- if: always()
|
||||
run: python3 tools/fork/privacy-check.py
|
||||
# The admin reads each expression field's allowed values and variables
|
||||
# from the schema; they're generated from the registry and must match it.
|
||||
- if: always()
|
||||
run: python3 tools/fork/expr-schema.py --check
|
||||
- if: always()
|
||||
run: python3 -m unittest discover -s tools/fork/tests
|
||||
|
||||
# The build, and that every test target compiles. The suites are not run:
|
||||
# they need a store, fixed ports and containers (docs/spec/
|
||||
# container-tests.md), and are run by hand.
|
||||
build:
|
||||
if: ${{ vars.BUILD_ON == 'github' && github.ref_type == 'branch' }}
|
||||
runs-on: ubuntu-latest
|
||||
env:
|
||||
CARGO_INCREMENTAL: "0"
|
||||
# Debug info is most of a dev target dir, and nothing here runs a
|
||||
# debugger. Without it the dev and test builds fit the runner's disk and
|
||||
# the cache below stays small enough to be worth restoring.
|
||||
CARGO_PROFILE_DEV_DEBUG: "0"
|
||||
CARGO_PROFILE_TEST_DEBUG: "0"
|
||||
steps:
|
||||
# The hosted image carries toolchains this build never touches; a dev,
|
||||
# test and release build of RocksDB and the workspace needs the room.
|
||||
- run: |
|
||||
sudo rm -rf /usr/share/dotnet /usr/local/lib/android /opt/ghc /opt/hostedtoolcache/CodeQL
|
||||
df -h /
|
||||
- uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1
|
||||
# Current stable, as Gitea's rust:1 image is.
|
||||
- id: rust
|
||||
run: |
|
||||
rustup toolchain install stable --profile minimal
|
||||
rustup default stable
|
||||
echo "version=$(rustc -V | cut -d' ' -f2)" >> "$GITHUB_OUTPUT"
|
||||
- run: sudo apt-get update -qq && sudo apt-get install -y -qq --no-install-recommends clang >/dev/null
|
||||
# Cargo's download cache and the dev/test target dir, keyed on the
|
||||
# lockfile and the compiler. Saved from main only, so the one cache
|
||||
# every branch restores is main's, and branches cannot evict it.
|
||||
- uses: actions/cache/restore@55cc8345863c7cc4c66a329aec7e433d2d1c52a9 # v6.1.0
|
||||
with:
|
||||
path: |
|
||||
~/.cargo/registry/index
|
||||
~/.cargo/registry/cache
|
||||
~/.cargo/git/db
|
||||
target/debug
|
||||
key: cargo-${{ steps.rust.outputs.version }}-${{ hashFiles('Cargo.lock') }}
|
||||
restore-keys: cargo-${{ steps.rust.outputs.version }}-
|
||||
- run: cargo build -p inbuxa --locked
|
||||
# --no-run: compiles every test target without running them, which
|
||||
# catches a test that no longer builds without needing a store.
|
||||
- run: cargo test --workspace --locked --no-run
|
||||
- if: github.ref == 'refs/heads/main'
|
||||
uses: actions/cache/save@55cc8345863c7cc4c66a329aec7e433d2d1c52a9 # v6.1.0
|
||||
with:
|
||||
path: |
|
||||
~/.cargo/registry/index
|
||||
~/.cargo/registry/cache
|
||||
~/.cargo/git/db
|
||||
target/debug
|
||||
key: cargo-${{ steps.rust.outputs.version }}-${{ hashFiles('Cargo.lock') }}
|
||||
# The release profile, on main only. It is the profile the image is
|
||||
# built with, and it fails in ways the dev profile does not: v2026.9.24
|
||||
# was tagged on a commit whose CI was green and whose release build
|
||||
# could not compile the scim crate at all.
|
||||
- if: github.ref == 'refs/heads/main'
|
||||
run: cargo build -p inbuxa --locked --release
|
||||
|
||||
# --------------------------------------------------------------- tags ------
|
||||
# Two guards before anything is pushed, the same as Gitea's publish.yml:
|
||||
# * the tag must be v<brand_version!>. The version is a string in
|
||||
# crates/types/src/branding.rs, not Cargo.toml, and the image is tagged
|
||||
# with it, so a tag beside an unbumped macro would publish an image that
|
||||
# reports a different version from its tag.
|
||||
# * the tag must be on main or on a release/* branch, so an image never
|
||||
# describes code that was never reviewed onto one of them. A release/*
|
||||
# branch carries a hotfix cut from an earlier release tag.
|
||||
version:
|
||||
if: ${{ vars.BUILD_ON == 'github' && github.ref_type == 'tag' && startsWith(github.ref_name, 'v') }}
|
||||
runs-on: ubuntu-latest
|
||||
outputs:
|
||||
version: ${{ steps.v.outputs.version }}
|
||||
steps:
|
||||
# Full history, and every branch as origin/*: the ancestry check cannot
|
||||
# be answered from a shallow clone.
|
||||
- uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1
|
||||
with:
|
||||
fetch-depth: 0
|
||||
- id: v
|
||||
env:
|
||||
TAG: ${{ github.ref_name }}
|
||||
run: |
|
||||
set -euo pipefail
|
||||
# Scoped to the macro body: branding.rs holds other string literals,
|
||||
# and tagging an image from one of those would be worse than failing.
|
||||
V="$(awk '/macro_rules! brand_version /,/^}/' crates/types/src/branding.rs \
|
||||
| grep -om1 '"[0-9][^"]*"' | tr -d '"')"
|
||||
[ -n "$V" ] || { echo "could not read brand_version! from branding.rs" >&2; exit 1; }
|
||||
if [ "$TAG" != "v$V" ]; then
|
||||
echo "Tag $TAG names a commit whose brand_version! says $V." >&2
|
||||
echo "Refusing to publish an image that would report the wrong version." >&2
|
||||
exit 1
|
||||
fi
|
||||
commit="$(git rev-parse "${TAG}^{commit}")"
|
||||
on=""
|
||||
for ref in origin/main $(git for-each-ref --format='%(refname:short)' 'refs/remotes/origin/release/*'); do
|
||||
if git merge-base --is-ancestor "$commit" "$ref"; then on="$ref"; break; fi
|
||||
done
|
||||
[ -n "$on" ] || { echo "$TAG is not on main or a release/* branch" >&2; exit 1; }
|
||||
echo "$TAG is on $on"
|
||||
echo "version=$V" >> "$GITHUB_OUTPUT"
|
||||
|
||||
# Each architecture on its own native runner, side by side. The Dockerfile
|
||||
# cross-compiles from the build platform, and on the self-hosted runners one
|
||||
# machine built both one after the other; here two machines build at once,
|
||||
# each natively (the builder stage picks the matching target, and the
|
||||
# aarch64 toolchain it installs exists on arm64 too), and the small final
|
||||
# stage needs no QEMU. amd64 also moves :<version> as soon as it is done, so
|
||||
# a production deploy can start from it; :latest waits for the index below,
|
||||
# so it never names an image without arm64.
|
||||
publish:
|
||||
needs: [version]
|
||||
runs-on: ${{ matrix.runner }}
|
||||
strategy:
|
||||
fail-fast: false
|
||||
matrix:
|
||||
include:
|
||||
- arch: amd64
|
||||
runner: ubuntu-latest
|
||||
- arch: arm64
|
||||
runner: ubuntu-24.04-arm
|
||||
env:
|
||||
VERSION: ${{ needs.version.outputs.version }}
|
||||
steps:
|
||||
- run: |
|
||||
sudo rm -rf /usr/share/dotnet /usr/local/lib/android /opt/ghc /opt/hostedtoolcache/CodeQL
|
||||
echo "IMAGE=${{ vars.REGISTRY }}/${GITHUB_REPOSITORY,,}" >> "$GITHUB_ENV"
|
||||
# The release link (fat LTO, one codegen unit) outgrows the runner's
|
||||
# 16 GB: v2026.9.30's arm64 link was killed for memory. Swap gives it
|
||||
# room; buildx's container has no memory limit of its own, so it
|
||||
# reaches the host's swap.
|
||||
- run: |
|
||||
sudo fallocate -l 16G /swap.release
|
||||
sudo chmod 600 /swap.release
|
||||
sudo mkswap /swap.release >/dev/null
|
||||
sudo swapon /swap.release
|
||||
free -g
|
||||
df -h /
|
||||
- uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1
|
||||
- uses: docker/setup-buildx-action@594f3bf4285d9ea8dc53c9a0c9c4092420091003 # v4.4.0
|
||||
- uses: docker/login-action@dbcb813823bdd20940b903addbd779551569679f # v4.6.0
|
||||
with:
|
||||
registry: ${{ vars.REGISTRY }}
|
||||
username: jcoffey-dev
|
||||
password: ${{ secrets.GITEA_TOKEN }}
|
||||
# Attestations off: they add manifests of their own, and the index
|
||||
# should hold the two images and nothing else. No build cache: GitHub
|
||||
# scopes a tag run's cache to that tag, so the next release could never
|
||||
# read it, and each one would park several GB in the repository's 10 GB
|
||||
# cache and evict main's cargo cache.
|
||||
- uses: docker/build-push-action@c3c9e263c25d99ce0380d002d59b67737d91b0dc # v7.4.0
|
||||
with:
|
||||
context: .
|
||||
platforms: linux/${{ matrix.arch }}
|
||||
provenance: false
|
||||
sbom: false
|
||||
push: true
|
||||
tags: |
|
||||
${{ env.IMAGE }}:${{ env.VERSION }}-${{ matrix.arch }}
|
||||
${{ matrix.arch == 'amd64' && format('{0}:{1}', env.IMAGE, env.VERSION) || '' }}
|
||||
|
||||
# Joins the two per-architecture tags into :<version> and :latest. Built
|
||||
# from the per-architecture tags rather than :<version>, which by now is
|
||||
# the amd64 image and would be read as such.
|
||||
index:
|
||||
needs: [version, publish]
|
||||
runs-on: ubuntu-latest
|
||||
env:
|
||||
VERSION: ${{ needs.version.outputs.version }}
|
||||
steps:
|
||||
- run: echo "IMAGE=${{ vars.REGISTRY }}/${GITHUB_REPOSITORY,,}" >> "$GITHUB_ENV"
|
||||
- uses: docker/setup-buildx-action@594f3bf4285d9ea8dc53c9a0c9c4092420091003 # v4.4.0
|
||||
- uses: docker/login-action@dbcb813823bdd20940b903addbd779551569679f # v4.6.0
|
||||
with:
|
||||
registry: ${{ vars.REGISTRY }}
|
||||
username: jcoffey-dev
|
||||
password: ${{ secrets.GITEA_TOKEN }}
|
||||
- run: |
|
||||
docker buildx imagetools create \
|
||||
--tag "$IMAGE:$VERSION" \
|
||||
--tag "$IMAGE:latest" \
|
||||
"$IMAGE:$VERSION-amd64" "$IMAGE:$VERSION-arm64"
|
||||
docker buildx imagetools inspect "$IMAGE:$VERSION"
|
||||
# Gitea keeps a container package on its owner; linking it shows it on
|
||||
# the repository's Packages tab. Idempotent.
|
||||
- env:
|
||||
GITEA_TOKEN: ${{ secrets.GITEA_TOKEN }}
|
||||
run: |
|
||||
owner="${GITHUB_REPOSITORY%%/*}"; name="${GITHUB_REPOSITORY#*/}"
|
||||
curl -fsS -o /dev/null -X POST -H "Authorization: token $GITEA_TOKEN" \
|
||||
"$GITEA_URL/api/v1/packages/${owner,,}/container/$name/-/link/$name" \
|
||||
|| echo "package already linked (or link refused); not fatal"
|
||||
|
||||
# The weekly release creates its Release (and so the tag) on Gitea first; a
|
||||
# tag pushed by hand has none. Either way the tag ends up with exactly one
|
||||
# Release there, created once the image exists so its pull instructions
|
||||
# work.
|
||||
release:
|
||||
needs: [version, index]
|
||||
runs-on: ubuntu-latest
|
||||
steps:
|
||||
- env:
|
||||
TAG: ${{ github.ref_name }}
|
||||
VERSION: ${{ needs.version.outputs.version }}
|
||||
GITEA_TOKEN: ${{ secrets.GITEA_TOKEN }}
|
||||
REGISTRY: ${{ vars.REGISTRY }}
|
||||
run: |
|
||||
set -euo pipefail
|
||||
api="$GITEA_URL/api/v1/repos/$GITHUB_REPOSITORY"
|
||||
code="$(curl -sS -o /dev/null -w '%{http_code}' -H "Authorization: token $GITEA_TOKEN" "$api/releases/tags/$TAG")"
|
||||
if [ "$code" = 200 ]; then echo "$TAG already has a release"; exit 0; fi
|
||||
[ "$code" = 404 ] || { echo "looking up the release for $TAG answered $code" >&2; exit 1; }
|
||||
image="$REGISTRY/${GITHUB_REPOSITORY,,}:$VERSION"
|
||||
body="Container image: \`$image\` (linux/amd64, linux/arm64); also \`:latest\`.
|
||||
|
||||
Binaries for a host install are attached: \`inbuxa-linux-amd64.tar.gz\` and \`inbuxa-linux-arm64.tar.gz\`, with \`SHA256SUMS\`. Each is the binary out of this release's image for that architecture, so it is the same build. The image grants it \`cap_net_bind_service\`; a host install has to grant that itself (\`setcap\`, or \`AmbientCapabilities\` in the unit) to bind port 25."
|
||||
jq -n --arg tag "$TAG" --arg name "INBUXA $VERSION" --arg body "$body" \
|
||||
'{tag_name:$tag, name:$name, body:$body}' |
|
||||
curl -fsS -X POST -H "Authorization: token $GITEA_TOKEN" -H 'Content-Type: application/json' \
|
||||
--data @- "$api/releases" | jq -r '"created release " + .tag_name'
|
||||
|
||||
# The binaries for a host install, taken out of the image that was just
|
||||
# pushed rather than compiled again: the binary in the tarball is the file
|
||||
# the image runs. `docker create` starts nothing, so copying a file out of
|
||||
# the arm64 image on an amd64 runner needs no emulation.
|
||||
binaries:
|
||||
needs: [version, index, release]
|
||||
runs-on: ubuntu-latest
|
||||
env:
|
||||
VERSION: ${{ needs.version.outputs.version }}
|
||||
TAG: ${{ github.ref_name }}
|
||||
GITEA_TOKEN: ${{ secrets.GITEA_TOKEN }}
|
||||
steps:
|
||||
- run: echo "IMAGE=${{ vars.REGISTRY }}/${GITHUB_REPOSITORY,,}" >> "$GITHUB_ENV"
|
||||
- uses: docker/login-action@dbcb813823bdd20940b903addbd779551569679f # v4.6.0
|
||||
with:
|
||||
registry: ${{ vars.REGISTRY }}
|
||||
username: jcoffey-dev
|
||||
password: ${{ secrets.GITEA_TOKEN }}
|
||||
- name: take the binaries out of the image
|
||||
run: |
|
||||
set -euo pipefail
|
||||
mkdir -p out && cd out
|
||||
for arch in amd64 arm64; do
|
||||
docker pull -q --platform "linux/$arch" "$IMAGE:$VERSION"
|
||||
id="$(docker create --platform "linux/$arch" "$IMAGE:$VERSION")"
|
||||
docker cp "$id:/usr/local/bin/inbuxa" inbuxa
|
||||
docker rm -f "$id" >/dev/null
|
||||
chmod 0755 inbuxa
|
||||
tar -czf "inbuxa-linux-$arch.tar.gz" inbuxa
|
||||
rm inbuxa
|
||||
done
|
||||
sha256sum inbuxa-linux-*.tar.gz > SHA256SUMS
|
||||
cat SHA256SUMS
|
||||
# A re-run of a tag replaces its assets rather than leaving two files
|
||||
# with the same name and different contents.
|
||||
#
|
||||
# The uploads cross Cloudflare, which dropped 50 MB HTTP/2 uploads
|
||||
# part-way for v2026.9.30.1 (curl 92, PROTOCOL_ERROR; origin logged
|
||||
# 400), once on each of two runs. Uploads go over HTTP/1.1 and retry.
|
||||
- name: attach them to the release
|
||||
run: |
|
||||
set -euo pipefail
|
||||
api="$GITEA_URL/api/v1/repos/$GITHUB_REPOSITORY"
|
||||
auth="Authorization: token $GITEA_TOKEN"
|
||||
retry=(--retry 5 --retry-all-errors --retry-delay 15)
|
||||
rel="$(curl -fsS "${retry[@]}" -H "$auth" "$api/releases/tags/$TAG" | jq -r .id)"
|
||||
assets="$(curl -fsS "${retry[@]}" -H "$auth" "$api/releases/$rel/assets")"
|
||||
for f in out/inbuxa-linux-amd64.tar.gz out/inbuxa-linux-arm64.tar.gz out/SHA256SUMS; do
|
||||
name="$(basename "$f")"
|
||||
old="$(jq -r --arg n "$name" '.[] | select(.name == $n) | .id' <<<"$assets")"
|
||||
for id in $old; do curl -fsS "${retry[@]}" -o /dev/null -X DELETE -H "$auth" "$api/releases/$rel/assets/$id"; done
|
||||
curl -fsS --http1.1 "${retry[@]}" -o /dev/null -X POST -H "$auth" -F "attachment=@$f" "$api/releases/$rel/assets?name=$name"
|
||||
echo "attached $name"
|
||||
done
|
||||
|
||||
# ------------------------------------------------------ ghcr replica ------
|
||||
# Copies the release image from the Gitea registry, which stays the
|
||||
# authoritative one, to ghcr.io under the same version tag and :latest. It is
|
||||
# a copy, not a second build: the digest on GHCR is the digest on the
|
||||
# registry, so `docker pull ghcr.io/...` gets exactly the same image. Left
|
||||
# out of the report to Gitea, like the release copy, so a GHCR problem
|
||||
# cannot fail a release.
|
||||
ghcr:
|
||||
if: ${{ vars.BUILD_ON == 'github' && github.ref_type == 'tag' }}
|
||||
needs: [version, index]
|
||||
runs-on: ubuntu-latest
|
||||
permissions:
|
||||
contents: read
|
||||
packages: write
|
||||
steps:
|
||||
- env:
|
||||
GH_TOKEN: ${{ github.token }}
|
||||
TAG: ${{ needs.version.outputs.version }}
|
||||
run: |
|
||||
set -euo pipefail
|
||||
src="${{ vars.REGISTRY }}/${GITHUB_REPOSITORY,,}"
|
||||
dst="ghcr.io/${GITHUB_REPOSITORY,,}"
|
||||
tag="$TAG"
|
||||
echo "$GH_TOKEN" | docker login ghcr.io -u "$GITHUB_ACTOR" --password-stdin
|
||||
docker buildx imagetools create -t "$dst:$tag" -t "$dst:latest" "$src:$tag"
|
||||
want="$(docker buildx imagetools inspect "$src:$tag" --format '{{json .Manifest.Digest}}')"
|
||||
got="$(docker buildx imagetools inspect "$dst:$tag" --format '{{json .Manifest.Digest}}')"
|
||||
echo "registry $src:$tag = $want"
|
||||
echo "ghcr $dst:$tag = $got"
|
||||
[ "$want" = "$got" ] || echo "::warning::GHCR digest differs from the registry's"
|
||||
docker logout ghcr.io
|
||||
|
||||
# ---------------------------------------------------- github release ------
|
||||
# Copies this tag's Gitea release -- notes and files -- to a GitHub release,
|
||||
# so the replica's Releases page, and anyone watching it, keeps up. Gitea's
|
||||
# release is the real one; this is left out of the report to Gitea, so a
|
||||
# failure here cannot fail a release. PR and issue numbers in the notes are
|
||||
# rewritten to Gitea links: on GitHub a bare #16 is some other PR.
|
||||
github-release:
|
||||
if: ${{ vars.BUILD_ON == 'github' && github.ref_type == 'tag' }}
|
||||
needs: [binaries]
|
||||
runs-on: ubuntu-latest
|
||||
permissions:
|
||||
contents: write
|
||||
env:
|
||||
GITEA_URL: ${{ vars.GITEA_URL }}
|
||||
GH_TOKEN: ${{ github.token }}
|
||||
TAG: ${{ github.ref_name }}
|
||||
steps:
|
||||
- run: |
|
||||
set -euo pipefail
|
||||
if gh release view "$TAG" --repo "$GITHUB_REPOSITORY" >/dev/null 2>&1; then
|
||||
echo "GitHub already has a release for $TAG"; exit 0
|
||||
fi
|
||||
# The Gitea release exists by now if this run made it; if the weekly
|
||||
# release job made it, it came before the tag. Allow a few minutes.
|
||||
code=0
|
||||
for _ in $(seq 1 15); do
|
||||
code="$(curl -sS -o rel.json -w '%{http_code}' "$GITEA_URL/api/v1/repos/$GITHUB_REPOSITORY/releases/tags/$TAG")"
|
||||
[ "$code" = 200 ] && break
|
||||
sleep 20
|
||||
done
|
||||
if [ "$code" != 200 ]; then echo "No Gitea release for $TAG; nothing to copy"; exit 0; fi
|
||||
if [ "$(jq -r .draft rel.json)" = true ]; then echo "The Gitea release is a draft; not copying"; exit 0; fi
|
||||
export BASE="$(jq -r '.html_url | sub("/releases/tag/.*$"; "")' rel.json)"
|
||||
jq -r '.body // ""' rel.json | perl -pe 's{(?<![\w/&\[])#(\d+)\b}{[#$1]($ENV{BASE}/pulls/$1)}g' > notes.md
|
||||
printf '\n\n_Mirrored from [the Gitea release](%s); report issues on [Gitea](%s/issues)._\n' \
|
||||
"$(jq -r .html_url rel.json)" "$BASE" >> notes.md
|
||||
files=()
|
||||
mkdir -p files
|
||||
while IFS=$'\t' read -r name url; do
|
||||
curl -fsSL -o "files/$name" "$url"; files+=("files/$name")
|
||||
done < <(jq -r '.assets[]? | [.name, .browser_download_url] | @tsv' rel.json)
|
||||
title="$(jq -r '.name // ""' rel.json)"; [ -n "$title" ] || title="$TAG"
|
||||
if [ "$(jq -r .prerelease rel.json)" = true ]; then kind=--prerelease; else kind=--latest; fi
|
||||
gh release create "$TAG" --repo "$GITHUB_REPOSITORY" --verify-tag --title "$title" \
|
||||
--notes-file notes.md "$kind" "${files[@]}"
|
||||
echo "created the GitHub release for $TAG with ${#files[@]} file(s)"
|
||||
|
||||
# ------------------------------------------------------------- report ------
|
||||
# One commit status on Gitea for the whole run: what Gitea's ci.yml and
|
||||
# publish.yml wait on. Skipped jobs (the tag jobs on a branch, and the other
|
||||
# way round) count as passing; a failed or cancelled one does not.
|
||||
report:
|
||||
if: ${{ always() && vars.BUILD_ON == 'github' }}
|
||||
needs: [start, fork-checks, build, version, publish, index, release, binaries]
|
||||
runs-on: ubuntu-latest
|
||||
steps:
|
||||
- env:
|
||||
GITEA_TOKEN: ${{ secrets.GITEA_TOKEN }}
|
||||
STATE: ${{ contains(needs.*.result, 'failure') && 'failure' || (contains(needs.*.result, 'cancelled') && 'cancelled' || 'success') }}
|
||||
run: |
|
||||
# A cancelled run was superseded by a newer run for the same commit (the
|
||||
# mirror can push one commit twice); that run reports. Posting "failure"
|
||||
# here would fail the Gitea check while the real build is still going.
|
||||
if [ "$STATE" = cancelled ]; then echo "cancelled: leaving the result to the newer run"; exit 0; fi
|
||||
jq -n --arg s "$STATE" --arg c "$STATUS_CONTEXT" \
|
||||
--arg u "$GITHUB_SERVER_URL/$GITHUB_REPOSITORY/actions/runs/$GITHUB_RUN_ID" \
|
||||
'{state:$s, context:$c, target_url:$u, description:"GitHub Actions"}' |
|
||||
curl -fsS -o /dev/null -X POST -H "Authorization: token $GITEA_TOKEN" \
|
||||
-H 'Content-Type: application/json' --data @- \
|
||||
"$GITEA_URL/api/v1/repos/$GITHUB_REPOSITORY/statuses/$GITHUB_SHA"
|
||||
echo "$STATUS_CONTEXT: $STATE"
|
||||
@@ -1,69 +0,0 @@
|
||||
# Prune old image versions from GHCR.
|
||||
#
|
||||
# Releases are kept forever -- they carry no assets and their generated notes
|
||||
# are this project's only changelog, so deleting one destroys history that
|
||||
# cannot be reconstructed for nothing saved. Images are the opposite: a
|
||||
# multi-arch build a week, and the by-digest push in publish.yml leaves two
|
||||
# untagged per-architecture manifests behind each time on top of the tagged
|
||||
# index. Those accumulate and nobody wants fifty of them.
|
||||
#
|
||||
# THE FOOTGUN: the obvious tool for this -- delete-package-versions with
|
||||
# `delete-only-untagged-versions` -- will happily delete the per-architecture
|
||||
# manifests that a multi-arch tag points *at*, because they are untagged by
|
||||
# design. Nothing appears to break: the tag still exists, and pulls simply
|
||||
# start failing for one architecture. This action understands manifest lists
|
||||
# and will not orphan a retained index, and `validate` re-checks every
|
||||
# multi-arch manifest against the registry afterwards.
|
||||
#
|
||||
# Separate from publish.yml, and dispatchable on its own, so `dry_run` can show
|
||||
# exactly what would be deleted without rebuilding and re-pushing an image to
|
||||
# find out.
|
||||
name: Prune images
|
||||
|
||||
on:
|
||||
workflow_call:
|
||||
inputs:
|
||||
dry_run:
|
||||
type: boolean
|
||||
default: false
|
||||
workflow_dispatch:
|
||||
inputs:
|
||||
dry_run:
|
||||
description: "List what would be deleted, delete nothing"
|
||||
type: boolean
|
||||
default: true
|
||||
|
||||
jobs:
|
||||
prune:
|
||||
runs-on: ubuntu-latest
|
||||
permissions:
|
||||
packages: write
|
||||
steps:
|
||||
# The only third-party action here that is not published by GitHub or
|
||||
# Docker, and the one with the most to lose: it is handed
|
||||
# `packages: write` and its whole job is deletion, so a ref repointed at
|
||||
# something else -- by a compromise or a mistake upstream -- is a bad
|
||||
# day. It was pinned to a commit long before the rest of them were.
|
||||
- uses: dataaxiom/ghcr-cleanup-action@d52806a0dc70b430571a37da1fde39733ffd640f # v1.2.2
|
||||
with:
|
||||
owner: inbuxa
|
||||
package: inbuxa-server
|
||||
token: ${{ secrets.GITHUB_TOKEN }}
|
||||
# Ten weekly releases is roughly a quarter of history, which is more
|
||||
# than enough to roll back to and far less than the year's worth that
|
||||
# would otherwise pile up. Older *releases* stay either way; this
|
||||
# only removes the images.
|
||||
keep-n-tagged: 10
|
||||
# Belt and braces on top of the action's own manifest awareness:
|
||||
# `latest` is never a candidate for deletion under any counting.
|
||||
exclude-tags: latest
|
||||
delete-untagged: true
|
||||
# Sweeps the wreckage of a half-failed run: an index whose platform
|
||||
# images did not all land, and referrers whose parent is gone.
|
||||
delete-partial-images: true
|
||||
delete-orphaned-images: true
|
||||
# Checks every remaining multi-architecture manifest still resolves
|
||||
# in the registry. This is the step that would catch the footgun
|
||||
# above rather than leaving a reader to discover it on `docker pull`.
|
||||
validate: true
|
||||
dry-run: ${{ inputs.dry_run }}
|
||||
@@ -1,198 +0,0 @@
|
||||
# Publish the container image to GHCR.
|
||||
#
|
||||
# The README and the docs site have told people to run
|
||||
# `ghcr.io/inbuxa/inbuxa-server:latest` for a long time, and nothing ever
|
||||
# pushed it: `docker pull` answered `denied`, because the package did not
|
||||
# exist. This is the workflow that makes those instructions true. It is also
|
||||
# the prerequisite for the self-hosted app catalogs -- TrueNAS and Unraid
|
||||
# both install by pulling an image and neither builds from source.
|
||||
#
|
||||
# FIRST RUN: a package GHCR creates for the first time is **private**, even in
|
||||
# a public repository, and an anonymous `docker pull` will still answer
|
||||
# `denied`. Nothing in a workflow can change that -- the visibility is set once
|
||||
# by hand under the package's settings, and until it is, this looks like it
|
||||
# worked while the docs stay just as wrong as before. Check with a logged-out
|
||||
# pull, not with one from a machine that has credentials.
|
||||
#
|
||||
# Two architectures, each built on its own native runner rather than under
|
||||
# QEMU. Emulated arm64 has to run `npm ci` and the Vite build through
|
||||
# instruction translation, which takes tens of minutes and occasionally runs
|
||||
# out of memory; `ubuntu-24.04-arm` is free for public repositories and does
|
||||
# the same work at native speed. The cost is the by-digest dance below: each
|
||||
# runner pushes an untagged image, and a final job joins the two digests into
|
||||
# one multi-arch tag.
|
||||
name: Publish image
|
||||
|
||||
on:
|
||||
release:
|
||||
types: [published]
|
||||
# Callable, so release.yml can build the release it just cut. This is not a
|
||||
# stylistic choice: a release created with GITHUB_TOKEN does **not** raise a
|
||||
# `release` event -- GitHub refuses to let a token trigger another workflow,
|
||||
# to stop a workflow looping on its own output. A scheduled job that cut a
|
||||
# release and expected this file to notice would silently never publish. The
|
||||
# alternatives are a personal access token kept as a secret, or calling the
|
||||
# workflow directly. This is the one that needs no credential.
|
||||
workflow_call:
|
||||
inputs:
|
||||
ref:
|
||||
description: "Tag, branch or SHA to build"
|
||||
required: true
|
||||
type: string
|
||||
tag_latest:
|
||||
description: "Also move :latest to this build"
|
||||
type: boolean
|
||||
default: false
|
||||
# Same reasoning as ci.yml's dispatch trigger: a run GitHub queues and then
|
||||
# orphans can be neither rerun nor canceled, and this workflow otherwise
|
||||
# only fires on a release -- which is not something to cut twice because a
|
||||
# runner died. `ref` also allows publishing an image for a tag that predates
|
||||
# this workflow, which is how the first one gets built.
|
||||
workflow_dispatch:
|
||||
inputs:
|
||||
ref:
|
||||
description: "Tag, branch or SHA to build"
|
||||
required: true
|
||||
default: main
|
||||
tag_latest:
|
||||
description: "Also move :latest to this build"
|
||||
type: boolean
|
||||
default: false
|
||||
|
||||
env:
|
||||
# Hardcoded rather than derived from github.repository, which would have to
|
||||
# be lowercased to be a legal registry path. This is the string the docs name.
|
||||
IMAGE: ghcr.io/inbuxa/inbuxa-server
|
||||
|
||||
jobs:
|
||||
# The version is read once and handed to both builds, so the two
|
||||
# architectures cannot disagree about what they are. It is read from the
|
||||
# macro the binary itself compiles in, which the weekly release commits
|
||||
# before this runs -- so the image is tagged with the version it reports.
|
||||
version:
|
||||
runs-on: ubuntu-latest
|
||||
outputs:
|
||||
version: ${{ steps.v.outputs.version }}
|
||||
steps:
|
||||
- uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1
|
||||
with:
|
||||
ref: ${{ inputs.ref || github.ref }}
|
||||
- id: v
|
||||
run: |
|
||||
set -euo pipefail
|
||||
# Scoped to the macro body: branding.rs holds other string literals,
|
||||
# and tagging an image from one of those would be worse than failing.
|
||||
V="$(awk '/macro_rules! brand_version/,/^}/' crates/types/src/branding.rs \
|
||||
| grep -om1 '"[0-9][^"]*"' | tr -d '"')"
|
||||
[ -n "$V" ] || { echo "could not read brand_version! from branding.rs" >&2; exit 1; }
|
||||
# A date version carries nothing a Docker tag objects to, so there is
|
||||
# no second, sanitized form of it here.
|
||||
echo "version=$V" >> "$GITHUB_OUTPUT"
|
||||
echo "version $V"
|
||||
|
||||
build:
|
||||
needs: version
|
||||
runs-on: ${{ matrix.runner }}
|
||||
permissions:
|
||||
contents: read
|
||||
packages: write
|
||||
strategy:
|
||||
fail-fast: false
|
||||
matrix:
|
||||
include:
|
||||
- platform: linux/amd64
|
||||
runner: ubuntu-latest
|
||||
- platform: linux/arm64
|
||||
runner: ubuntu-24.04-arm
|
||||
steps:
|
||||
- uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1
|
||||
with:
|
||||
ref: ${{ inputs.ref || github.ref }}
|
||||
- uses: docker/setup-buildx-action@594f3bf4285d9ea8dc53c9a0c9c4092420091003 # v4.4.0
|
||||
- uses: docker/login-action@dbcb813823bdd20940b903addbd779551569679f # v4.6.0
|
||||
with:
|
||||
registry: ghcr.io
|
||||
username: ${{ github.actor }}
|
||||
password: ${{ secrets.GITHUB_TOKEN }}
|
||||
- name: Build and push by digest
|
||||
id: push
|
||||
uses: docker/build-push-action@c3c9e263c25d99ce0380d002d59b67737d91b0dc # v7.4.0
|
||||
with:
|
||||
context: .
|
||||
platforms: ${{ matrix.platform }}
|
||||
# Attestations are off deliberately: they add manifests of their own
|
||||
# to the index, and `imagetools create` below expects the two entries
|
||||
# it pushed rather than four.
|
||||
provenance: false
|
||||
sbom: false
|
||||
cache-from: type=gha,scope=${{ matrix.platform }}
|
||||
cache-to: type=gha,mode=max,scope=${{ matrix.platform }}
|
||||
outputs: type=image,name=${{ env.IMAGE }},push-by-digest=true,name-canonical=true,push=true
|
||||
- name: Save the digest
|
||||
run: |
|
||||
mkdir -p /tmp/digests
|
||||
# The prefix is stripped here and put back in the merge job, so the
|
||||
# filename is the bare hash. Leaving it on produces
|
||||
# `image@sha256:sha256:...` when the reference is rebuilt.
|
||||
digest="${{ steps.push.outputs.digest }}"
|
||||
touch "/tmp/digests/${digest#sha256:}"
|
||||
- uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7.0.1
|
||||
with:
|
||||
# One artifact per platform; the merge job globs them back together.
|
||||
name: digest-${{ strategy.job-index }}
|
||||
path: /tmp/digests/*
|
||||
retention-days: 1
|
||||
if-no-files-found: error
|
||||
|
||||
# Joins the per-architecture digests into a single tagged manifest, so
|
||||
# `docker pull ghcr.io/inbuxa/inbuxa-server:<tag>` resolves on both.
|
||||
publish:
|
||||
needs: [version, build]
|
||||
runs-on: ubuntu-latest
|
||||
permissions:
|
||||
contents: read
|
||||
packages: write
|
||||
steps:
|
||||
- uses: actions/download-artifact@3e5f45b2cfb9172054b4087a40e8e0b5a5461e7c # v8.0.1
|
||||
with:
|
||||
path: /tmp/digests
|
||||
pattern: digest-*
|
||||
merge-multiple: true
|
||||
- uses: docker/setup-buildx-action@594f3bf4285d9ea8dc53c9a0c9c4092420091003 # v4.4.0
|
||||
- uses: docker/login-action@dbcb813823bdd20940b903addbd779551569679f # v4.6.0
|
||||
with:
|
||||
registry: ghcr.io
|
||||
username: ${{ github.actor }}
|
||||
password: ${{ secrets.GITHUB_TOKEN }}
|
||||
- name: Create the manifest
|
||||
run: |
|
||||
# Arrays rather than a string: the tags and the digest references
|
||||
# have to reach docker as separate arguments, and building them by
|
||||
# word-splitting an unquoted variable is the version of this that
|
||||
# breaks the day a value contains a space.
|
||||
tags=(-t "${IMAGE}:${{ needs.version.outputs.version }}")
|
||||
# :latest follows real releases only. A prerelease that moved it
|
||||
# would hand every `:latest` deployment an unfinished build, and a
|
||||
# dispatch run has to ask for it on purpose.
|
||||
if [ "${{ github.event_name }}" = "release" ] && [ "${{ github.event.release.prerelease }}" = "false" ]; then
|
||||
tags+=(-t "${IMAGE}:latest")
|
||||
elif [ "${{ inputs.tag_latest }}" = "true" ]; then
|
||||
tags+=(-t "${IMAGE}:latest")
|
||||
fi
|
||||
refs=()
|
||||
for f in /tmp/digests/*; do
|
||||
refs+=("${IMAGE}@sha256:$(basename "$f")")
|
||||
done
|
||||
echo "tags: ${tags[*]}"
|
||||
echo "refs: ${refs[*]}"
|
||||
docker buildx imagetools create "${tags[@]}" "${refs[@]}"
|
||||
- name: Show what landed
|
||||
run: docker buildx imagetools inspect "${IMAGE}:${{ needs.version.outputs.version }}"
|
||||
|
||||
# Runs only after a successful publish, because that is the only moment the
|
||||
# package grows. See cleanup.yml for why this is not the obvious one-liner.
|
||||
prune:
|
||||
needs: publish
|
||||
permissions:
|
||||
packages: write
|
||||
uses: ./.github/workflows/cleanup.yml
|
||||
@@ -1,246 +0,0 @@
|
||||
# Cut a release once a week, but only if there is something in it.
|
||||
#
|
||||
# It does nothing on a quiet week. A release with no commits in it is worse
|
||||
# than no release: it moves `:latest` to an identical build, spends a version
|
||||
# number, and mails everybody watching the repository about nothing.
|
||||
#
|
||||
# INBUXA's version is a string in crates/types/src/branding.rs, deliberately
|
||||
# not in Cargo.toml so that upstream's version bumps merge without conflicts.
|
||||
# So this writes it: the bump is committed to main, and the tag names that
|
||||
# commit. The tree a tag points at therefore reports the version the tag
|
||||
# claims, which a tag placed beside an unbumped macro cannot promise.
|
||||
name: Weekly release
|
||||
|
||||
on:
|
||||
schedule:
|
||||
# Mondays, 10:07 UTC, and last of the three: INBUXA Admin and the webmail
|
||||
# release ahead of the server they talk to. Staggered rather than
|
||||
# simultaneous so three releases do not compete for runners, and so a bad
|
||||
# Monday names one repository instead of three. GitHub runs scheduled jobs
|
||||
# best-effort and can delay a run considerably, so the exact minute is not
|
||||
# a promise; the odd minute keeps it off the crowded top of the hour.
|
||||
#
|
||||
# Note also that GitHub disables scheduled workflows in a repository with
|
||||
# no activity for 60 days, which is worth checking for before assuming
|
||||
# this file is broken.
|
||||
- cron: "7 10 * * 1"
|
||||
workflow_dispatch:
|
||||
inputs:
|
||||
dry_run:
|
||||
description: "Work out what would be released, then stop"
|
||||
type: boolean
|
||||
default: false
|
||||
|
||||
# One at a time. Two overlapping runs would race to write the same version and
|
||||
# create the same tag, and the loser fails noisily for a reason that has
|
||||
# nothing to do with the code.
|
||||
concurrency:
|
||||
group: weekly-release
|
||||
cancel-in-progress: false
|
||||
|
||||
jobs:
|
||||
check:
|
||||
runs-on: ubuntu-latest
|
||||
permissions:
|
||||
contents: read
|
||||
outputs:
|
||||
should_release: ${{ steps.decide.outputs.should_release }}
|
||||
version: ${{ steps.decide.outputs.version }}
|
||||
tag: ${{ steps.decide.outputs.tag }}
|
||||
previous: ${{ steps.decide.outputs.previous }}
|
||||
count: ${{ steps.decide.outputs.count }}
|
||||
steps:
|
||||
- uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1
|
||||
with:
|
||||
ref: main
|
||||
fetch-depth: 0
|
||||
- id: decide
|
||||
env:
|
||||
GH_TOKEN: ${{ secrets.GITHUB_TOKEN }}
|
||||
run: |
|
||||
set -euo pipefail
|
||||
|
||||
# The newest published release, or empty on a repository that has
|
||||
# never had one -- in which case everything counts as new. Drafts are
|
||||
# excluded: an unpublished draft is not a release anybody has, so
|
||||
# counting from it would hide commits that have never shipped.
|
||||
previous="$(gh release list --limit 1 --exclude-drafts --json tagName --jq '.[0].tagName // ""')"
|
||||
# A tag named by a release is normally present after a full checkout,
|
||||
# but a release can outlive its tag. Falling back to the whole
|
||||
# history is the safe direction to be wrong in: it over-counts, which
|
||||
# cuts a release that was due anyway, where under-counting would skip
|
||||
# one that was.
|
||||
if [ -n "$previous" ] && git rev-parse -q --verify "refs/tags/${previous}" >/dev/null; then
|
||||
count="$(git rev-list --count "${previous}..HEAD")"
|
||||
else
|
||||
count="$(git rev-list --count HEAD)"
|
||||
fi
|
||||
|
||||
# INBUXA's version is the date: YYYY.M.D, unpadded, as branding.rs
|
||||
# documents. A second release on one day takes a `.N` suffix,
|
||||
# counting from 2, which is why this asks the tags rather than
|
||||
# assuming today is free.
|
||||
today="$(date -u +%Y.%-m.%-d)"
|
||||
version="$today"
|
||||
n=2
|
||||
while git rev-parse -q --verify "refs/tags/v${version}" >/dev/null; do
|
||||
version="${today}.${n}"
|
||||
n=$((n + 1))
|
||||
done
|
||||
|
||||
should_release=true
|
||||
reason=""
|
||||
if [ "$count" -eq 0 ]; then
|
||||
should_release=false
|
||||
reason="no commits since ${previous}"
|
||||
fi
|
||||
|
||||
{
|
||||
echo "should_release=$should_release"
|
||||
echo "version=$version"
|
||||
echo "tag=v${version}"
|
||||
echo "previous=$previous"
|
||||
echo "count=$count"
|
||||
} >> "$GITHUB_OUTPUT"
|
||||
|
||||
# Written to the run summary so a skipped week reads as a decision
|
||||
# rather than as a workflow that quietly did nothing.
|
||||
{
|
||||
echo "### Weekly release"
|
||||
echo
|
||||
if [ "$should_release" = "true" ]; then
|
||||
echo "Releasing **v${version}** — ${count} commit(s) since ${previous:-the beginning}."
|
||||
else
|
||||
echo "Nothing to release: ${reason}."
|
||||
fi
|
||||
} >> "$GITHUB_STEP_SUMMARY"
|
||||
|
||||
cut:
|
||||
needs: check
|
||||
if: needs.check.outputs.should_release == 'true' && !inputs.dry_run
|
||||
runs-on: ubuntu-latest
|
||||
permissions:
|
||||
contents: write
|
||||
pull-requests: write
|
||||
outputs:
|
||||
sha: ${{ steps.land.outputs.sha }}
|
||||
steps:
|
||||
- uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1
|
||||
with:
|
||||
ref: main
|
||||
fetch-depth: 0
|
||||
- id: bump
|
||||
env:
|
||||
VERSION: ${{ needs.check.outputs.version }}
|
||||
BRANCH: release/v${{ needs.check.outputs.version }}
|
||||
run: |
|
||||
set -euo pipefail
|
||||
|
||||
# Scoped to the macro body rather than replacing the first quoted
|
||||
# string in the file, and asserted to have matched exactly once.
|
||||
# branding.rs holds other string literals, and a bump that silently
|
||||
# edited one of those -- or none -- would ship a build whose version
|
||||
# disagrees with its tag.
|
||||
python3 - <<'PY'
|
||||
import os, re
|
||||
path = "crates/types/src/branding.rs"
|
||||
src = open(path, encoding="utf-8").read()
|
||||
pattern = re.compile(r'(macro_rules! brand_version \{\s*\(\) => \{\s*")[^"]+(")')
|
||||
out, n = pattern.subn(lambda m: m.group(1) + os.environ["VERSION"] + m.group(2), src, count=1)
|
||||
assert n == 1, f"brand_version! not found in {path}"
|
||||
open(path, "w", encoding="utf-8").write(out)
|
||||
PY
|
||||
|
||||
git config user.name "github-actions[bot]"
|
||||
git config user.email "41898282+github-actions[bot]@users.noreply.github.com"
|
||||
git add crates/types/src/branding.rs
|
||||
git commit -m "Version ${VERSION}"
|
||||
git push origin "HEAD:refs/heads/${BRANCH}"
|
||||
|
||||
# main is protected: it takes a pull request with a green build, and
|
||||
# GITHUB_TOKEN is not among the bypass actors. So the bump lands the way
|
||||
# every other change does. The alternative was to hand the release a
|
||||
# credential that outranks the rule, which is a worse thing to own than
|
||||
# a slower Monday.
|
||||
- id: land
|
||||
env:
|
||||
VERSION: ${{ needs.check.outputs.version }}
|
||||
BRANCH: release/v${{ needs.check.outputs.version }}
|
||||
GH_TOKEN: ${{ secrets.GITHUB_TOKEN }}
|
||||
run: |
|
||||
set -euo pipefail
|
||||
|
||||
url="$(gh pr create --base main --head "${BRANCH}" \
|
||||
--title "Version ${VERSION}" \
|
||||
--body "Weekly release. Bumps \`brand_version!\` to ${VERSION} so the tag names a tree that reports the version the tag claims.")"
|
||||
# The number, not the branch: the branch is deleted on merge, and a
|
||||
# deleted branch no longer resolves to its pull request.
|
||||
pr="${url##*/}"
|
||||
echo "Opened #${pr}"
|
||||
|
||||
# The build is what the rule actually requires, and it is also the
|
||||
# thing worth waiting for: a release cut from a tree that does not
|
||||
# compile is the failure this whole arrangement exists to prevent.
|
||||
# A full build of this tree is long, so the deadline is generous.
|
||||
deadline=$(( SECONDS + 3600 ))
|
||||
while :; do
|
||||
state="$(gh pr view "${pr}" --json statusCheckRollup \
|
||||
--jq '[.statusCheckRollup[]? | .conclusion // "PENDING"] | join(",")')"
|
||||
case "${state}" in
|
||||
*FAILURE*|*CANCELLED*|*TIMED_OUT*)
|
||||
echo "::error::CI failed on ${BRANCH} (${state}); no release cut. PR #${pr} is left open."
|
||||
exit 1 ;;
|
||||
*SUCCESS*) break ;;
|
||||
esac
|
||||
if [ "${SECONDS}" -ge "${deadline}" ]; then
|
||||
echo "::error::timed out waiting for CI on ${BRANCH}. PR #${pr} is left open."
|
||||
exit 1
|
||||
fi
|
||||
sleep 30
|
||||
done
|
||||
|
||||
gh pr merge "${pr}" --rebase --delete-branch
|
||||
|
||||
# A rebase merge rewrites the commit, so the sha to tag is the one
|
||||
# GitHub recorded for the merge, not the tip that was pushed. It can
|
||||
# take a moment to appear.
|
||||
sha=""
|
||||
for _ in $(seq 1 30); do
|
||||
sha="$(gh pr view "${pr}" --json mergeCommit --jq '.mergeCommit.oid // ""')"
|
||||
[ -n "${sha}" ] && break
|
||||
sleep 5
|
||||
done
|
||||
if [ -z "${sha}" ]; then
|
||||
echo "::error::#${pr} merged but GitHub reported no merge commit; nothing safe to tag."
|
||||
exit 1
|
||||
fi
|
||||
|
||||
echo "sha=${sha}" >> "$GITHUB_OUTPUT"
|
||||
- env:
|
||||
GH_TOKEN: ${{ secrets.GITHUB_TOKEN }}
|
||||
run: |
|
||||
set -euo pipefail
|
||||
args=(--target "${{ steps.land.outputs.sha }}"
|
||||
--title "INBUXA ${{ needs.check.outputs.version }}"
|
||||
--generate-notes)
|
||||
# Bound the notes to what is actually new. Without a start tag the
|
||||
# generator reaches back to whatever it decides is previous, which on
|
||||
# a repository carrying upstream's tag shapes is not always the last
|
||||
# release.
|
||||
if [ -n "${{ needs.check.outputs.previous }}" ]; then
|
||||
args+=(--notes-start-tag "${{ needs.check.outputs.previous }}")
|
||||
fi
|
||||
gh release create "${{ needs.check.outputs.tag }}" "${args[@]}"
|
||||
|
||||
# Called rather than left to the `release` trigger on purpose: see the note
|
||||
# at the top of publish.yml. A release created with GITHUB_TOKEN raises no
|
||||
# event, so without this the tag would exist and no image would follow it.
|
||||
publish:
|
||||
needs: [check, cut]
|
||||
permissions:
|
||||
contents: read
|
||||
packages: write
|
||||
uses: ./.github/workflows/publish.yml
|
||||
with:
|
||||
ref: ${{ needs.cut.outputs.sha }}
|
||||
tag_latest: true
|
||||
@@ -2,6 +2,33 @@
|
||||
|
||||
All notable changes to this project will be documented in this file. This project adheres to [Semantic Versioning](http://semver.org/).
|
||||
|
||||
## [0.16.25] - 2026-10-05
|
||||
|
||||
If you are upgrading from v0.16.x, replace the binary (or run `docker pull`). If you are upgrading from v0.15.x and below, please read the [upgrading documentation](https://github.com/stalwartlabs/stalwart/blob/main/UPGRADING/v0_16.md) for more information on how to upgrade from previous versions.
|
||||
|
||||
## Added
|
||||
|
||||
## Changed
|
||||
|
||||
## Fixed
|
||||
- JMAP: Creating a `MaskedEmail` with `emailDomain` fails with `forbidden` for every domain when the account has addresses on more than one domain.
|
||||
- Autodiscover: Requests for a response schema other than Outlook's, such as ActiveSync (`mobilesync`), are answered with the Outlook settings instead of error 601.
|
||||
- IMAP:
|
||||
- `LOGIN` and `AUTHENTICATE` with a wrong, expired or unknown app password or API key are answered with an untagged `NO`, so clients keep waiting for the command to complete until the connection times out.
|
||||
- The failed login that exceeds the maximum number of authentication failures is answered with an untagged `NO` before the connection is closed.
|
||||
- DKIM:
|
||||
- A rotation moves the active key to retiring even when its successor fails to publish or propagate, so outgoing mail is sent unsigned until a retry publishes the new key. The DNS write failure is also not logged and the task reports success.
|
||||
- Keys created while DNS management was manual, or before DKIM was added to the published records, are never rotated after DNS management becomes automatic. Domains already affected start rotating once a `DkimManagement` task is created for them.
|
||||
- After switching DNS management from automatic to manual, a due rotation activates a new key that was never published in DNS, so signatures fail verification, and retiring the old key is retried forever.
|
||||
- Spam filter:
|
||||
- Messages with no text line long enough for a Pyzor digest are checked with the digest of empty input and tagged `PYZOR`.
|
||||
- DNSBL answers with several return codes, such as a Spamhaus ZEN listing in both SBL and PBL, are scored for only the first code returned.
|
||||
- DNSBL lookups that return "not listed" are cached for 24 hours regardless of the zone's negative TTL.
|
||||
- Removing a duplicate training sample of a message reclassified on the same day clears the blob link of the sample that is kept.
|
||||
- MTA: Queue quotas with an empty `match` expression are never enforced, including the global queue quota created on first start.
|
||||
- RocksDB: The info log (`LOG`, `LOG.old.*`) grows without limit because log rotation and retention are left at RocksDB defaults.
|
||||
- WebUI: A blob store read error at startup, such as an S3 authentication failure, stops the web interface from being downloaded.
|
||||
|
||||
## [0.16.24] - 2026-09-27
|
||||
|
||||
If you are upgrading from v0.16.x, replace the binary (or run `docker pull`). If you are upgrading from v0.15.x and below, please read the [upgrading documentation](https://github.com/stalwartlabs/stalwart/blob/main/UPGRADING/v0_16.md) for more information on how to upgrade from previous versions.
|
||||
|
||||
+1
-1
@@ -60,7 +60,7 @@ representative at an online or offline event.
|
||||
|
||||
Instances of abusive, harassing, or otherwise unacceptable behavior may be
|
||||
reported to the community leaders responsible for enforcement at
|
||||
**johnellisATlinuxDOTcom**.
|
||||
**communityATcoffeylabsDOTorg**.
|
||||
All complaints will be reviewed and investigated promptly and fairly.
|
||||
|
||||
All community leaders are obligated to respect the privacy and security of the
|
||||
|
||||
+1
-1
@@ -54,7 +54,7 @@ Coffey Labs" line in place. New files carry:
|
||||
|
||||
```
|
||||
/*
|
||||
* SPDX-FileCopyrightText: 2026 Coffey Labs
|
||||
* SPDX-FileCopyrightText: 2026 Coffey Labs LLC
|
||||
*
|
||||
* SPDX-License-Identifier: AGPL-3.0-only
|
||||
*/
|
||||
|
||||
Generated
+87
-69
@@ -277,9 +277,9 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "async-compression"
|
||||
version = "0.4.48"
|
||||
version = "0.4.50"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "fb61aea1a7def73ee7c350a184f0e70b32c182344e2e75bf70c9b621b83417fd"
|
||||
checksum = "ee19bd99b43e3691acbad4e840420a4881cea6c0b66a208125a824f8fd53f5a1"
|
||||
dependencies = [
|
||||
"compression-codecs",
|
||||
"compression-core",
|
||||
@@ -1292,7 +1292,7 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "common"
|
||||
version = "0.16.24"
|
||||
version = "0.16.25"
|
||||
dependencies = [
|
||||
"aes-gcm-siv",
|
||||
"ahash",
|
||||
@@ -1392,9 +1392,9 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "compression-codecs"
|
||||
version = "0.4.43"
|
||||
version = "0.4.45"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "bef16c47ba2797aa6a909cc37d39911f3a6743811fe7408ac0b0cc0276b656e9"
|
||||
checksum = "98fc98460ba0ad5317075d3632b8dfc45d0be8c4a49347c2a38272019717614a"
|
||||
dependencies = [
|
||||
"compression-core",
|
||||
"flate2",
|
||||
@@ -1477,7 +1477,7 @@ checksum = "3d52eff69cd5e647efe296129160853a42795992097e8af39800e1060caeea9b"
|
||||
|
||||
[[package]]
|
||||
name = "coordinator"
|
||||
version = "0.16.24"
|
||||
version = "0.16.25"
|
||||
dependencies = [
|
||||
"async-nats",
|
||||
"futures",
|
||||
@@ -1889,7 +1889,7 @@ checksum = "4583a4551df46e2792f82ceeac45e850d2e2d5debba0b91f102385cda5b11f06"
|
||||
|
||||
[[package]]
|
||||
name = "dav"
|
||||
version = "0.16.24"
|
||||
version = "0.16.25"
|
||||
dependencies = [
|
||||
"calcard",
|
||||
"chrono",
|
||||
@@ -1912,7 +1912,7 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "dav-proto"
|
||||
version = "0.16.24"
|
||||
version = "0.16.25"
|
||||
dependencies = [
|
||||
"calcard",
|
||||
"chrono",
|
||||
@@ -2125,7 +2125,7 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "directory"
|
||||
version = "0.16.24"
|
||||
version = "0.16.25"
|
||||
dependencies = [
|
||||
"ahash",
|
||||
"argon2 0.6.0",
|
||||
@@ -2366,7 +2366,7 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "email"
|
||||
version = "0.16.24"
|
||||
version = "0.16.25"
|
||||
dependencies = [
|
||||
"aes 0.9.3",
|
||||
"aes-gcm 0.11.1",
|
||||
@@ -2406,6 +2406,16 @@ dependencies = [
|
||||
"log",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "encodify"
|
||||
version = "1.0.1"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "798c447647dd23f673748f2b868ef309a01dd86aaae999182559d36f06ac82f0"
|
||||
dependencies = [
|
||||
"memchr",
|
||||
"simdutf8",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "encoding_rs"
|
||||
version = "0.8.42"
|
||||
@@ -2474,7 +2484,7 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "event_macro"
|
||||
version = "0.16.24"
|
||||
version = "0.16.25"
|
||||
dependencies = [
|
||||
"quote",
|
||||
"syn 3.0.6",
|
||||
@@ -3002,7 +3012,7 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "groupware"
|
||||
version = "0.16.24"
|
||||
version = "0.16.25"
|
||||
dependencies = [
|
||||
"ahash",
|
||||
"calcard",
|
||||
@@ -3289,7 +3299,7 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "http"
|
||||
version = "0.16.24"
|
||||
version = "0.16.25"
|
||||
dependencies = [
|
||||
"async-stream",
|
||||
"base64 0.23.1",
|
||||
@@ -3385,7 +3395,7 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "http_proto"
|
||||
version = "0.16.24"
|
||||
version = "0.16.25"
|
||||
dependencies = [
|
||||
"common",
|
||||
"compact_str",
|
||||
@@ -3872,7 +3882,7 @@ checksum = "65b27460c2c92b037f3f94c538ed9a3342f3fdf923606781629ccb35f82d042a"
|
||||
|
||||
[[package]]
|
||||
name = "imap"
|
||||
version = "0.16.24"
|
||||
version = "0.16.25"
|
||||
dependencies = [
|
||||
"ahash",
|
||||
"common",
|
||||
@@ -3897,7 +3907,7 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "imap_proto"
|
||||
version = "0.16.24"
|
||||
version = "0.16.25"
|
||||
dependencies = [
|
||||
"ahash",
|
||||
"base64 0.23.1",
|
||||
@@ -3912,7 +3922,7 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "inbuxa"
|
||||
version = "0.16.24"
|
||||
version = "0.16.25"
|
||||
dependencies = [
|
||||
"common",
|
||||
"coordinator",
|
||||
@@ -3920,7 +3930,7 @@ dependencies = [
|
||||
"directory",
|
||||
"email",
|
||||
"groupware",
|
||||
"http 0.16.24",
|
||||
"http 0.16.25",
|
||||
"http_proto",
|
||||
"imap",
|
||||
"jmap",
|
||||
@@ -3947,9 +3957,14 @@ name = "inbuxa-features"
|
||||
version = "0.16.22"
|
||||
dependencies = [
|
||||
"ahash",
|
||||
"aho-corasick",
|
||||
"base64 0.23.1",
|
||||
"flate2",
|
||||
"jmap_proto",
|
||||
"mail-builder 1.0.0",
|
||||
"mail-parser",
|
||||
"quick-xml 0.41.0",
|
||||
"regex",
|
||||
"registry",
|
||||
"serde",
|
||||
"serde_json",
|
||||
@@ -3961,6 +3976,7 @@ dependencies = [
|
||||
"types",
|
||||
"utils",
|
||||
"xxhash-rust",
|
||||
"zip",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
@@ -4203,7 +4219,7 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "jmap"
|
||||
version = "0.16.24"
|
||||
version = "0.16.25"
|
||||
dependencies = [
|
||||
"async-stream",
|
||||
"base64 0.23.1",
|
||||
@@ -4252,14 +4268,14 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "jmap-client"
|
||||
version = "0.4.2"
|
||||
version = "0.4.3"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "4deab22e057d24e32122f0fc6e2d667a124fdd6a0d8ef3ed4f8a89923c11084f"
|
||||
checksum = "f5b5bc66252cc8e779ef1238f40ab93971d54ad00b5d7afb5c1d447a9647ed62"
|
||||
dependencies = [
|
||||
"ahash",
|
||||
"async-stream",
|
||||
"base64 0.22.1",
|
||||
"chrono",
|
||||
"encodify",
|
||||
"futures-util",
|
||||
"maybe-async",
|
||||
"parking_lot",
|
||||
@@ -4286,7 +4302,7 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "jmap_proto"
|
||||
version = "0.16.24"
|
||||
version = "0.16.25"
|
||||
dependencies = [
|
||||
"ahash",
|
||||
"calcard",
|
||||
@@ -4496,9 +4512,9 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "lazy_static"
|
||||
version = "1.5.0"
|
||||
version = "1.5.1"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "bbd2bcb4c963f2ddae06a2efc7e9f3591312473c50c6685e1f298068316e66fe"
|
||||
checksum = "20870f649af7073d53e38067b2a84312175d56ea15217e1b15bc83506ec50afb"
|
||||
dependencies = [
|
||||
"spin 0.9.9",
|
||||
]
|
||||
@@ -4792,7 +4808,7 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "managesieve"
|
||||
version = "0.16.24"
|
||||
version = "0.16.25"
|
||||
dependencies = [
|
||||
"common",
|
||||
"compact_str",
|
||||
@@ -4927,7 +4943,7 @@ checksum = "c797b9d6bb23aab2fc369c65f871be49214f5c759af65bde26ffaaa2b646b492"
|
||||
|
||||
[[package]]
|
||||
name = "migration"
|
||||
version = "0.16.24"
|
||||
version = "0.16.25"
|
||||
dependencies = [
|
||||
"common",
|
||||
"email",
|
||||
@@ -5177,7 +5193,7 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "nlp"
|
||||
version = "0.16.24"
|
||||
version = "0.16.25"
|
||||
dependencies = [
|
||||
"ahash",
|
||||
"hashify",
|
||||
@@ -5497,8 +5513,8 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "opentelemetry"
|
||||
version = "0.32.0"
|
||||
source = "git+https://github.com/stalwartlabs/opentelemetry-rust#80a14a3b6846f62f85506d68d2600c948fccc9d2"
|
||||
version = "0.33.0"
|
||||
source = "git+https://github.com/stalwartlabs/opentelemetry-rust#ae66e97b140f70e477ab710686aafce665cc2f8b"
|
||||
dependencies = [
|
||||
"futures-core",
|
||||
"futures-sink",
|
||||
@@ -5510,8 +5526,8 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "opentelemetry-http"
|
||||
version = "0.32.0"
|
||||
source = "git+https://github.com/stalwartlabs/opentelemetry-rust#80a14a3b6846f62f85506d68d2600c948fccc9d2"
|
||||
version = "0.33.0"
|
||||
source = "git+https://github.com/stalwartlabs/opentelemetry-rust#ae66e97b140f70e477ab710686aafce665cc2f8b"
|
||||
dependencies = [
|
||||
"async-trait",
|
||||
"bytes",
|
||||
@@ -5522,8 +5538,8 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "opentelemetry-otlp"
|
||||
version = "0.32.0"
|
||||
source = "git+https://github.com/stalwartlabs/opentelemetry-rust#80a14a3b6846f62f85506d68d2600c948fccc9d2"
|
||||
version = "0.33.0"
|
||||
source = "git+https://github.com/stalwartlabs/opentelemetry-rust#ae66e97b140f70e477ab710686aafce665cc2f8b"
|
||||
dependencies = [
|
||||
"http 1.5.0",
|
||||
"httpdate",
|
||||
@@ -5541,8 +5557,8 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "opentelemetry-proto"
|
||||
version = "0.32.0"
|
||||
source = "git+https://github.com/stalwartlabs/opentelemetry-rust#80a14a3b6846f62f85506d68d2600c948fccc9d2"
|
||||
version = "0.33.0"
|
||||
source = "git+https://github.com/stalwartlabs/opentelemetry-rust#ae66e97b140f70e477ab710686aafce665cc2f8b"
|
||||
dependencies = [
|
||||
"opentelemetry",
|
||||
"opentelemetry_sdk",
|
||||
@@ -5553,13 +5569,13 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "opentelemetry-semantic-conventions"
|
||||
version = "0.32.1"
|
||||
source = "git+https://github.com/stalwartlabs/opentelemetry-rust#80a14a3b6846f62f85506d68d2600c948fccc9d2"
|
||||
version = "0.33.0"
|
||||
source = "git+https://github.com/stalwartlabs/opentelemetry-rust#ae66e97b140f70e477ab710686aafce665cc2f8b"
|
||||
|
||||
[[package]]
|
||||
name = "opentelemetry_sdk"
|
||||
version = "0.32.1"
|
||||
source = "git+https://github.com/stalwartlabs/opentelemetry-rust#80a14a3b6846f62f85506d68d2600c948fccc9d2"
|
||||
version = "0.33.0"
|
||||
source = "git+https://github.com/stalwartlabs/opentelemetry-rust#ae66e97b140f70e477ab710686aafce665cc2f8b"
|
||||
dependencies = [
|
||||
"futures-channel",
|
||||
"futures-executor",
|
||||
@@ -6009,7 +6025,7 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "pop3"
|
||||
version = "0.16.24"
|
||||
version = "0.16.25"
|
||||
dependencies = [
|
||||
"common",
|
||||
"directory",
|
||||
@@ -6379,9 +6395,9 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "quinn-proto"
|
||||
version = "0.11.18"
|
||||
version = "0.11.19"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "a9746dbde176634f4f2f1faf2404e30a31b2bc1e9cafb5329c95d8177a18c9fc"
|
||||
checksum = "0e750cca55fe4f0439a15d0bb529da9651e79993e8e72c61a899a36d462befbe"
|
||||
dependencies = [
|
||||
"aws-lc-rs",
|
||||
"bytes",
|
||||
@@ -6404,9 +6420,9 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "quinn-udp"
|
||||
version = "0.5.15"
|
||||
version = "0.5.16"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "35a133f956daabe89a61a685c2649f13d82d5aa4bd5d12d1277e1072a21c0694"
|
||||
checksum = "af66907df18639dcf4db56ca65490cabc4b27a97dbadd96f2926cca73298f016"
|
||||
dependencies = [
|
||||
"cfg_aliases",
|
||||
"libc",
|
||||
@@ -6829,7 +6845,7 @@ checksum = "d6f6ff9a378485b298a5286656da665ba74413d36db0979633275d2e708145d4"
|
||||
|
||||
[[package]]
|
||||
name = "registry"
|
||||
version = "0.16.24"
|
||||
version = "0.16.25"
|
||||
dependencies = [
|
||||
"ahash",
|
||||
"hashify",
|
||||
@@ -7386,7 +7402,7 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "scim"
|
||||
version = "0.16.24"
|
||||
version = "0.16.25"
|
||||
dependencies = [
|
||||
"ahash",
|
||||
"base64 0.23.1",
|
||||
@@ -7412,7 +7428,7 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "scim-proto"
|
||||
version = "0.16.24"
|
||||
version = "0.16.25"
|
||||
dependencies = [
|
||||
"hashify",
|
||||
"serde",
|
||||
@@ -7666,9 +7682,9 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "serde_with"
|
||||
version = "3.23.0"
|
||||
version = "3.24.0"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "935177bb8c0cd8ca1a4e6d1a2ac8988bea69cab4f9d3a31311e012ad27868ea4"
|
||||
checksum = "df9adc193c780ef8f159aee8b61e2d5801aaa555e6eb0947fe45530ec506296f"
|
||||
dependencies = [
|
||||
"base64 0.23.1",
|
||||
"bs58",
|
||||
@@ -7687,9 +7703,9 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "serde_with_macros"
|
||||
version = "3.23.0"
|
||||
version = "3.24.0"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "1d607aa01a3cb0ad757d6fd216136910db3c97b102fe686585689615a02dbcdc"
|
||||
checksum = "3e17bbc68e28663bbbb90df47e058aa7eda4fb445b89fe70457bb94fbccf6e49"
|
||||
dependencies = [
|
||||
"darling 0.24.1",
|
||||
"proc-macro2",
|
||||
@@ -7747,7 +7763,7 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "services"
|
||||
version = "0.16.24"
|
||||
version = "0.16.25"
|
||||
dependencies = [
|
||||
"aes-gcm 0.11.1",
|
||||
"aho-corasick",
|
||||
@@ -7756,11 +7772,13 @@ dependencies = [
|
||||
"common",
|
||||
"dns-update",
|
||||
"email",
|
||||
"futures",
|
||||
"groupware",
|
||||
"hkdf 0.13.0",
|
||||
"inbuxa-features",
|
||||
"jmap-tools",
|
||||
"jmap_proto",
|
||||
"mail-auth",
|
||||
"mail-builder 1.0.0",
|
||||
"mail-parser",
|
||||
"memory-stats",
|
||||
@@ -8060,7 +8078,7 @@ checksum = "f9395f0f0eee849a9b707b2f06bb92a6a422090e2123bb2ef8e87a0e61892a8e"
|
||||
|
||||
[[package]]
|
||||
name = "smtp"
|
||||
version = "0.16.24"
|
||||
version = "0.16.25"
|
||||
dependencies = [
|
||||
"ahash",
|
||||
"base64 0.23.1",
|
||||
@@ -8099,9 +8117,9 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "smtp-proto"
|
||||
version = "0.2.4"
|
||||
version = "0.2.5"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "707104487221ff447b5b796b5049e5c09cf52ff9fc1c8abba4695b89ac4b0f37"
|
||||
checksum = "142a5a642c6bd7ffd7e1b525ad6a6e9921ccef992dba4663974f8519209a00c1"
|
||||
dependencies = [
|
||||
"memchr",
|
||||
"rkyv",
|
||||
@@ -8151,7 +8169,7 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "spam-filter"
|
||||
version = "0.16.24"
|
||||
version = "0.16.25"
|
||||
dependencies = [
|
||||
"common",
|
||||
"compact_str",
|
||||
@@ -8271,7 +8289,7 @@ checksum = "a2eb9349b6444b326872e140eb1cf5e7c522154d69e7a0ffb0fb81c06b37543f"
|
||||
|
||||
[[package]]
|
||||
name = "store"
|
||||
version = "0.16.24"
|
||||
version = "0.16.25"
|
||||
dependencies = [
|
||||
"ahash",
|
||||
"arc-swap",
|
||||
@@ -8531,7 +8549,7 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "tests"
|
||||
version = "0.16.24"
|
||||
version = "0.16.25"
|
||||
dependencies = [
|
||||
"ahash",
|
||||
"aws-lc-rs",
|
||||
@@ -8553,7 +8571,7 @@ dependencies = [
|
||||
"form_urlencoded",
|
||||
"futures",
|
||||
"groupware",
|
||||
"http 0.16.24",
|
||||
"http 0.16.25",
|
||||
"http_proto",
|
||||
"hyper",
|
||||
"hyper-util",
|
||||
@@ -8810,9 +8828,9 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "tokio-rustls"
|
||||
version = "0.26.5"
|
||||
version = "0.26.6"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "b0c85f2c3ef0b1cd58b36682f4b17aaa995f0e5db534d85692b4903abce21f67"
|
||||
checksum = "c9cc2678c2cdd569ef8215e2afd7954ada2ae20b4fdd2c5fe6139a3b02d105db"
|
||||
dependencies = [
|
||||
"rustls",
|
||||
"tokio",
|
||||
@@ -9146,7 +9164,7 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "trc"
|
||||
version = "0.16.24"
|
||||
version = "0.16.25"
|
||||
dependencies = [
|
||||
"ahash",
|
||||
"base64 0.23.1",
|
||||
@@ -9255,7 +9273,7 @@ checksum = "b6f5e870be6c3b371b77fe0ee0bafb859fa4964b4404c27de1d380043c4dda20"
|
||||
|
||||
[[package]]
|
||||
name = "types"
|
||||
version = "0.16.24"
|
||||
version = "0.16.25"
|
||||
dependencies = [
|
||||
"blake3",
|
||||
"compact_str",
|
||||
@@ -9424,7 +9442,7 @@ checksum = "b6c140620e7ffbb22c2dee59cafe6084a59b5ffc27a8859a5f0d494b5d52b6be"
|
||||
|
||||
[[package]]
|
||||
name = "utils"
|
||||
version = "0.16.24"
|
||||
version = "0.16.25"
|
||||
dependencies = [
|
||||
"ahash",
|
||||
"arcstr",
|
||||
@@ -10109,9 +10127,9 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "xxhash-rust"
|
||||
version = "0.8.18"
|
||||
version = "0.8.19"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "aee1b19627c7c60102ab80d3a9cbe18de90bfe03bfa6c3715447681f0e8c8af6"
|
||||
checksum = "550a2b930b62486a393c52d5c3b84bff264b28aa437ed64694d31e93b1757af7"
|
||||
|
||||
[[package]]
|
||||
name = "yasna"
|
||||
@@ -10136,9 +10154,9 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "yoke-derive"
|
||||
version = "0.8.3"
|
||||
version = "0.8.4"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "33811428bee40dbceb6d545e95754741d17a6aef9a4849f0fd62e2ba4f412a78"
|
||||
checksum = "ec8ebde2db3681e8c9980cc27822030e68752690ddfa9473e739aeb4dbde6d71"
|
||||
dependencies = [
|
||||
"proc-macro2",
|
||||
"quote",
|
||||
|
||||
+1
-1
@@ -4,7 +4,7 @@
|
||||
# *****************
|
||||
# Base image for planner & builder
|
||||
# *****************
|
||||
FROM --platform=$BUILDPLATFORM rust:slim-trixie AS base
|
||||
FROM --platform=$BUILDPLATFORM rust:1.98.1-slim-trixie AS base
|
||||
|
||||
ENV DEBIAN_FRONTEND="noninteractive" \
|
||||
BINSTALL_DISABLE_TELEMETRY=true \
|
||||
|
||||
@@ -8,6 +8,10 @@
|
||||
|
||||
---
|
||||
|
||||
> [!NOTE]
|
||||
> Development happens on [git.coffeylabs.org/inbuxa/inbuxa-server](https://git.coffeylabs.org/inbuxa/inbuxa-server); the copy on GitHub is a read-only mirror.
|
||||
> Report issues at **[git.coffeylabs.org/inbuxa/inbuxa-server/issues](https://git.coffeylabs.org/inbuxa/inbuxa-server/issues)**, and join discussions at **[community.coffeylabs.org](https://community.coffeylabs.org)**.
|
||||
|
||||
**inbuxa** is a mail and collaboration server: JMAP, IMAP, POP3, SMTP,
|
||||
CalDAV, CardDAV and WebDAV, in one Rust binary, with ihasmail as its web front
|
||||
end. It is a fork of [Stalwart](https://github.com/stalwartlabs/stalwart).
|
||||
|
||||
+2
-2
@@ -17,7 +17,7 @@ visible to everyone, including whoever would use it, before there is a fix.
|
||||
|
||||
Report it privately by email to:
|
||||
|
||||
**johnellisATlinuxDOTcom**
|
||||
**securityATcoffeylabsDOTorg**
|
||||
|
||||
Include as much as you can of:
|
||||
|
||||
@@ -36,7 +36,7 @@ to Stalwart Labs with credit to you, and you'll be told that has happened.
|
||||
This repository is the mail server. The web front ends have their own:
|
||||
|
||||
- [inbuxa-admin](https://git.coffeylabs.org/inbuxa/inbuxa-admin)
|
||||
- [ihasmail-inbuxa](https://git.coffeylabs.org/inbuxa/ihasmail-inbuxa)
|
||||
- [inbuxa-webmail](https://git.coffeylabs.org/inbuxa/inbuxa-webmail)
|
||||
|
||||
Upstream's own security documents are kept in `.github-upstream/` for
|
||||
reference. They describe Stalwart Labs' process, not this project's.
|
||||
@@ -1,6 +1,6 @@
|
||||
[package]
|
||||
name = "common"
|
||||
version = "0.16.24"
|
||||
version = "0.16.25"
|
||||
edition = "2024"
|
||||
build = "build.rs"
|
||||
|
||||
|
||||
@@ -1,5 +1,5 @@
|
||||
/*
|
||||
* SPDX-FileCopyrightText: 2026 Coffey Labs
|
||||
* SPDX-FileCopyrightText: 2026 Coffey Labs LLC
|
||||
*
|
||||
* SPDX-License-Identifier: AGPL-3.0-only
|
||||
*/
|
||||
@@ -445,6 +445,63 @@ impl Server {
|
||||
}
|
||||
}
|
||||
|
||||
/// MA-D0a: a message sent from an address that isn't the sender's own:
|
||||
/// a group's or a shared mailbox's. The message itself only says
|
||||
/// `From:` that address, so the audit log is where the person who sent
|
||||
/// it is named. A locked account's delegate's send is AL-9's record, not
|
||||
/// this one.
|
||||
pub async fn audit_send_as(
|
||||
&self,
|
||||
token: &AccessToken,
|
||||
submission_account_id: u32,
|
||||
submission_id: u32,
|
||||
address: &str,
|
||||
) {
|
||||
let Ok(Some(as_account_id)) = self.account_id_from_email(address, true).await else {
|
||||
return;
|
||||
};
|
||||
if as_account_id == token.account_id()
|
||||
|| token
|
||||
.delegation(as_account_id)
|
||||
.is_some_and(|delegation| delegation.kind.is_lock())
|
||||
{
|
||||
return;
|
||||
}
|
||||
let actor = self.audit_actor(token).await;
|
||||
let tenant_id = self
|
||||
.account(as_account_id)
|
||||
.await
|
||||
.ok()
|
||||
.and_then(|account| account.id_tenant);
|
||||
let details = if submission_account_id == as_account_id {
|
||||
format!("Sent as {address}")
|
||||
} else {
|
||||
format!(
|
||||
"Sent as {address}, from {}",
|
||||
self.audit_account_name(submission_account_id).await
|
||||
)
|
||||
};
|
||||
self.audit_note(Record {
|
||||
at: ms(),
|
||||
actor,
|
||||
via: token.origin().cloned(),
|
||||
remote_ip: None,
|
||||
action: Action::Create,
|
||||
target: Target {
|
||||
kind: "EmailSubmission".into(),
|
||||
id: Some(Id::from(submission_id).to_string()),
|
||||
name: Some(address.to_string()),
|
||||
account_id: Some(as_account_id),
|
||||
tenant_id,
|
||||
},
|
||||
changes: vec![],
|
||||
details: Some(details),
|
||||
reason: None,
|
||||
outcome: Outcome::success(),
|
||||
})
|
||||
.await;
|
||||
}
|
||||
|
||||
/// AU-7: removes entries past the retention period.
|
||||
pub async fn audit_purge(&self) -> trc::Result<usize> {
|
||||
let settings = log::settings(self.store()).await?;
|
||||
|
||||
@@ -36,6 +36,32 @@ use utils::map::bitmap::{Bitmap, BitmapItem};
|
||||
use xxhash_rust::xxh3;
|
||||
|
||||
impl Server {
|
||||
/// inbuxa: MA-C: whether people in `owner`'s tenant may share their mail
|
||||
/// (the server's switch, narrowed by the tenant's).
|
||||
pub async fn mail_sharing_allowed(&self, owner: u32) -> trc::Result<bool> {
|
||||
let tenant_id = self.account(owner).await.ok().and_then(|account| account.id_tenant);
|
||||
Ok(
|
||||
inbuxa_features::security::sharing_policy::effective_for(self.store(), tenant_id)
|
||||
.await
|
||||
.caused_by(trc::location!())?
|
||||
.mail_sharing,
|
||||
)
|
||||
}
|
||||
|
||||
/// inbuxa: MA-C: whether `owner`'s mail shares give access now. A locked
|
||||
/// account's or shared mailbox's grants are an administrator's and always
|
||||
/// do; anyone else's only while their tenant allows mail sharing.
|
||||
pub async fn mail_shares_honored(&self, owner: u32) -> trc::Result<bool> {
|
||||
if inbuxa_features::lock::get(self.store(), owner)
|
||||
.await
|
||||
.caused_by(trc::location!())?
|
||||
.is_some()
|
||||
{
|
||||
return Ok(true);
|
||||
}
|
||||
self.mail_sharing_allowed(owner).await
|
||||
}
|
||||
|
||||
async fn build_access_token(
|
||||
&self,
|
||||
account: Account,
|
||||
@@ -46,19 +72,22 @@ impl Server {
|
||||
// inbuxa: AL-2, AL-5: whether this account is locked, and which
|
||||
// locked accounts are handed to it. The token is their cache: every
|
||||
// change to a lock invalidates the tokens it touches.
|
||||
let locked = inbuxa_features::lock::get(self.store(), account_id)
|
||||
let lock_kind = inbuxa_features::lock::get(self.store(), account_id)
|
||||
.await
|
||||
.caused_by(trc::location!())?
|
||||
.is_some();
|
||||
.map(|lock| lock.kind);
|
||||
let locked = lock_kind.is_some();
|
||||
let shared_mailbox = lock_kind == Some(inbuxa_features::lock::Kind::SharedMailbox);
|
||||
let now_secs = now();
|
||||
let delegations: Box<[super::Delegation]> =
|
||||
inbuxa_features::lock::delegated_to(self.store(), account_id)
|
||||
.await
|
||||
.caused_by(trc::location!())?
|
||||
.into_iter()
|
||||
.filter(|(_, delegate)| delegate.is_current(now_secs))
|
||||
.map(|(locked_id, delegate)| super::Delegation {
|
||||
.filter(|(_, delegate, _)| delegate.is_current(now_secs))
|
||||
.map(|(locked_id, delegate, kind)| super::Delegation {
|
||||
account_id: locked_id,
|
||||
kind,
|
||||
access: delegate.access,
|
||||
send_as: delegate.send_as,
|
||||
until: delegate.until,
|
||||
@@ -97,6 +126,9 @@ impl Server {
|
||||
.map(|m| m.id() as u32)
|
||||
.collect::<TinyVec<[u32; 3]>>();
|
||||
let mut access_to: Vec<AccessTo> = Vec::new();
|
||||
// inbuxa: MA-C: whether an owner's mail shares are honored,
|
||||
// looked up once per owner
|
||||
let mut mail_shares_honored: Vec<(u32, bool)> = Vec::new();
|
||||
for grant_account_id in [account_id].into_iter().chain(member_of.iter().copied()) {
|
||||
for acl_item in self
|
||||
.store()
|
||||
@@ -117,6 +149,27 @@ impl Server {
|
||||
.caused_by(trc::location!()));
|
||||
}
|
||||
|
||||
// inbuxa: MA-C: a mail share from an account whose
|
||||
// tenant (or server) has mail sharing off gives
|
||||
// nothing while it is off. It stays stored, so it
|
||||
// comes back when sharing does. A lock's and a
|
||||
// shared mailbox's grants are an administrator's,
|
||||
// and always count.
|
||||
if collection == Collection::Mailbox {
|
||||
let owner = acl_item.to_account_id;
|
||||
let honored = match mail_shares_honored.iter().find(|(id, _)| *id == owner) {
|
||||
Some((_, honored)) => *honored,
|
||||
None => {
|
||||
let honored = self.mail_shares_honored(owner).await?;
|
||||
mail_shares_honored.push((owner, honored));
|
||||
honored
|
||||
}
|
||||
};
|
||||
if !honored {
|
||||
continue;
|
||||
}
|
||||
}
|
||||
|
||||
let mut collections: Bitmap<Collection> = Bitmap::new();
|
||||
if acl.contains(Acl::Read) {
|
||||
collections.insert(collection);
|
||||
@@ -247,6 +300,7 @@ impl Server {
|
||||
.map(ConcurrencyLimiter::new),
|
||||
obj_size: 0,
|
||||
locked,
|
||||
shared_mailbox,
|
||||
delegations: delegations.clone(),
|
||||
revision,
|
||||
revision_account,
|
||||
@@ -300,6 +354,7 @@ impl Server {
|
||||
.map(ConcurrencyLimiter::new),
|
||||
obj_size: 0,
|
||||
locked,
|
||||
shared_mailbox,
|
||||
delegations: delegations.clone(),
|
||||
revision,
|
||||
revision_account,
|
||||
@@ -553,6 +608,16 @@ impl AccessToken {
|
||||
|| self.inner.access_to.iter().any(|a| a.account_id == account_id)
|
||||
}
|
||||
|
||||
/// inbuxa: MA-D0: in the account only because it is a group this token
|
||||
/// belongs to. Such a member has the group's mailbox but may not share it
|
||||
/// on: who is in a group is an administrator's decision, and a share
|
||||
/// would let anyone in.
|
||||
pub fn is_group_member_only(&self, account_id: u32) -> bool {
|
||||
self.inner.account_id != account_id
|
||||
&& self.inner.member_of.contains(&account_id)
|
||||
&& !self.has_permission(Permission::Impersonate)
|
||||
}
|
||||
|
||||
pub fn is_account_id(&self, account_id: u32) -> bool {
|
||||
self.inner.account_id == account_id
|
||||
}
|
||||
@@ -648,6 +713,7 @@ impl AccessToken {
|
||||
credential_version: old_inner.credential_version,
|
||||
obj_size: old_inner.obj_size,
|
||||
locked: old_inner.locked,
|
||||
shared_mailbox: old_inner.shared_mailbox,
|
||||
delegations: old_inner.delegations.clone(),
|
||||
};
|
||||
|
||||
@@ -838,6 +904,18 @@ impl AccessToken {
|
||||
self.inner.locked
|
||||
}
|
||||
|
||||
/// inbuxa: MA-S: the account is a shared mailbox (a lock of that kind).
|
||||
pub fn is_shared_mailbox(&self) -> bool {
|
||||
self.inner.shared_mailbox
|
||||
}
|
||||
|
||||
/// inbuxa: MA-S: this account's delegation into `account_id` is to a
|
||||
/// shared mailbox, not a locked account.
|
||||
pub fn delegated_shared_mailbox(&self, account_id: u32) -> bool {
|
||||
self.delegation(account_id)
|
||||
.is_some_and(|d| d.kind == inbuxa_features::lock::Kind::SharedMailbox)
|
||||
}
|
||||
|
||||
/// inbuxa: AL-5: this account's delegation into a locked account, if it
|
||||
/// has one that hasn't ended.
|
||||
/// inbuxa: AL-6, AL-7: a delegate at organize or full, who may add to
|
||||
@@ -918,6 +996,7 @@ impl AccessToken {
|
||||
credential_version: Default::default(),
|
||||
obj_size: Default::default(),
|
||||
locked: false,
|
||||
shared_mailbox: false,
|
||||
delegations: Default::default(),
|
||||
}),
|
||||
}
|
||||
@@ -978,6 +1057,7 @@ impl AccessTokenInner {
|
||||
credential_version: Default::default(),
|
||||
obj_size: Default::default(),
|
||||
locked: false,
|
||||
shared_mailbox: false,
|
||||
delegations: Default::default(),
|
||||
}
|
||||
}
|
||||
|
||||
@@ -152,6 +152,8 @@ pub struct AccessTokenInner {
|
||||
pub(crate) obj_size: u64,
|
||||
// inbuxa: AL-2: the account is locked; it may not authenticate
|
||||
pub(crate) locked: bool,
|
||||
// inbuxa: MA-S: the lock is a shared mailbox
|
||||
pub(crate) shared_mailbox: bool,
|
||||
// inbuxa: AL-5: locked accounts handed to this one
|
||||
pub(crate) delegations: Box<[Delegation]>,
|
||||
}
|
||||
@@ -165,6 +167,8 @@ pub struct Delegation {
|
||||
pub send_as: bool,
|
||||
/// Seconds since the epoch.
|
||||
pub until: Option<u64>,
|
||||
/// MA-S: a locked account, or a shared mailbox.
|
||||
pub kind: inbuxa_features::lock::Kind,
|
||||
}
|
||||
|
||||
#[derive(Debug, Default, Hash, Clone)]
|
||||
|
||||
@@ -111,6 +111,9 @@ impl Server {
|
||||
Permission::SysLegalHoldCreate,
|
||||
Permission::SysLegalHoldUpdate,
|
||||
Permission::SysLegalHoldExport,
|
||||
// inbuxa: DL-20: the lists and the check are the server's
|
||||
Permission::SysDeliverabilityUpdate,
|
||||
Permission::SysDeliverabilityCheck,
|
||||
] {
|
||||
permissions.disabled.set(permission as usize);
|
||||
}
|
||||
@@ -165,6 +168,14 @@ impl AccessToken {
|
||||
mut requested_permissions: Permissions,
|
||||
) -> Result<(), Vec<Permission>> {
|
||||
requested_permissions.difference(self.permissions_bits());
|
||||
// inbuxa: journaling, JR-18: whoever sets up journals may give
|
||||
// others (or, through a role, themselves) the reading of them,
|
||||
// which administrators don't hold by default; the role change is
|
||||
// in the audit log
|
||||
if self.has_permission(Permission::SysJournalUpdate) {
|
||||
requested_permissions.clear(Permission::SysJournalSearch as usize);
|
||||
requested_permissions.clear(Permission::SysJournalExport as usize);
|
||||
}
|
||||
if requested_permissions.is_empty() {
|
||||
Ok(())
|
||||
} else {
|
||||
@@ -296,6 +307,37 @@ impl Default for DefaultPermissions {
|
||||
default.superuser.push(permission);
|
||||
default.tenant.push(permission);
|
||||
}
|
||||
// inbuxa: deliverability spec, DL-20: a tenant administrator
|
||||
// reads its own domains' findings; the lists and the check
|
||||
// itself are the server's
|
||||
Permission::SysDeliverabilityGet => {
|
||||
default.superuser.push(permission);
|
||||
default.tenant.push(permission);
|
||||
}
|
||||
Permission::SysDeliverabilityUpdate | Permission::SysDeliverabilityCheck => {
|
||||
default.superuser.push(permission);
|
||||
}
|
||||
// inbuxa: DLP and mail flow rules, and held mail, are the
|
||||
// server's: never a tenant's (dlp-and-mail-flow-rules spec,
|
||||
// settled answer 3)
|
||||
Permission::SysMailRuleGet
|
||||
| Permission::SysMailRuleUpdate
|
||||
| Permission::SysDlpPolicyGet
|
||||
| Permission::SysDlpPolicyUpdate
|
||||
| Permission::SysDlpReviewGet
|
||||
| Permission::SysDlpReviewUpdate
|
||||
// inbuxa: every security check is server-wide (security
|
||||
// to-do list spec)
|
||||
| Permission::SysSecurityAccept => {
|
||||
default.superuser.push(permission);
|
||||
}
|
||||
// inbuxa: journals are the server's; administrators set them
|
||||
// up but read what's journaled only if granted it
|
||||
// (journaling spec, JR-18, settled answer 5)
|
||||
Permission::SysJournalGet | Permission::SysJournalUpdate => {
|
||||
default.superuser.push(permission);
|
||||
}
|
||||
Permission::SysJournalSearch | Permission::SysJournalExport => {}
|
||||
// inbuxa: AL-12: tenant administrators lock and delegate
|
||||
// within their tenant
|
||||
Permission::SysAccountLockGet
|
||||
|
||||
@@ -72,6 +72,10 @@ pub struct Http {
|
||||
pub cors_origins: Vec<hyper::header::HeaderValue>,
|
||||
pub use_forwarded: bool,
|
||||
pub redirect_root: Option<String>,
|
||||
/// inbuxa: HTTP Basic accepted on every endpoint, not only DAV (contract
|
||||
/// C-23). True in bootstrap and recovery mode, or with
|
||||
/// `INBUXA_HTTP_BASIC_AUTH=all`.
|
||||
pub basic_auth_everywhere: bool,
|
||||
}
|
||||
|
||||
#[derive(Clone)]
|
||||
@@ -453,6 +457,35 @@ impl Http {
|
||||
.collect()
|
||||
};
|
||||
|
||||
// inbuxa: outside DAV, HTTP sign-in is a token unless the operator
|
||||
// says otherwise (contract C-23). The integration suites sign in with
|
||||
// passwords over JMAP and the API, so test builds accept Basic
|
||||
// everywhere.
|
||||
#[cfg(feature = "test_mode")]
|
||||
let basic_auth_everywhere = true;
|
||||
|
||||
#[cfg(not(feature = "test_mode"))]
|
||||
let basic_auth_everywhere = bp.registry.is_recovery_mode()
|
||||
|| bp.registry.is_bootstrap_mode()
|
||||
|| match types::branding::env_var("HTTP_BASIC_AUTH") {
|
||||
Ok(value) if value.trim().eq_ignore_ascii_case("all") => true,
|
||||
Ok(value)
|
||||
if value.trim().is_empty() || value.trim().eq_ignore_ascii_case("dav") =>
|
||||
{
|
||||
false
|
||||
}
|
||||
Ok(value) => {
|
||||
bp.build_warning(
|
||||
ObjectType::Http.singleton(),
|
||||
format!(
|
||||
"INBUXA_HTTP_BASIC_AUTH is {value:?}; expected \"dav\" or \"all\". Basic authentication stays on DAV only."
|
||||
),
|
||||
);
|
||||
false
|
||||
}
|
||||
Err(_) => false,
|
||||
};
|
||||
|
||||
if use_permissive_cors {
|
||||
http_headers.push((
|
||||
hyper::header::ACCESS_CONTROL_ALLOW_ORIGIN,
|
||||
@@ -512,6 +545,7 @@ impl Http {
|
||||
cors_origins,
|
||||
use_forwarded: http.use_x_forwarded,
|
||||
redirect_root: http.redirect_root,
|
||||
basic_auth_everywhere,
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@@ -1,5 +1,5 @@
|
||||
/*
|
||||
* SPDX-FileCopyrightText: 2026 Coffey Labs
|
||||
* SPDX-FileCopyrightText: 2026 Coffey Labs LLC
|
||||
*
|
||||
* SPDX-License-Identifier: AGPL-3.0-only
|
||||
*/
|
||||
|
||||
@@ -1,5 +1,5 @@
|
||||
/*
|
||||
* SPDX-FileCopyrightText: 2026 Coffey Labs
|
||||
* SPDX-FileCopyrightText: 2026 Coffey Labs LLC
|
||||
*
|
||||
* SPDX-License-Identifier: AGPL-3.0-only
|
||||
*/
|
||||
|
||||
@@ -1,5 +1,5 @@
|
||||
/*
|
||||
* SPDX-FileCopyrightText: 2026 Coffey Labs
|
||||
* SPDX-FileCopyrightText: 2026 Coffey Labs LLC
|
||||
*
|
||||
* SPDX-License-Identifier: AGPL-3.0-only
|
||||
*/
|
||||
|
||||
@@ -88,6 +88,8 @@ pub enum BroadcastEvent {
|
||||
QueueRefresh,
|
||||
// inbuxa: AL-3: end an account's open sessions on every node
|
||||
EndSessions(u32),
|
||||
// inbuxa: deliverability spec, DL-15: every node checks itself now
|
||||
DeliverabilityCheck,
|
||||
}
|
||||
|
||||
#[derive(Debug, Clone, Copy)]
|
||||
|
||||
@@ -70,6 +70,7 @@ pub mod cache;
|
||||
pub mod audit; // inbuxa: the audit log (audit-hold-lock spec, AU)
|
||||
pub mod hold; // inbuxa: legal holds (audit-hold-lock spec, LH)
|
||||
pub mod privacy; // inbuxa: the personal-data catalog, evaluated
|
||||
pub mod reachability; // inbuxa: whether the outside world reaches each node's ports
|
||||
pub mod config;
|
||||
pub mod expr;
|
||||
pub mod i18n;
|
||||
@@ -129,6 +130,8 @@ pub const KV_LOCK_QUEUE_MESSAGE: u8 = 21;
|
||||
pub const KV_LOCK_TASK: u8 = 23;
|
||||
pub const KV_LOCK_DAV: u8 = 25;
|
||||
pub const KV_SIEVE_ID: u8 = 26;
|
||||
// inbuxa: far above upstream's prefixes, so a new one of theirs never collides
|
||||
pub const KV_PORT_REACHABILITY: u8 = 200;
|
||||
|
||||
#[derive(Clone)]
|
||||
pub struct Server {
|
||||
@@ -222,7 +225,7 @@ pub struct Caches {
|
||||
pub dns_ipv6: CacheWithTtl<Box<str>, RecordSet<Ipv6Addr>>,
|
||||
pub dns_tlsa: CacheWithTtl<Box<str>, Arc<Tlsa>>,
|
||||
pub dns_mta_sts: CacheWithTtl<Box<str>, Arc<Policy>>,
|
||||
pub dns_rbl: CacheWithTtl<Box<str>, Option<Arc<IpResolver>>>,
|
||||
pub dns_rbl: CacheWithTtl<Box<str>, Option<Arc<[IpResolver]>>>,
|
||||
|
||||
pub negative_cache_ttl: Duration,
|
||||
}
|
||||
|
||||
@@ -227,10 +227,22 @@ impl WebApplicationManager {
|
||||
let cached = if force_refresh {
|
||||
None
|
||||
} else {
|
||||
server
|
||||
match server
|
||||
.blob_store()
|
||||
.get_blob(self.blob_key.as_slice(), 0..usize::MAX)
|
||||
.await?
|
||||
.await
|
||||
{
|
||||
Ok(cached) => cached,
|
||||
Err(err) => {
|
||||
trc::event!(
|
||||
Resource(trc::ResourceEvent::Error),
|
||||
Reason = err,
|
||||
Url = self.url.clone(),
|
||||
Details = "Failed to read cached application bundle, downloading it again"
|
||||
);
|
||||
None
|
||||
}
|
||||
}
|
||||
};
|
||||
let is_cached = cached.is_some();
|
||||
let bundle = match cached {
|
||||
|
||||
@@ -1,5 +1,5 @@
|
||||
/*
|
||||
* SPDX-FileCopyrightText: 2026 Coffey Labs
|
||||
* SPDX-FileCopyrightText: 2026 Coffey Labs LLC
|
||||
*
|
||||
* SPDX-License-Identifier: AGPL-3.0-only
|
||||
*/
|
||||
@@ -65,6 +65,14 @@ const OFFICER: &[Permission] = &[
|
||||
Permission::SysLegalHoldUpdate,
|
||||
Permission::SysLegalHoldExport,
|
||||
Permission::SysAccountLockGet,
|
||||
// dlp-and-mail-flow-rules spec, §2.8: see DLP rules, review held mail
|
||||
Permission::SysDlpPolicyGet,
|
||||
Permission::SysDlpReviewGet,
|
||||
Permission::SysDlpReviewUpdate,
|
||||
// journaling spec, JR-18: see journals, search and export them
|
||||
Permission::SysJournalGet,
|
||||
Permission::SysJournalSearch,
|
||||
Permission::SysJournalExport,
|
||||
];
|
||||
|
||||
/// What a tenant's officer holds besides [`READS`].
|
||||
@@ -113,6 +121,11 @@ fn created_key(tenant: Option<Id>) -> ValueClass {
|
||||
})
|
||||
}
|
||||
|
||||
/// The server-level Compliance Officer role the server made, if it has.
|
||||
pub async fn server_role(data: &Store) -> trc::Result<Option<Id>> {
|
||||
recorded(data, None).await
|
||||
}
|
||||
|
||||
async fn recorded(data: &Store, tenant: Option<Id>) -> trc::Result<Option<Id>> {
|
||||
Ok(data
|
||||
.get_value::<u64>(ValueKey::from(created_key(tenant)))
|
||||
@@ -172,13 +185,19 @@ pub async fn ensure_compliance_roles(registry: &RegistryStore, data: &Store) ->
|
||||
|
||||
/// A new tenant gets its Compliance Officer role.
|
||||
pub async fn tenant_created(registry: &RegistryStore, data: &Store, tenant: Id) -> trc::Result<()> {
|
||||
create_once(registry, data, Some(tenant), tenant_role(tenant)).await.map(|_| ())
|
||||
create_once(registry, data, Some(tenant), tenant_role(tenant))
|
||||
.await
|
||||
.map(|_| ())
|
||||
}
|
||||
|
||||
/// Before a tenant is deleted: removes its Compliance Officer role if nobody
|
||||
/// holds it, so the role doesn't block the delete. Returns whether it did,
|
||||
/// so a delete refused for another reason can put it back.
|
||||
pub async fn tenant_deleting(registry: &RegistryStore, data: &Store, tenant: Id) -> trc::Result<bool> {
|
||||
pub async fn tenant_deleting(
|
||||
registry: &RegistryStore,
|
||||
data: &Store,
|
||||
tenant: Id,
|
||||
) -> trc::Result<bool> {
|
||||
let Some(role) = recorded(data, Some(tenant)).await? else {
|
||||
return Ok(false);
|
||||
};
|
||||
@@ -220,7 +239,9 @@ mod tests {
|
||||
// Beyond what any user holds for their own account
|
||||
for permission in all.into_iter().filter(|p| !user.contains(p)) {
|
||||
let name = permission.as_str();
|
||||
let holds = name.starts_with("sysLegalHold");
|
||||
// Placing holds and reviewing held mail are the officer's
|
||||
// job, not settings (settled answers 2 and 4)
|
||||
let holds = name.starts_with("sysLegalHold") || name.starts_with("sysDlpReview");
|
||||
assert!(
|
||||
!(name.ends_with("Update") && !holds)
|
||||
&& !(name.ends_with("Create") && !holds)
|
||||
@@ -249,7 +270,11 @@ mod tests {
|
||||
assert!(officer.contains(&hold));
|
||||
assert!(!tenant.contains(&hold));
|
||||
}
|
||||
for both in [Permission::SysComplianceGet, Permission::SysAuditGet, Permission::SysAccountGet] {
|
||||
for both in [
|
||||
Permission::SysComplianceGet,
|
||||
Permission::SysAuditGet,
|
||||
Permission::SysAccountGet,
|
||||
] {
|
||||
assert!(officer.contains(&both) && tenant.contains(&both));
|
||||
}
|
||||
assert!(!officer.contains(&Permission::SysAuditSettingsUpdate));
|
||||
@@ -257,9 +282,15 @@ mod tests {
|
||||
|
||||
#[test]
|
||||
fn records_are_per_place() {
|
||||
let ValueClass::Any(server) = created_key(None) else { panic!() };
|
||||
let ValueClass::Any(a) = created_key(Some(Id::from(1u64))) else { panic!() };
|
||||
let ValueClass::Any(b) = created_key(Some(Id::from(2u64))) else { panic!() };
|
||||
let ValueClass::Any(server) = created_key(None) else {
|
||||
panic!()
|
||||
};
|
||||
let ValueClass::Any(a) = created_key(Some(Id::from(1u64))) else {
|
||||
panic!()
|
||||
};
|
||||
let ValueClass::Any(b) = created_key(Some(Id::from(2u64))) else {
|
||||
panic!()
|
||||
};
|
||||
assert_eq!(server.key, b"Pc");
|
||||
assert_ne!(a.key, b.key);
|
||||
assert!(a.key.starts_with(b"Pc"));
|
||||
|
||||
@@ -1,5 +1,5 @@
|
||||
/*
|
||||
* SPDX-FileCopyrightText: 2026 Coffey Labs
|
||||
* SPDX-FileCopyrightText: 2026 Coffey Labs LLC
|
||||
*
|
||||
* SPDX-License-Identifier: AGPL-3.0-only
|
||||
*/
|
||||
@@ -14,7 +14,7 @@
|
||||
//! application names another;
|
||||
//! - INBUXA Admin hosted elsewhere, as `inbuxa-admin`, when `INBUXA_ADMIN_URL`
|
||||
//! is set;
|
||||
//! - ihasmail-inbuxa, as the confidential client `ihasmail-inbuxa`, when
|
||||
//! - inbuxa-webmail, as the confidential client `ihasmail-inbuxa`, when
|
||||
//! `INBUXA_WEBMAIL_URL` and `INBUXA_WEBMAIL_CLIENT_SECRET` are set.
|
||||
//!
|
||||
//! inbuxa: the environment variables stand in for `x:FrontEnds` (C-4) until
|
||||
@@ -22,7 +22,7 @@
|
||||
//! it instead.
|
||||
//!
|
||||
//! A missing client is created. An existing one gains any redirect URI it
|
||||
//! lacks and, for ihasmail-inbuxa, the configured secret; nothing an operator
|
||||
//! lacks and, for inbuxa-webmail, the configured secret; nothing an operator
|
||||
//! added is removed.
|
||||
|
||||
use directory::core::secret::{hash_secret, verify_secret_hash};
|
||||
|
||||
@@ -1,5 +1,5 @@
|
||||
/*
|
||||
* SPDX-FileCopyrightText: 2026 Coffey Labs
|
||||
* SPDX-FileCopyrightText: 2026 Coffey Labs LLC
|
||||
*
|
||||
* SPDX-License-Identifier: AGPL-3.0-only
|
||||
*/
|
||||
@@ -31,7 +31,9 @@ use types::id::Id;
|
||||
/// Granted to the default administrator roles: "Explain this"
|
||||
/// (ai-explain spec, EX-4: superuser by default), the audit log, account
|
||||
/// locks and legal holds (audit-hold-lock spec, AU-9, AL-12, LH-13), and
|
||||
/// the data inventory (personal-data catalog spec).
|
||||
/// the data inventory (personal-data catalog spec), accepting security
|
||||
/// to-do items (security to-do list spec), and the deliverability check
|
||||
/// (deliverability spec).
|
||||
const ADMIN_GRANTS: &[Permission] = &[
|
||||
Permission::SysAiExplain,
|
||||
Permission::SysAuditGet,
|
||||
@@ -46,11 +48,37 @@ const ADMIN_GRANTS: &[Permission] = &[
|
||||
Permission::SysLegalHoldUpdate,
|
||||
Permission::SysLegalHoldExport,
|
||||
Permission::SysComplianceGet,
|
||||
Permission::SysMailRuleGet,
|
||||
Permission::SysMailRuleUpdate,
|
||||
Permission::SysDlpPolicyGet,
|
||||
Permission::SysDlpPolicyUpdate,
|
||||
Permission::SysDlpReviewGet,
|
||||
Permission::SysDlpReviewUpdate,
|
||||
Permission::SysJournalGet,
|
||||
Permission::SysJournalUpdate,
|
||||
Permission::SysSecurityAccept,
|
||||
Permission::SysDeliverabilityGet,
|
||||
Permission::SysDeliverabilityUpdate,
|
||||
Permission::SysDeliverabilityCheck,
|
||||
];
|
||||
|
||||
/// Granted to the server-level Compliance Officer role once it exists:
|
||||
/// seeing DLP rules and reviewing held mail (dlp-and-mail-flow-rules spec,
|
||||
/// §2.8, settled answer 4). A new install's role has them from the start.
|
||||
const OFFICER_GRANTS: &[Permission] = &[
|
||||
Permission::SysDlpPolicyGet,
|
||||
Permission::SysDlpReviewGet,
|
||||
Permission::SysDlpReviewUpdate,
|
||||
// journaling spec, JR-18: see journals, search and export them
|
||||
Permission::SysJournalGet,
|
||||
Permission::SysJournalSearch,
|
||||
Permission::SysJournalExport,
|
||||
];
|
||||
|
||||
/// Granted to the default tenant administrator roles: reading and exporting
|
||||
/// the tenant's audit log (AU-9), locking and delegating its accounts
|
||||
/// (AL-12), and the tenant's slice of the data inventory.
|
||||
/// (AL-12), the tenant's slice of the data inventory, and its own domains'
|
||||
/// deliverability findings (DL-20).
|
||||
const TENANT_GRANTS: &[Permission] = &[
|
||||
Permission::SysAuditGet,
|
||||
Permission::SysAuditExport,
|
||||
@@ -59,19 +87,23 @@ const TENANT_GRANTS: &[Permission] = &[
|
||||
Permission::SysAccountLockUpdate,
|
||||
Permission::SysAccountLockDestroy,
|
||||
Permission::SysComplianceGet,
|
||||
Permission::SysDeliverabilityGet,
|
||||
];
|
||||
|
||||
#[derive(Clone, Copy, PartialEq, Eq)]
|
||||
enum Audience {
|
||||
Admin,
|
||||
Tenant,
|
||||
Officer,
|
||||
}
|
||||
|
||||
fn granted_key(permission: Permission, audience: Audience) -> ValueClass {
|
||||
let mut key = b"Pg".to_vec();
|
||||
// Admin grants keep the key they were first recorded under
|
||||
if audience == Audience::Tenant {
|
||||
key.extend_from_slice(b"tenant:");
|
||||
match audience {
|
||||
Audience::Admin => {}
|
||||
Audience::Tenant => key.extend_from_slice(b"tenant:"),
|
||||
Audience::Officer => key.extend_from_slice(b"officer:"),
|
||||
}
|
||||
key.extend_from_slice(permission.as_str().as_bytes());
|
||||
ValueClass::Any(AnyClass {
|
||||
@@ -82,7 +114,8 @@ fn granted_key(permission: Permission, audience: Audience) -> ValueClass {
|
||||
|
||||
pub(crate) async fn grant_new_admin_permissions(bp: &mut Bootstrap) -> trc::Result<()> {
|
||||
grant(bp, Audience::Admin, ADMIN_GRANTS).await?;
|
||||
grant(bp, Audience::Tenant, TENANT_GRANTS).await
|
||||
grant(bp, Audience::Tenant, TENANT_GRANTS).await?;
|
||||
grant(bp, Audience::Officer, OFFICER_GRANTS).await
|
||||
}
|
||||
|
||||
async fn grant(bp: &mut Bootstrap, audience: Audience, grants: &[Permission]) -> trc::Result<()> {
|
||||
@@ -101,10 +134,17 @@ async fn grant(bp: &mut Bootstrap, audience: Audience, grants: &[Permission]) ->
|
||||
if pending.is_empty() {
|
||||
return Ok(());
|
||||
}
|
||||
// The officer role is the one the server made, if it has made it yet: a
|
||||
// new install makes it after this, with the permissions already in it
|
||||
let admin_roles: Vec<Id> = if audience == Audience::Officer {
|
||||
super::compliance_roles::server_role(&bp.data_store)
|
||||
.await?
|
||||
.into_iter()
|
||||
.collect()
|
||||
} else {
|
||||
// An administrator's default roles include the plain User role, which
|
||||
// every user also holds; only roles that are the audience's alone get it
|
||||
let admin_roles: Vec<Id> = bp
|
||||
.registry
|
||||
bp.registry
|
||||
.object::<Authentication>(Id::singleton())
|
||||
.await?
|
||||
.map(|auth| {
|
||||
@@ -118,7 +158,7 @@ async fn grant(bp: &mut Bootstrap, audience: Audience, grants: &[Permission]) ->
|
||||
]
|
||||
.concat(),
|
||||
),
|
||||
Audience::Tenant => (
|
||||
Audience::Tenant | Audience::Officer => (
|
||||
auth.default_tenant_role_ids.as_slice(),
|
||||
[
|
||||
auth.default_user_role_ids.as_slice(),
|
||||
@@ -133,7 +173,8 @@ async fn grant(bp: &mut Bootstrap, audience: Audience, grants: &[Permission]) ->
|
||||
.copied()
|
||||
.collect()
|
||||
})
|
||||
.unwrap_or_default();
|
||||
.unwrap_or_default()
|
||||
};
|
||||
// Fetched by id: the registry's listing doesn't reach stored roles
|
||||
for role_id in admin_roles {
|
||||
let Some(stored) = bp
|
||||
|
||||
@@ -1,5 +1,5 @@
|
||||
/*
|
||||
* SPDX-FileCopyrightText: 2026 Coffey Labs
|
||||
* SPDX-FileCopyrightText: 2026 Coffey Labs LLC
|
||||
*
|
||||
* SPDX-License-Identifier: AGPL-3.0-only
|
||||
*/
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
/*
|
||||
* SPDX-FileCopyrightText: 2020 Stalwart Labs LLC <[email protected]>
|
||||
* SPDX-FileCopyrightText: 2026 Coffey Labs
|
||||
* SPDX-FileCopyrightText: 2026 Coffey Labs LLC
|
||||
*
|
||||
* SPDX-License-Identifier: AGPL-3.0-only OR LicenseRef-SEL
|
||||
*
|
||||
|
||||
@@ -11,7 +11,7 @@ use quick_xml::Reader;
|
||||
use quick_xml::XmlVersion;
|
||||
use quick_xml::events::Event;
|
||||
use registry::schema::{enums::ServiceProtocol, structs::Service};
|
||||
use std::fmt::Write;
|
||||
use std::{borrow::Cow, fmt::Write};
|
||||
use utils::map::vec_map::VecMap;
|
||||
|
||||
impl Server {
|
||||
@@ -20,32 +20,78 @@ impl Server {
|
||||
body: Option<Vec<u8>>,
|
||||
) -> trc::Result<Resource<Vec<u8>>> {
|
||||
// Obtain parameters
|
||||
let emailaddress = parse_autodiscover_request(body.as_deref().unwrap_or_default())
|
||||
.map_err(|err| {
|
||||
let request =
|
||||
parse_autodiscover_request(body.as_deref().unwrap_or_default()).map_err(|err| {
|
||||
trc::ResourceEvent::BadParameters
|
||||
.into_err()
|
||||
.details("Failed to parse autodiscover request")
|
||||
.ctx(trc::Key::Reason, err)
|
||||
})?;
|
||||
// inbuxa: legacy-protocols LP-7, LP-14a
|
||||
let legacy_off = match emailaddress.rsplit_once('@') {
|
||||
let legacy_off = match request.email.rsplit_once('@') {
|
||||
Some((_, domain)) => self.legacy_off_for(domain).await?,
|
||||
None => self.legacy_off_for("").await?,
|
||||
};
|
||||
|
||||
Ok(Resource::new(
|
||||
"application/xml; charset=utf-8",
|
||||
build_autodiscover_response(
|
||||
&emailaddress,
|
||||
let response = match request.response_schema {
|
||||
ResponseSchema::Outlook => build_autodiscover_response(
|
||||
&request.email,
|
||||
&self.core.network.server_name,
|
||||
&self.core.network.info.services,
|
||||
|protocol| legacy_off.service(protocol),
|
||||
)
|
||||
.into_bytes(),
|
||||
))
|
||||
ResponseSchema::Unsupported => PROVIDER_NOT_AVAILABLE_RESPONSE.as_bytes().to_vec(),
|
||||
};
|
||||
|
||||
Ok(Resource::new("application/xml; charset=utf-8", response))
|
||||
}
|
||||
}
|
||||
|
||||
const OUTLOOK_RESPONSE_SCHEMA: &str =
|
||||
"http://schemas.microsoft.com/exchange/autodiscover/outlook/responseschema/2006a";
|
||||
|
||||
const PROVIDER_NOT_AVAILABLE_RESPONSE: &str = concat!(
|
||||
"<?xml version=\"1.0\" encoding=\"UTF-8\"?>\n",
|
||||
"<Autodiscover xmlns=\"http://schemas.microsoft.com/exchange/autodiscover/responseschema/2006\">\n",
|
||||
"\t<Response>\n",
|
||||
"\t\t<Error>\n",
|
||||
"\t\t\t<ErrorCode>601</ErrorCode>\n",
|
||||
"\t\t\t<Message>Provider is not available</Message>\n",
|
||||
"\t\t\t<DebugData />\n",
|
||||
"\t\t</Error>\n",
|
||||
"\t</Response>\n",
|
||||
"</Autodiscover>\n",
|
||||
);
|
||||
|
||||
#[derive(Debug, Clone, Copy, PartialEq, Eq)]
|
||||
enum ResponseSchema {
|
||||
Outlook,
|
||||
Unsupported,
|
||||
}
|
||||
|
||||
impl ResponseSchema {
|
||||
fn parse(value: &str) -> Self {
|
||||
if value.trim().eq_ignore_ascii_case(OUTLOOK_RESPONSE_SCHEMA) {
|
||||
ResponseSchema::Outlook
|
||||
} else {
|
||||
ResponseSchema::Unsupported
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
#[derive(Debug, PartialEq, Eq)]
|
||||
struct AutodiscoverRequest {
|
||||
email: String,
|
||||
response_schema: ResponseSchema,
|
||||
}
|
||||
|
||||
#[derive(Clone, Copy)]
|
||||
enum RequestField {
|
||||
EmailAddress,
|
||||
ResponseSchema,
|
||||
}
|
||||
|
||||
fn build_autodiscover_response(
|
||||
emailaddress: &str,
|
||||
default_host: &str,
|
||||
@@ -124,7 +170,7 @@ fn build_autodiscover_response(
|
||||
config
|
||||
}
|
||||
|
||||
fn parse_autodiscover_request(bytes: &[u8]) -> Result<String, String> {
|
||||
fn parse_autodiscover_request(bytes: &[u8]) -> Result<AutodiscoverRequest, String> {
|
||||
if bytes.is_empty() {
|
||||
return Err("Empty request body".to_string());
|
||||
}
|
||||
@@ -132,8 +178,9 @@ fn parse_autodiscover_request(bytes: &[u8]) -> Result<String, String> {
|
||||
let mut reader = Reader::from_reader(bytes);
|
||||
reader.config_mut().trim_text(true);
|
||||
let mut buf = Vec::with_capacity(128);
|
||||
let mut value_buf = Vec::with_capacity(128);
|
||||
|
||||
'outer: for tag_name in ["Autodiscover", "Request", "EMailAddress"] {
|
||||
'outer: for tag_name in ["Autodiscover", "Request"] {
|
||||
loop {
|
||||
match reader.read_event_into(&mut buf) {
|
||||
Ok(Event::Start(e)) => {
|
||||
@@ -143,30 +190,6 @@ fn parse_autodiscover_request(bytes: &[u8]) -> Result<String, String> {
|
||||
.eq_ignore_ascii_case(found_tag_name.as_ref())
|
||||
{
|
||||
continue 'outer;
|
||||
} else if tag_name == "EMailAddress" {
|
||||
// Skip unsupported tags under Request, such as AcceptableResponseSchema
|
||||
let mut tag_count = 0;
|
||||
loop {
|
||||
match reader.read_event_into(&mut buf) {
|
||||
Ok(Event::End(_)) => {
|
||||
if tag_count == 0 {
|
||||
break;
|
||||
} else {
|
||||
tag_count -= 1;
|
||||
}
|
||||
}
|
||||
Ok(Event::Start(_)) => {
|
||||
tag_count += 1;
|
||||
}
|
||||
Ok(Event::Eof) => {
|
||||
return Err(format!(
|
||||
"Expected value, found unexpected EOF at position {}.",
|
||||
reader.buffer_position()
|
||||
));
|
||||
}
|
||||
_ => (),
|
||||
}
|
||||
}
|
||||
} else {
|
||||
return Err(format!(
|
||||
"Expected tag {}, found unexpected tag {} at position {}.",
|
||||
@@ -195,38 +218,172 @@ fn parse_autodiscover_request(bytes: &[u8]) -> Result<String, String> {
|
||||
}
|
||||
}
|
||||
|
||||
if let Ok(Event::Text(text)) = reader.read_event_into(&mut buf)
|
||||
&& let Ok(text) = text.xml_content(XmlVersion::Implicit1_0)
|
||||
&& text.contains('@')
|
||||
{
|
||||
return Ok(text.trim().to_lowercase());
|
||||
let mut email = None;
|
||||
let mut response_schema = ResponseSchema::Outlook;
|
||||
|
||||
loop {
|
||||
match reader.read_event_into(&mut buf) {
|
||||
Ok(Event::Start(e)) => {
|
||||
let local_name = e.local_name();
|
||||
let field = hashify::tiny_map_ignore_case!(local_name.as_ref(),
|
||||
b"EMailAddress" => RequestField::EmailAddress,
|
||||
b"AcceptableResponseSchema" => RequestField::ResponseSchema,
|
||||
);
|
||||
|
||||
let value = match reader.read_event_into(&mut value_buf) {
|
||||
Ok(Event::End(_)) => None,
|
||||
Ok(event) => {
|
||||
let value = match event {
|
||||
Event::Text(text) => text
|
||||
.xml_content(XmlVersion::Implicit1_0)
|
||||
.ok()
|
||||
.map(Cow::into_owned),
|
||||
_ => None,
|
||||
};
|
||||
reader
|
||||
.read_to_end_into(e.name(), &mut value_buf)
|
||||
.map_err(|err| {
|
||||
format!("Error at position {}: {:?}", reader.buffer_position(), err)
|
||||
})?;
|
||||
value
|
||||
}
|
||||
Err(err) => {
|
||||
return Err(format!(
|
||||
"Error at position {}: {:?}",
|
||||
reader.buffer_position(),
|
||||
err
|
||||
));
|
||||
}
|
||||
};
|
||||
|
||||
match (field, value) {
|
||||
(Some(RequestField::EmailAddress), Some(value)) => {
|
||||
email = Some(value);
|
||||
}
|
||||
(Some(RequestField::ResponseSchema), Some(value)) => {
|
||||
response_schema = ResponseSchema::parse(&value);
|
||||
}
|
||||
_ => (),
|
||||
}
|
||||
}
|
||||
Ok(Event::End(_) | Event::Eof) => break,
|
||||
Ok(_) => (),
|
||||
Err(e) => {
|
||||
return Err(format!(
|
||||
"Error at position {}: {:?}",
|
||||
reader.buffer_position(),
|
||||
e
|
||||
));
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
Err(format!(
|
||||
match email {
|
||||
Some(email) if email.contains('@') => Ok(AutodiscoverRequest {
|
||||
email: email.trim().to_lowercase(),
|
||||
response_schema,
|
||||
}),
|
||||
_ => Err(format!(
|
||||
"Expected email address, found unexpected value at position {}.",
|
||||
reader.buffer_position()
|
||||
))
|
||||
)),
|
||||
}
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::{AutodiscoverRequest, ResponseSchema, parse_autodiscover_request};
|
||||
|
||||
#[test]
|
||||
fn parse_autodiscover() {
|
||||
let r = r#"<?xml version="1.0" encoding="utf-8"?>
|
||||
const OUTLOOK: &str =
|
||||
"http://schemas.microsoft.com/exchange/autodiscover/outlook/responseschema/2006a";
|
||||
const MOBILESYNC: &str =
|
||||
"http://schemas.microsoft.com/exchange/autodiscover/mobilesync/responseschema/2006";
|
||||
|
||||
for (request, expected) in [
|
||||
(
|
||||
format!(
|
||||
r#"<?xml version="1.0" encoding="utf-8"?>
|
||||
<Autodiscover xmlns="http://schemas.microsoft.com/exchange/autodiscover/outlook/requestschema/2006">
|
||||
<Request>
|
||||
<EMailAddress>email@example.com</EMailAddress>
|
||||
<AcceptableResponseSchema>http://schemas.microsoft.com/exchange/autodiscover/outlook/responseschema/2006a</AcceptableResponseSchema>
|
||||
<EMailAddress>Email@Example.com</EMailAddress>
|
||||
<AcceptableResponseSchema>{OUTLOOK}</AcceptableResponseSchema>
|
||||
</Request>
|
||||
</Autodiscover>"#;
|
||||
|
||||
</Autodiscover>"#
|
||||
),
|
||||
ResponseSchema::Outlook,
|
||||
),
|
||||
(
|
||||
format!(
|
||||
r#"<Autodiscover xmlns="http://schemas.microsoft.com/exchange/autodiscover/outlook/requestschema/2006">
|
||||
<Request>
|
||||
<AcceptableResponseSchema>{OUTLOOK}</AcceptableResponseSchema>
|
||||
<EMailAddress>[email protected]</EMailAddress>
|
||||
</Request>
|
||||
</Autodiscover>"#
|
||||
),
|
||||
ResponseSchema::Outlook,
|
||||
),
|
||||
(
|
||||
r#"<Autodiscover>
|
||||
<Request>
|
||||
<EMailAddress>[email protected]</EMailAddress>
|
||||
</Request>
|
||||
</Autodiscover>"#
|
||||
.to_string(),
|
||||
ResponseSchema::Outlook,
|
||||
),
|
||||
(
|
||||
format!(
|
||||
r#"<?xml version="1.0" encoding="utf-8"?>
|
||||
<Autodiscover xmlns="http://schemas.microsoft.com/exchange/autodiscover/mobilesync/requestschema/2006">
|
||||
<Request>
|
||||
<EMailAddress>[email protected]</EMailAddress>
|
||||
<AcceptableResponseSchema>{MOBILESYNC}</AcceptableResponseSchema>
|
||||
</Request>
|
||||
</Autodiscover>"#
|
||||
),
|
||||
ResponseSchema::Unsupported,
|
||||
),
|
||||
(
|
||||
format!(
|
||||
r#"<Autodiscover>
|
||||
<Request>
|
||||
<LegacyDN>/o=Example/ou=Users/cn=email</LegacyDN>
|
||||
<Unknown><Nested>value</Nested><Empty/></Unknown>
|
||||
<AcceptableResponseSchema>{MOBILESYNC}</AcceptableResponseSchema>
|
||||
<EMailAddress>[email protected]</EMailAddress>
|
||||
</Request>
|
||||
</Autodiscover>"#
|
||||
),
|
||||
ResponseSchema::Unsupported,
|
||||
),
|
||||
] {
|
||||
assert_eq!(
|
||||
super::parse_autodiscover_request(r.as_bytes()).unwrap(),
|
||||
"[email protected]"
|
||||
parse_autodiscover_request(request.as_bytes()).expect("valid request"),
|
||||
AutodiscoverRequest {
|
||||
email: "[email protected]".to_string(),
|
||||
response_schema: expected,
|
||||
},
|
||||
"{request}"
|
||||
);
|
||||
}
|
||||
|
||||
for request in [
|
||||
"",
|
||||
"<Autodiscover><Request></Request></Autodiscover>",
|
||||
"<Autodiscover><Request><EMailAddress>no-domain</EMailAddress></Request></Autodiscover>",
|
||||
"<Autodiscover><Request><EMailAddress>[email protected]</Request></Autodiscover>",
|
||||
"<Request><EMailAddress>[email protected]</EMailAddress></Request>",
|
||||
] {
|
||||
assert!(
|
||||
parse_autodiscover_request(request.as_bytes()).is_err(),
|
||||
"{request}"
|
||||
);
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn autodiscover_encryption() {
|
||||
use registry::schema::{enums::ServiceProtocol, structs::Service};
|
||||
|
||||
@@ -1,5 +1,5 @@
|
||||
/*
|
||||
* SPDX-FileCopyrightText: 2026 Coffey Labs
|
||||
* SPDX-FileCopyrightText: 2026 Coffey Labs LLC
|
||||
*
|
||||
* SPDX-License-Identifier: AGPL-3.0-only
|
||||
*/
|
||||
|
||||
@@ -1,5 +1,5 @@
|
||||
/*
|
||||
* SPDX-FileCopyrightText: 2026 Coffey Labs
|
||||
* SPDX-FileCopyrightText: 2026 Coffey Labs LLC
|
||||
*
|
||||
* SPDX-License-Identifier: AGPL-3.0-only
|
||||
*/
|
||||
|
||||
@@ -1,5 +1,5 @@
|
||||
/*
|
||||
* SPDX-FileCopyrightText: 2026 Coffey Labs
|
||||
* SPDX-FileCopyrightText: 2026 Coffey Labs LLC
|
||||
*
|
||||
* SPDX-License-Identifier: AGPL-3.0-only
|
||||
*/
|
||||
|
||||
@@ -0,0 +1,293 @@
|
||||
/*
|
||||
* SPDX-FileCopyrightText: 2026 Coffey Labs LLC
|
||||
*
|
||||
* SPDX-License-Identifier: AGPL-3.0-only
|
||||
*/
|
||||
|
||||
//! Whether the outside world can reach each node's ports (settings-reorg,
|
||||
//! Ports: the reachability check).
|
||||
//!
|
||||
//! A server can't answer this about itself: a connection to its own public
|
||||
//! address never leaves the machine, so it passes whatever the firewall in
|
||||
//! front says. In a cluster the other nodes are outside that machine. Every
|
||||
//! ten minutes each node resolves every other active node's hostname, as a
|
||||
//! sender would, and tries a TCP connection to each listener port on each
|
||||
//! address. What it saw goes in the shared in-memory store for an hour, under
|
||||
//! (target, prober), so whichever node the admin asks can report it all.
|
||||
//!
|
||||
//! A single server has no one outside to ask. It reports only whether each
|
||||
//! port is listening, and says so.
|
||||
//!
|
||||
//! A connection is all that's tried: nothing is sent, so no protocol logs a
|
||||
//! session and no rate limit counts it.
|
||||
|
||||
use crate::{KV_PORT_REACHABILITY, Server};
|
||||
use registry::schema::{enums::ClusterNodeStatus, structs::NetworkListener};
|
||||
use serde::{Deserialize, Serialize};
|
||||
use serde_json::{Value, json};
|
||||
use std::{
|
||||
collections::BTreeSet,
|
||||
net::{IpAddr, Ipv4Addr, Ipv6Addr, SocketAddr},
|
||||
time::{Duration, Instant},
|
||||
};
|
||||
use store::{dispatch::lookup::KeyValue, write::now};
|
||||
|
||||
/// How often each node probes the others.
|
||||
pub const PROBE_INTERVAL: Duration = Duration::from_secs(600);
|
||||
/// How long one node's view of another is kept: long enough to span a missed round.
|
||||
const KEEP_FOR: u64 = 3600;
|
||||
const CONNECT_TIMEOUT: Duration = Duration::from_secs(5);
|
||||
|
||||
#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)]
|
||||
pub struct Probe {
|
||||
pub port: u16,
|
||||
pub address: String,
|
||||
pub ok: bool,
|
||||
#[serde(default, skip_serializing_if = "Option::is_none")]
|
||||
pub error: Option<String>,
|
||||
}
|
||||
|
||||
#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)]
|
||||
pub struct Report {
|
||||
/// Unix seconds.
|
||||
pub checked_at: u64,
|
||||
pub probes: Vec<Probe>,
|
||||
/// The hostname didn't resolve, so nothing could be tried.
|
||||
#[serde(default, skip_serializing_if = "Option::is_none")]
|
||||
pub error: Option<String>,
|
||||
}
|
||||
|
||||
/// The ports a sender or client could reach: every listener's port, leaving
|
||||
/// out listeners bound only to loopback, which are private by design.
|
||||
pub fn public_ports<'x>(listeners: impl IntoIterator<Item = &'x NetworkListener>) -> Vec<u16> {
|
||||
listeners
|
||||
.into_iter()
|
||||
.flat_map(|l| l.bind.iter())
|
||||
.map(|addr| addr.0)
|
||||
.filter(|addr| !addr.ip().is_loopback())
|
||||
.map(|addr| addr.port())
|
||||
.collect::<BTreeSet<_>>()
|
||||
.into_iter()
|
||||
.collect()
|
||||
}
|
||||
|
||||
fn key(target: &str, prober: &str) -> Vec<u8> {
|
||||
format!("{target}\n{prober}").into_bytes()
|
||||
}
|
||||
|
||||
async fn connect(address: SocketAddr) -> Result<(), String> {
|
||||
match tokio::time::timeout(CONNECT_TIMEOUT, tokio::net::TcpStream::connect(address)).await {
|
||||
Ok(Ok(_)) => Ok(()),
|
||||
Ok(Err(err)) => Err(err.to_string()),
|
||||
Err(_) => Err("no answer within 5 seconds".into()),
|
||||
}
|
||||
}
|
||||
|
||||
/// Tries each port on each address `hostname` resolves to.
|
||||
pub async fn probe_host(hostname: &str, ports: &[u16]) -> Report {
|
||||
let checked_at = now();
|
||||
let addresses = match tokio::net::lookup_host((hostname, 0)).await {
|
||||
Ok(found) => found.map(|a| a.ip()).collect::<BTreeSet<_>>(),
|
||||
Err(err) => {
|
||||
return Report {
|
||||
checked_at,
|
||||
probes: vec![],
|
||||
error: Some(format!("{hostname} doesn't resolve: {err}")),
|
||||
};
|
||||
}
|
||||
};
|
||||
let tries = addresses.iter().flat_map(|ip| {
|
||||
ports.iter().map(move |port| {
|
||||
let address = SocketAddr::new(*ip, *port);
|
||||
async move {
|
||||
let result = connect(address).await;
|
||||
Probe {
|
||||
port: *port,
|
||||
address: ip.to_string(),
|
||||
ok: result.is_ok(),
|
||||
error: result.err(),
|
||||
}
|
||||
}
|
||||
})
|
||||
});
|
||||
Report {
|
||||
checked_at,
|
||||
probes: futures::future::join_all(tries).await,
|
||||
error: None,
|
||||
}
|
||||
}
|
||||
|
||||
async fn listeners(server: &Server) -> trc::Result<Vec<NetworkListener>> {
|
||||
Ok(server
|
||||
.registry()
|
||||
.list::<NetworkListener>()
|
||||
.await?
|
||||
.into_iter()
|
||||
.map(|l| l.object)
|
||||
.collect())
|
||||
}
|
||||
|
||||
/// Where to knock to see a port listening on this machine: the bound
|
||||
/// address, or loopback of the same family for a wildcard bind.
|
||||
pub fn local_targets<'x>(
|
||||
listeners: impl IntoIterator<Item = &'x NetworkListener>,
|
||||
) -> Vec<SocketAddr> {
|
||||
listeners
|
||||
.into_iter()
|
||||
.flat_map(|l| l.bind.iter())
|
||||
.map(|addr| addr.0)
|
||||
.filter(|addr| !addr.ip().is_loopback())
|
||||
.map(|addr| match addr.ip() {
|
||||
IpAddr::V4(ip) if ip.is_unspecified() => {
|
||||
SocketAddr::new(Ipv4Addr::LOCALHOST.into(), addr.port())
|
||||
}
|
||||
IpAddr::V6(ip) if ip.is_unspecified() => {
|
||||
SocketAddr::new(Ipv6Addr::LOCALHOST.into(), addr.port())
|
||||
}
|
||||
_ => addr,
|
||||
})
|
||||
.collect::<BTreeSet<_>>()
|
||||
.into_iter()
|
||||
.collect()
|
||||
}
|
||||
|
||||
/// One round: this node probes every other active node and records what it saw.
|
||||
pub async fn probe_peers(server: &Server) -> trc::Result<()> {
|
||||
let nodes = server.registry().cluster_node_list().await?;
|
||||
let me = server.registry().node_id() as u64;
|
||||
let Some(prober) = nodes
|
||||
.iter()
|
||||
.find(|n| n.node_id == me)
|
||||
.map(|n| n.hostname.clone())
|
||||
else {
|
||||
return Ok(());
|
||||
};
|
||||
let ports = public_ports(&listeners(server).await?);
|
||||
for target in nodes.iter().filter(|n| {
|
||||
n.node_id != me && n.status == ClusterNodeStatus::Active && n.hostname != prober
|
||||
}) {
|
||||
let report = probe_host(&target.hostname, &ports).await;
|
||||
server
|
||||
.in_memory_store()
|
||||
.key_set(
|
||||
KeyValue::with_prefix(
|
||||
KV_PORT_REACHABILITY,
|
||||
key(&target.hostname, &prober),
|
||||
serde_json::to_vec(&report).unwrap_or_default(),
|
||||
)
|
||||
.expires(KEEP_FOR),
|
||||
)
|
||||
.await?;
|
||||
}
|
||||
Ok(())
|
||||
}
|
||||
|
||||
/// What `GET /api/ports/check` answers.
|
||||
pub async fn report(server: &Server) -> trc::Result<Value> {
|
||||
let listeners = listeners(server).await?;
|
||||
let ports = public_ports(&listeners);
|
||||
let nodes = if server.core.storage.coordinator.is_enabled() {
|
||||
server.registry().cluster_node_list().await?
|
||||
} else {
|
||||
vec![]
|
||||
};
|
||||
let active = nodes
|
||||
.iter()
|
||||
.filter(|n| n.status == ClusterNodeStatus::Active)
|
||||
.collect::<Vec<_>>();
|
||||
|
||||
if active.len() < 2 {
|
||||
// No one outside to ask: only whether each port is listening here.
|
||||
let started = Instant::now();
|
||||
let listening = futures::future::join_all(local_targets(&listeners).into_iter().map(
|
||||
|address| async move {
|
||||
let result = connect(address).await;
|
||||
json!({ "port": address.port(), "address": address.ip().to_string(), "listening": result.is_ok() })
|
||||
},
|
||||
))
|
||||
.await;
|
||||
return Ok(json!({
|
||||
"mode": "local",
|
||||
"ports": ports,
|
||||
"listening": listening,
|
||||
"ms": started.elapsed().as_millis() as u64,
|
||||
}));
|
||||
}
|
||||
|
||||
let mut out = Vec::new();
|
||||
for target in &active {
|
||||
let mut seen_by = Vec::new();
|
||||
for prober in active.iter().filter(|p| p.node_id != target.node_id) {
|
||||
let stored = server
|
||||
.in_memory_store()
|
||||
.key_get::<String>(KeyValue::<()>::build_key(
|
||||
KV_PORT_REACHABILITY,
|
||||
key(&target.hostname, &prober.hostname),
|
||||
))
|
||||
.await?;
|
||||
let report = stored.and_then(|raw| serde_json::from_str::<Report>(&raw).ok());
|
||||
seen_by.push(json!({ "prober": prober.hostname, "report": report }));
|
||||
}
|
||||
out.push(json!({ "hostname": target.hostname, "seenBy": seen_by }));
|
||||
}
|
||||
Ok(json!({
|
||||
"mode": "cluster",
|
||||
"ports": ports,
|
||||
"intervalSeconds": PROBE_INTERVAL.as_secs(),
|
||||
"nodes": out,
|
||||
}))
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::*;
|
||||
|
||||
fn listener(binds: &[&str]) -> NetworkListener {
|
||||
NetworkListener {
|
||||
bind: registry::schema::prelude::Map::new(
|
||||
binds.iter().map(|b| b.parse().unwrap()).collect(),
|
||||
),
|
||||
..Default::default()
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn public_ports_leave_out_loopback_only_listeners() {
|
||||
let listeners = [
|
||||
listener(&["[::]:25"]),
|
||||
listener(&["0.0.0.0:993", "[::]:993"]),
|
||||
listener(&["127.0.0.1:8080"]),
|
||||
listener(&["203.0.113.5:465"]),
|
||||
];
|
||||
assert_eq!(public_ports(listeners.iter()), vec![25, 465, 993]);
|
||||
assert_eq!(
|
||||
local_targets(listeners.iter())
|
||||
.iter()
|
||||
.map(ToString::to_string)
|
||||
.collect::<Vec<_>>(),
|
||||
vec!["127.0.0.1:993", "203.0.113.5:465", "[::1]:25", "[::1]:993"]
|
||||
);
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn probe_host_reports_open_and_closed_ports() {
|
||||
let open = tokio::net::TcpListener::bind("127.0.0.1:0").await.unwrap();
|
||||
let open_port = open.local_addr().unwrap().port();
|
||||
let closed_port = {
|
||||
let l = std::net::TcpListener::bind("127.0.0.1:0").unwrap();
|
||||
l.local_addr().unwrap().port()
|
||||
};
|
||||
let report = probe_host("127.0.0.1", &[open_port, closed_port]).await;
|
||||
assert_eq!(report.error, None);
|
||||
let ok = |port| report.probes.iter().find(|p| p.port == port).unwrap().ok;
|
||||
assert!(ok(open_port));
|
||||
assert!(!ok(closed_port));
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn probe_host_says_when_a_name_does_not_resolve() {
|
||||
let report = probe_host("does-not-exist.invalid", &[25]).await;
|
||||
assert!(report.probes.is_empty());
|
||||
assert!(report.error.unwrap().contains("doesn't resolve"));
|
||||
}
|
||||
}
|
||||
@@ -1,5 +1,5 @@
|
||||
/*
|
||||
* SPDX-FileCopyrightText: 2026 Coffey Labs
|
||||
* SPDX-FileCopyrightText: 2026 Coffey Labs LLC
|
||||
*
|
||||
* SPDX-License-Identifier: AGPL-3.0-only
|
||||
*/
|
||||
|
||||
@@ -1,5 +1,5 @@
|
||||
/*
|
||||
* SPDX-FileCopyrightText: 2026 Coffey Labs
|
||||
* SPDX-FileCopyrightText: 2026 Coffey Labs LLC
|
||||
*
|
||||
* SPDX-License-Identifier: AGPL-3.0-only
|
||||
*/
|
||||
|
||||
@@ -1,5 +1,5 @@
|
||||
/*
|
||||
* SPDX-FileCopyrightText: 2026 Coffey Labs
|
||||
* SPDX-FileCopyrightText: 2026 Coffey Labs LLC
|
||||
*
|
||||
* SPDX-License-Identifier: AGPL-3.0-only
|
||||
*/
|
||||
|
||||
@@ -1,5 +1,5 @@
|
||||
/*
|
||||
* SPDX-FileCopyrightText: 2026 Coffey Labs
|
||||
* SPDX-FileCopyrightText: 2026 Coffey Labs LLC
|
||||
*
|
||||
* SPDX-License-Identifier: AGPL-3.0-only
|
||||
*/
|
||||
@@ -104,14 +104,31 @@ impl StoredMetric {
|
||||
pub fn timestamp(&self) -> u64 {
|
||||
SnowflakeIdGenerator::to_timestamp(self.id)
|
||||
}
|
||||
|
||||
/// The node that wrote the sample. Histogram totals are per node, so a
|
||||
/// reader diffs them per node.
|
||||
pub fn node_id(&self) -> u64 {
|
||||
SnowflakeIdGenerator::to_node_id(self.id)
|
||||
}
|
||||
}
|
||||
|
||||
/// What the node wrote last, so counters and histograms are written as
|
||||
/// changes (MON-4). Per process: a restart counts from the start.
|
||||
static LAST: Mutex<Option<AHashMap<MetricType, (u64, u64)>>> = Mutex::new(None);
|
||||
|
||||
/// One tick's samples (MON-4 to MON-6).
|
||||
pub fn sample() -> Vec<Metric> {
|
||||
/// Gauges that count the whole cluster's data, not this node's. Only the node
|
||||
/// that computes them (the metrics-calculation role) has a true reading; on
|
||||
/// the others the queue gauge only moves with local queue events and drifts
|
||||
/// below zero, and the account and domain counts stay at 0.
|
||||
const CLUSTER_GAUGES: [MetricType; 3] = [
|
||||
MetricType::QueueCount,
|
||||
MetricType::UserCount,
|
||||
MetricType::DomainCount,
|
||||
];
|
||||
|
||||
/// One tick's samples (MON-4 to MON-6). `calculates` is whether this node
|
||||
/// computes the cluster-wide gauges; a node that doesn't leaves them out.
|
||||
pub fn sample(calculates: bool) -> Vec<Metric> {
|
||||
let mut last_guard = LAST.lock().unwrap();
|
||||
let last = last_guard.get_or_insert_with(AHashMap::new);
|
||||
let mut samples = Vec::new();
|
||||
@@ -134,6 +151,9 @@ pub fn sample() -> Vec<Metric> {
|
||||
|
||||
// Gauges: the reading, always (MON-5)
|
||||
for gauge in Collector::collect_gauges() {
|
||||
if !calculates && CLUSTER_GAUGES.contains(&gauge.id()) {
|
||||
continue;
|
||||
}
|
||||
samples.push(Metric::Gauge(MetricCount {
|
||||
count: gauge.get(),
|
||||
metric: gauge.id(),
|
||||
@@ -175,7 +195,7 @@ impl Server {
|
||||
if store.is_none() {
|
||||
return;
|
||||
}
|
||||
let samples = sample();
|
||||
let samples = sample(self.core.network.roles.metrics_calculate);
|
||||
let count = samples.len();
|
||||
let started = std::time::Instant::now();
|
||||
match store.write_metrics(samples, now()).await {
|
||||
@@ -265,3 +285,41 @@ impl Server {
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::*;
|
||||
|
||||
fn gauges(samples: &[Metric]) -> Vec<MetricType> {
|
||||
samples
|
||||
.iter()
|
||||
.filter_map(|m| match m {
|
||||
Metric::Gauge(g) => Some(g.metric),
|
||||
_ => None,
|
||||
})
|
||||
.collect()
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn only_the_calculating_node_stores_cluster_gauges() {
|
||||
let all = gauges(&sample(true));
|
||||
let local = gauges(&sample(false));
|
||||
for metric in CLUSTER_GAUGES {
|
||||
assert!(
|
||||
all.contains(&metric),
|
||||
"{metric:?} missing on the calculating node"
|
||||
);
|
||||
assert!(
|
||||
!local.contains(&metric),
|
||||
"{metric:?} stored by a node that doesn't compute it"
|
||||
);
|
||||
}
|
||||
// Per-node gauges are stored either way
|
||||
for metric in [MetricType::ServerMemory, MetricType::HttpActiveConnections] {
|
||||
assert!(
|
||||
all.contains(&metric) && local.contains(&metric),
|
||||
"{metric:?}"
|
||||
);
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -1,5 +1,5 @@
|
||||
/*
|
||||
* SPDX-FileCopyrightText: 2026 Coffey Labs
|
||||
* SPDX-FileCopyrightText: 2026 Coffey Labs LLC
|
||||
*
|
||||
* SPDX-License-Identifier: AGPL-3.0-only
|
||||
*/
|
||||
|
||||
@@ -156,15 +156,7 @@ async fn post_webhook_events(
|
||||
|
||||
// Add HMAC-SHA256 signature
|
||||
let mut headers = settings.headers.clone();
|
||||
if !settings.key.is_empty() {
|
||||
let key = hmac::Key::new(hmac::HMAC_SHA256, settings.key.as_bytes());
|
||||
let tag = hmac::sign(&key, body.as_bytes());
|
||||
|
||||
headers.insert(
|
||||
"X-Signature",
|
||||
STANDARD.encode(tag.as_ref()).parse().unwrap(),
|
||||
);
|
||||
}
|
||||
sign(&mut headers, &settings.key, &body);
|
||||
|
||||
// Send request
|
||||
let response = settings
|
||||
@@ -188,3 +180,150 @@ async fn post_webhook_events(
|
||||
))
|
||||
}
|
||||
}
|
||||
|
||||
/// Adds the HMAC-SHA256 `X-Signature` a receiver checks, when the webhook has a key.
|
||||
fn sign(headers: &mut hyper::HeaderMap, key: &str, body: &str) {
|
||||
if !key.is_empty() {
|
||||
let key = hmac::Key::new(hmac::HMAC_SHA256, key.as_bytes());
|
||||
let tag = hmac::sign(&key, body.as_bytes());
|
||||
|
||||
headers.insert(
|
||||
"X-Signature",
|
||||
STANDARD.encode(tag.as_ref()).parse().unwrap(),
|
||||
);
|
||||
}
|
||||
}
|
||||
|
||||
/// inbuxa: "Send test" for a saved webhook (settings-reorg, Webhooks). One
|
||||
/// sample event, sent the way a real batch is: the same URL, headers, sign-in,
|
||||
/// signature, timeout and certificate checks. The event's type,
|
||||
/// `webhook.test`, is none the server raises, and an `X-Inbuxa-Test` header
|
||||
/// marks it, so a receiver can tell it apart. Answers the HTTP status, or why
|
||||
/// nothing came back.
|
||||
pub async fn send_test(hook: ®istry::schema::structs::WebHook) -> Result<u16, String> {
|
||||
let mut headers = hook
|
||||
.http_auth
|
||||
.build_headers(hook.http_headers.clone(), "application/json".into())
|
||||
.await
|
||||
.map_err(|err| format!("Unable to build HTTP headers: {err}"))?;
|
||||
let key = hook
|
||||
.signature_key
|
||||
.secret()
|
||||
.await
|
||||
.map_err(|err| format!("Unable to retrieve signature key: {err}"))?
|
||||
.unwrap_or_default()
|
||||
.into_owned();
|
||||
|
||||
let created = now();
|
||||
let body = serde_json::json!({
|
||||
"events": [{
|
||||
"id": format!("test-{created}"),
|
||||
"createdAt": mail_parser::DateTime::from_timestamp(created as i64).to_rfc3339(),
|
||||
"type": "webhook.test",
|
||||
"data": { "details": "A test from inbuxa Admin. Nothing happened on the server." },
|
||||
}]
|
||||
})
|
||||
.to_string();
|
||||
sign(&mut headers, &key, &body);
|
||||
headers.insert("X-Inbuxa-Test", "true".parse().unwrap());
|
||||
|
||||
let response = utils::http::http_client_builder(hook.allow_invalid_certs)
|
||||
.build()
|
||||
.map_err(|err| format!("Unable to build an HTTP client: {err}"))?
|
||||
.post(&hook.url)
|
||||
.timeout(hook.timeout.into_inner())
|
||||
.headers(headers)
|
||||
.body(body)
|
||||
.send()
|
||||
.await
|
||||
.map_err(|err| format!("Webhook request to {} failed: {err}", hook.url))?;
|
||||
Ok(response.status().as_u16())
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::*;
|
||||
use registry::schema::structs::{SecretKeyOptional, SecretKeyValue, WebHook};
|
||||
use tokio::io::{AsyncReadExt, AsyncWriteExt};
|
||||
|
||||
/// One request in, the given status out; hands back what was received.
|
||||
async fn receiver(status: &'static str) -> (String, tokio::task::JoinHandle<String>) {
|
||||
let listener = tokio::net::TcpListener::bind("127.0.0.1:0").await.unwrap();
|
||||
let url = format!("http://{}/hook", listener.local_addr().unwrap());
|
||||
let task = tokio::spawn(async move {
|
||||
let (mut socket, _) = listener.accept().await.unwrap();
|
||||
let mut buf = Vec::new();
|
||||
let mut chunk = [0u8; 4096];
|
||||
loop {
|
||||
let n = socket.read(&mut chunk).await.unwrap();
|
||||
buf.extend_from_slice(&chunk[..n]);
|
||||
let text = String::from_utf8_lossy(&buf);
|
||||
if let Some(end) = text.find("\r\n\r\n") {
|
||||
let length = text[..end]
|
||||
.lines()
|
||||
.find_map(|l| {
|
||||
l.to_ascii_lowercase()
|
||||
.strip_prefix("content-length:")
|
||||
.map(|v| v.trim().parse::<usize>().unwrap())
|
||||
})
|
||||
.unwrap_or(0);
|
||||
if buf.len() >= end + 4 + length || n == 0 {
|
||||
break;
|
||||
}
|
||||
}
|
||||
}
|
||||
socket
|
||||
.write_all(
|
||||
format!("HTTP/1.1 {status}\r\ncontent-length: 0\r\nconnection: close\r\n\r\n")
|
||||
.as_bytes(),
|
||||
)
|
||||
.await
|
||||
.unwrap();
|
||||
String::from_utf8_lossy(&buf).into_owned()
|
||||
});
|
||||
(url, task)
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn send_test_signs_and_marks_the_sample() {
|
||||
let (url, task) = receiver("204 No Content").await;
|
||||
let hook = WebHook {
|
||||
url,
|
||||
enable: false,
|
||||
signature_key: SecretKeyOptional::Value(SecretKeyValue { secret: "k".into() }),
|
||||
..Default::default()
|
||||
};
|
||||
assert_eq!(send_test(&hook).await, Ok(204));
|
||||
|
||||
let request = task.await.unwrap();
|
||||
let (head, body) = request.split_once("\r\n\r\n").unwrap();
|
||||
let head = head.to_ascii_lowercase();
|
||||
assert!(head.contains("x-inbuxa-test: true"), "{head}");
|
||||
let parsed: serde_json::Value = serde_json::from_str(body).unwrap();
|
||||
assert_eq!(parsed["events"][0]["type"], "webhook.test");
|
||||
let tag = hmac::sign(&hmac::Key::new(hmac::HMAC_SHA256, b"k"), body.as_bytes());
|
||||
assert!(
|
||||
head.contains(&format!(
|
||||
"x-signature: {}",
|
||||
STANDARD.encode(tag.as_ref()).to_ascii_lowercase()
|
||||
)),
|
||||
"{head}"
|
||||
);
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn send_test_reports_what_came_back() {
|
||||
let (url, _task) = receiver("403 Forbidden").await;
|
||||
let hook = WebHook {
|
||||
url,
|
||||
..Default::default()
|
||||
};
|
||||
assert_eq!(send_test(&hook).await, Ok(403));
|
||||
|
||||
let hook = WebHook {
|
||||
url: "http://127.0.0.1:9/hook".into(),
|
||||
..Default::default()
|
||||
};
|
||||
assert!(send_test(&hook).await.unwrap_err().contains("failed"));
|
||||
}
|
||||
}
|
||||
@@ -1,6 +1,6 @@
|
||||
[package]
|
||||
name = "coordinator"
|
||||
version = "0.16.24"
|
||||
version = "0.16.25"
|
||||
edition = "2024"
|
||||
|
||||
[dependencies]
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
[package]
|
||||
name = "dav-proto"
|
||||
version = "0.16.24"
|
||||
version = "0.16.25"
|
||||
edition = "2024"
|
||||
|
||||
[dependencies]
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
[package]
|
||||
name = "dav"
|
||||
version = "0.16.24"
|
||||
version = "0.16.25"
|
||||
edition = "2024"
|
||||
|
||||
[dependencies]
|
||||
|
||||
@@ -133,6 +133,10 @@ impl DavAclHandler for Server {
|
||||
{
|
||||
return Err(DavError::Code(StatusCode::FORBIDDEN));
|
||||
}
|
||||
// inbuxa: MA-D0: a group's members don't share what it owns on.
|
||||
if access_token.is_group_member_only(account_id) {
|
||||
return Err(DavError::Code(StatusCode::FORBIDDEN));
|
||||
}
|
||||
|
||||
// Validate ACEs
|
||||
let grants = self
|
||||
@@ -565,7 +569,13 @@ impl Privileges for AccessToken {
|
||||
grants: &ArchivedVec<ArchivedAclGrant>,
|
||||
is_calendar: bool,
|
||||
) -> Vec<Privilege> {
|
||||
if self.is_member(account_id) {
|
||||
if self.is_group_member_only(account_id) {
|
||||
// inbuxa: MA-D0: everything but sharing it on.
|
||||
Privilege::all(is_calendar)
|
||||
.into_iter()
|
||||
.filter(|privilege| !matches!(privilege, Privilege::All | Privilege::WriteAcl))
|
||||
.collect()
|
||||
} else if self.is_member(account_id) {
|
||||
Privilege::all(is_calendar)
|
||||
} else {
|
||||
current_user_privilege_set(grants.effective_acl(self))
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
[package]
|
||||
name = "directory"
|
||||
version = "0.16.24"
|
||||
version = "0.16.25"
|
||||
edition = "2024"
|
||||
|
||||
[dependencies]
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
[package]
|
||||
name = "email"
|
||||
version = "0.16.24"
|
||||
version = "0.16.25"
|
||||
edition = "2024"
|
||||
|
||||
[dependencies]
|
||||
|
||||
@@ -1,5 +1,5 @@
|
||||
/*
|
||||
* SPDX-FileCopyrightText: 2026 Coffey Labs
|
||||
* SPDX-FileCopyrightText: 2026 Coffey Labs LLC
|
||||
*
|
||||
* SPDX-License-Identifier: AGPL-3.0-only
|
||||
*/
|
||||
|
||||
@@ -20,7 +20,7 @@ use groupware::{
|
||||
scheduling::{ItipError, ItipMessages},
|
||||
};
|
||||
use mail_parser::{
|
||||
DateTime, Header, HeaderName, HeaderValue, Message, MessageParser, MimeHeaders, PartType,
|
||||
Header, HeaderName, HeaderValue, Message, MessageParser, MimeHeaders, PartType,
|
||||
parsers::fields::thread::thread_name,
|
||||
};
|
||||
use registry::{
|
||||
@@ -924,11 +924,7 @@ impl EmailIngest for Server {
|
||||
span_id: u64,
|
||||
) {
|
||||
if let Some(config) = &self.core.spam.classifier {
|
||||
let mut dt = DateTime::from_timestamp(now() as i64);
|
||||
dt.hour = 0;
|
||||
dt.minute = 0;
|
||||
dt.second = 0;
|
||||
let until = dt.to_timestamp() as u64 + config.hold_samples_for;
|
||||
let until = now() + config.hold_samples_for;
|
||||
|
||||
let sample = SpamTrainingSample {
|
||||
account_id: Some(Id::from(account_id)),
|
||||
|
||||
@@ -290,7 +290,11 @@ impl SieveScriptIngest for Server {
|
||||
// inbuxa: AL-4: a locked account answers no sender, so a
|
||||
// rejection is kept instead; sieve has already cleared
|
||||
// the implicit keep, so it is filed here
|
||||
Event::Reject { .. } if access_token.is_locked() => {
|
||||
// A shared mailbox (MA-S) is a role address and answers
|
||||
// as one: its Sieve script runs as written
|
||||
Event::Reject { .. }
|
||||
if access_token.is_locked() && !access_token.is_shared_mailbox() =>
|
||||
{
|
||||
if let Some(message) = messages.get_mut(0)
|
||||
&& !message.file_into.contains(&INBOX_ID)
|
||||
{
|
||||
@@ -403,7 +407,11 @@ impl SieveScriptIngest for Server {
|
||||
// inbuxa: AL-4: a locked account sends nothing on its
|
||||
// own: no redirect, vacation reply or notification. An
|
||||
// unsent redirect leaves the message to be kept.
|
||||
Event::SendMessage { .. } if access_token.is_locked() => {
|
||||
// A shared mailbox's acknowledgements and redirects go
|
||||
// out (MA-S).
|
||||
Event::SendMessage { .. }
|
||||
if access_token.is_locked() && !access_token.is_shared_mailbox() =>
|
||||
{
|
||||
trc::event!(
|
||||
Sieve(SieveEvent::ActionReject),
|
||||
Details = "Account is locked: nothing is sent",
|
||||
|
||||
@@ -21,6 +21,13 @@ base64 = "0.23"
|
||||
sha2 = "0.11"
|
||||
flate2 = "1.1"
|
||||
tokio = { version = "1.53", features = ["sync", "rt"] }
|
||||
# inbuxa: DLP detectors and attachment text (dlp-and-mail-flow-rules spec)
|
||||
regex = "1.13.1"
|
||||
aho-corasick = "1.1"
|
||||
zip = "8.6"
|
||||
quick-xml = "0.41"
|
||||
mail-parser = { version = "0.11", features = ["full_encoding"] }
|
||||
mail-builder = { version = "1.0" }
|
||||
|
||||
[dev-dependencies]
|
||||
tokio = { version = "1.53", features = ["macros", "rt"] }
|
||||
@@ -1,5 +1,5 @@
|
||||
/*
|
||||
* SPDX-FileCopyrightText: 2026 Coffey Labs
|
||||
* SPDX-FileCopyrightText: 2026 Coffey Labs LLC
|
||||
*
|
||||
* SPDX-License-Identifier: AGPL-3.0-only
|
||||
*/
|
||||
|
||||
@@ -1,5 +1,5 @@
|
||||
/*
|
||||
* SPDX-FileCopyrightText: 2026 Coffey Labs
|
||||
* SPDX-FileCopyrightText: 2026 Coffey Labs LLC
|
||||
*
|
||||
* SPDX-License-Identifier: AGPL-3.0-only
|
||||
*/
|
||||
|
||||
@@ -1,5 +1,5 @@
|
||||
/*
|
||||
* SPDX-FileCopyrightText: 2026 Coffey Labs
|
||||
* SPDX-FileCopyrightText: 2026 Coffey Labs LLC
|
||||
*
|
||||
* SPDX-License-Identifier: AGPL-3.0-only
|
||||
*/
|
||||
|
||||
@@ -1,5 +1,5 @@
|
||||
/*
|
||||
* SPDX-FileCopyrightText: 2026 Coffey Labs
|
||||
* SPDX-FileCopyrightText: 2026 Coffey Labs LLC
|
||||
*
|
||||
* SPDX-License-Identifier: AGPL-3.0-only
|
||||
*/
|
||||
|
||||
@@ -1,5 +1,5 @@
|
||||
/*
|
||||
* SPDX-FileCopyrightText: 2026 Coffey Labs
|
||||
* SPDX-FileCopyrightText: 2026 Coffey Labs LLC
|
||||
*
|
||||
* SPDX-License-Identifier: AGPL-3.0-only
|
||||
*/
|
||||
|
||||
@@ -1,5 +1,5 @@
|
||||
/*
|
||||
* SPDX-FileCopyrightText: 2026 Coffey Labs
|
||||
* SPDX-FileCopyrightText: 2026 Coffey Labs LLC
|
||||
*
|
||||
* SPDX-License-Identifier: AGPL-3.0-only
|
||||
*/
|
||||
|
||||
@@ -1,5 +1,5 @@
|
||||
/*
|
||||
* SPDX-FileCopyrightText: 2026 Coffey Labs
|
||||
* SPDX-FileCopyrightText: 2026 Coffey Labs LLC
|
||||
*
|
||||
* SPDX-License-Identifier: AGPL-3.0-only
|
||||
*/
|
||||
|
||||
@@ -1,5 +1,5 @@
|
||||
/*
|
||||
* SPDX-FileCopyrightText: 2026 Coffey Labs
|
||||
* SPDX-FileCopyrightText: 2026 Coffey Labs LLC
|
||||
*
|
||||
* SPDX-License-Identifier: AGPL-3.0-only
|
||||
*/
|
||||
|
||||
@@ -1,5 +1,5 @@
|
||||
/*
|
||||
* SPDX-FileCopyrightText: 2026 Coffey Labs
|
||||
* SPDX-FileCopyrightText: 2026 Coffey Labs LLC
|
||||
*
|
||||
* SPDX-License-Identifier: AGPL-3.0-only
|
||||
*/
|
||||
|
||||
@@ -1,5 +1,5 @@
|
||||
/*
|
||||
* SPDX-FileCopyrightText: 2026 Coffey Labs
|
||||
* SPDX-FileCopyrightText: 2026 Coffey Labs LLC
|
||||
*
|
||||
* SPDX-License-Identifier: AGPL-3.0-only
|
||||
*/
|
||||
|
||||
@@ -1,5 +1,5 @@
|
||||
/*
|
||||
* SPDX-FileCopyrightText: 2026 Coffey Labs
|
||||
* SPDX-FileCopyrightText: 2026 Coffey Labs LLC
|
||||
*
|
||||
* SPDX-License-Identifier: AGPL-3.0-only
|
||||
*/
|
||||
|
||||
@@ -1,5 +1,5 @@
|
||||
/*
|
||||
* SPDX-FileCopyrightText: 2026 Coffey Labs
|
||||
* SPDX-FileCopyrightText: 2026 Coffey Labs LLC
|
||||
*
|
||||
* SPDX-License-Identifier: AGPL-3.0-only
|
||||
*/
|
||||
|
||||
@@ -1,5 +1,5 @@
|
||||
/*
|
||||
* SPDX-FileCopyrightText: 2026 Coffey Labs
|
||||
* SPDX-FileCopyrightText: 2026 Coffey Labs LLC
|
||||
*
|
||||
* SPDX-License-Identifier: AGPL-3.0-only
|
||||
*/
|
||||
|
||||
@@ -1,5 +1,5 @@
|
||||
/*
|
||||
* SPDX-FileCopyrightText: 2026 Coffey Labs
|
||||
* SPDX-FileCopyrightText: 2026 Coffey Labs LLC
|
||||
*
|
||||
* SPDX-License-Identifier: AGPL-3.0-only
|
||||
*/
|
||||
|
||||
@@ -1,5 +1,5 @@
|
||||
/*
|
||||
* SPDX-FileCopyrightText: 2026 Coffey Labs
|
||||
* SPDX-FileCopyrightText: 2026 Coffey Labs LLC
|
||||
*
|
||||
* SPDX-License-Identifier: AGPL-3.0-only
|
||||
*/
|
||||
|
||||
@@ -1,5 +1,5 @@
|
||||
/*
|
||||
* SPDX-FileCopyrightText: 2026 Coffey Labs
|
||||
* SPDX-FileCopyrightText: 2026 Coffey Labs LLC
|
||||
*
|
||||
* SPDX-License-Identifier: AGPL-3.0-only
|
||||
*/
|
||||
|
||||
@@ -1,5 +1,5 @@
|
||||
/*
|
||||
* SPDX-FileCopyrightText: 2026 Coffey Labs
|
||||
* SPDX-FileCopyrightText: 2026 Coffey Labs LLC
|
||||
*
|
||||
* SPDX-License-Identifier: AGPL-3.0-only
|
||||
*/
|
||||
|
||||
@@ -1,5 +1,5 @@
|
||||
/*
|
||||
* SPDX-FileCopyrightText: 2026 Coffey Labs
|
||||
* SPDX-FileCopyrightText: 2026 Coffey Labs LLC
|
||||
*
|
||||
* SPDX-License-Identifier: AGPL-3.0-only
|
||||
*/
|
||||
|
||||
@@ -1,5 +1,5 @@
|
||||
/*
|
||||
* SPDX-FileCopyrightText: 2026 Coffey Labs
|
||||
* SPDX-FileCopyrightText: 2026 Coffey Labs LLC
|
||||
*
|
||||
* SPDX-License-Identifier: AGPL-3.0-only
|
||||
*/
|
||||
|
||||
@@ -1,5 +1,5 @@
|
||||
/*
|
||||
* SPDX-FileCopyrightText: 2026 Coffey Labs
|
||||
* SPDX-FileCopyrightText: 2026 Coffey Labs LLC
|
||||
*
|
||||
* SPDX-License-Identifier: AGPL-3.0-only
|
||||
*/
|
||||
|
||||
@@ -1,5 +1,5 @@
|
||||
/*
|
||||
* SPDX-FileCopyrightText: 2026 Coffey Labs
|
||||
* SPDX-FileCopyrightText: 2026 Coffey Labs LLC
|
||||
*
|
||||
* SPDX-License-Identifier: AGPL-3.0-only
|
||||
*/
|
||||
|
||||
@@ -0,0 +1,282 @@
|
||||
/*
|
||||
* SPDX-FileCopyrightText: 2026 Coffey Labs LLC
|
||||
*
|
||||
* SPDX-License-Identifier: AGPL-3.0-only
|
||||
*/
|
||||
|
||||
//! The blocklists a node asks about itself (deliverability spec, DL-6), and
|
||||
//! how to read each one's answer.
|
||||
//!
|
||||
//! A list answers with an address in 127.0.0.0/8. Each list says which of
|
||||
//! those mean "listed" and which mean "I won't answer you": Spamhaus, for
|
||||
//! one, answers `127.255.255.254` to a query that came through a public
|
||||
//! resolver. A refusal is never read as a listing (DL-4).
|
||||
|
||||
use std::net::{IpAddr, Ipv4Addr};
|
||||
|
||||
#[derive(Debug, Clone, Copy, PartialEq, Eq)]
|
||||
pub enum Scope {
|
||||
/// Looked up by the reversed address: `2.0.0.127.zen.spamhaus.org`.
|
||||
Ip,
|
||||
/// Looked up by name: `example.org.dbl.spamhaus.org`.
|
||||
Domain,
|
||||
}
|
||||
|
||||
#[derive(Debug, Clone, Copy)]
|
||||
pub struct BlockList {
|
||||
/// What the page and the settings call it.
|
||||
pub name: &'static str,
|
||||
pub zone: &'static str,
|
||||
pub scope: Scope,
|
||||
/// Where an administrator looks the address up and asks for removal.
|
||||
pub lookup: &'static str,
|
||||
/// Something the page says beside the list.
|
||||
pub note: Option<&'static str>,
|
||||
read: fn(Ipv4Addr) -> Answer,
|
||||
}
|
||||
|
||||
/// What a list's answer means.
|
||||
#[derive(Debug, Clone, PartialEq, Eq)]
|
||||
pub enum Answer {
|
||||
Listed(&'static str),
|
||||
/// The list won't answer this resolver, or not now.
|
||||
Refused(&'static str),
|
||||
/// A code the list doesn't define: neither listed nor clean.
|
||||
Unknown,
|
||||
}
|
||||
|
||||
impl BlockList {
|
||||
pub fn read(&self, answer: Ipv4Addr) -> Answer {
|
||||
(self.read)(answer)
|
||||
}
|
||||
|
||||
/// The name to look up for `subject`, or None when the subject doesn't
|
||||
/// suit the list (a domain on an IP list, or an IPv6 address: none of
|
||||
/// these lists publish IPv6 zones worth asking).
|
||||
pub fn query(&self, subject: &Subject<'_>) -> Option<String> {
|
||||
match (self.scope, subject) {
|
||||
(Scope::Ip, Subject::Ip(IpAddr::V4(ip))) => {
|
||||
let [a, b, c, d] = ip.octets();
|
||||
Some(format!("{d}.{c}.{b}.{a}.{}.", self.zone))
|
||||
}
|
||||
(Scope::Domain, Subject::Domain(domain)) => {
|
||||
Some(format!("{}.{}.", domain.trim_end_matches('.'), self.zone))
|
||||
}
|
||||
_ => None,
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
pub enum Subject<'x> {
|
||||
Ip(IpAddr),
|
||||
Domain(&'x str),
|
||||
}
|
||||
|
||||
/// Spamhaus' error codes, the same on every Spamhaus zone.
|
||||
fn spamhaus_refusal(ip: Ipv4Addr) -> Option<Answer> {
|
||||
match ip.octets() {
|
||||
[127, 255, 255, 252] => Some(Answer::Refused("The query was malformed")),
|
||||
[127, 255, 255, 254] => Some(Answer::Refused(
|
||||
"Spamhaus doesn't answer public resolvers; use the server's own",
|
||||
)),
|
||||
[127, 255, 255, 255] => Some(Answer::Refused("Too many queries from this resolver")),
|
||||
_ => None,
|
||||
}
|
||||
}
|
||||
|
||||
fn zen(ip: Ipv4Addr) -> Answer {
|
||||
if let Some(refused) = spamhaus_refusal(ip) {
|
||||
return refused;
|
||||
}
|
||||
match ip.octets() {
|
||||
[127, 0, 0, 2] => Answer::Listed("SBL: a known spam source"),
|
||||
[127, 0, 0, 3] => Answer::Listed("CSS: sent spam recently"),
|
||||
[127, 0, 0, 4..=7] => Answer::Listed("XBL: a compromised or infected host"),
|
||||
[127, 0, 0, 9] => Answer::Listed("DROP: a hijacked or criminal network"),
|
||||
[127, 0, 0, 10 | 11] => {
|
||||
Answer::Listed("PBL: an address that isn't meant to send mail directly")
|
||||
}
|
||||
_ => Answer::Unknown,
|
||||
}
|
||||
}
|
||||
|
||||
fn dbl(ip: Ipv4Addr) -> Answer {
|
||||
if let Some(refused) = spamhaus_refusal(ip) {
|
||||
return refused;
|
||||
}
|
||||
match ip.octets() {
|
||||
[127, 0, 1, 2] => Answer::Listed("A spam domain"),
|
||||
[127, 0, 1, 4] => Answer::Listed("A phishing domain"),
|
||||
[127, 0, 1, 5] => Answer::Listed("A malware domain"),
|
||||
[127, 0, 1, 6] => Answer::Listed("A botnet controller"),
|
||||
[127, 0, 1, 102..=106] => Answer::Listed("A legitimate domain being abused"),
|
||||
[127, 0, 1, 255] => Answer::Refused("The query was malformed"),
|
||||
_ => Answer::Unknown,
|
||||
}
|
||||
}
|
||||
|
||||
/// Most lists answer 127.0.0.2 for "listed" and define nothing else.
|
||||
fn just_two(ip: Ipv4Addr) -> Answer {
|
||||
match ip.octets() {
|
||||
[127, 0, 0, 2] => Answer::Listed("Listed"),
|
||||
_ => Answer::Unknown,
|
||||
}
|
||||
}
|
||||
|
||||
fn surbl(ip: Ipv4Addr) -> Answer {
|
||||
match ip.octets() {
|
||||
[127, 0, 0, 1] => Answer::Refused("SURBL doesn't answer this resolver"),
|
||||
[127, 0, 0, bits] if bits & (8 | 16 | 64 | 128) != 0 => {
|
||||
Answer::Listed("Seen in phishing, malware, abuse or cracked sites")
|
||||
}
|
||||
_ => Answer::Unknown,
|
||||
}
|
||||
}
|
||||
|
||||
fn uribl(ip: Ipv4Addr) -> Answer {
|
||||
match ip.octets() {
|
||||
[127, 0, 0, 1] => Answer::Refused("URIBL doesn't answer public resolvers"),
|
||||
[127, 0, 0, bits] if bits & (2 | 8) != 0 => Answer::Listed("Seen in spam"),
|
||||
[127, 0, 0, bits] if bits & 4 != 0 => {
|
||||
Answer::Listed("Grey: seen in bulk mail some people don't want")
|
||||
}
|
||||
_ => Answer::Unknown,
|
||||
}
|
||||
}
|
||||
|
||||
pub const LISTS: &[BlockList] = &[
|
||||
BlockList {
|
||||
name: "Spamhaus ZEN",
|
||||
zone: "zen.spamhaus.org",
|
||||
scope: Scope::Ip,
|
||||
lookup: "https://check.spamhaus.org/",
|
||||
note: None,
|
||||
read: zen,
|
||||
},
|
||||
BlockList {
|
||||
name: "SpamCop",
|
||||
zone: "bl.spamcop.net",
|
||||
scope: Scope::Ip,
|
||||
lookup: "https://www.spamcop.net/bl.shtml",
|
||||
note: None,
|
||||
read: just_two,
|
||||
},
|
||||
BlockList {
|
||||
name: "Barracuda",
|
||||
zone: "b.barracudacentral.org",
|
||||
scope: Scope::Ip,
|
||||
lookup: "https://www.barracudacentral.org/lookups",
|
||||
note: Some(
|
||||
"Barracuda answers only resolvers whose address is registered with it (free, at barracudacentral.org/rbl). Until then its lookups can't be checked.",
|
||||
),
|
||||
read: just_two,
|
||||
},
|
||||
BlockList {
|
||||
name: "UCEPROTECT level 1",
|
||||
zone: "dnsbl-1.uceprotect.net",
|
||||
scope: Scope::Ip,
|
||||
lookup: "https://www.uceprotect.net/en/rblcheck.php",
|
||||
note: None,
|
||||
read: just_two,
|
||||
},
|
||||
BlockList {
|
||||
name: "Mailspike",
|
||||
zone: "bl.mailspike.net",
|
||||
scope: Scope::Ip,
|
||||
lookup: "https://mailspike.org/iplookup.html",
|
||||
note: None,
|
||||
read: just_two,
|
||||
},
|
||||
BlockList {
|
||||
name: "PSBL",
|
||||
zone: "psbl.surriel.com",
|
||||
scope: Scope::Ip,
|
||||
lookup: "https://psbl.org/",
|
||||
note: None,
|
||||
read: just_two,
|
||||
},
|
||||
BlockList {
|
||||
name: "Spamhaus DBL",
|
||||
zone: "dbl.spamhaus.org",
|
||||
scope: Scope::Domain,
|
||||
lookup: "https://check.spamhaus.org/",
|
||||
note: None,
|
||||
read: dbl,
|
||||
},
|
||||
BlockList {
|
||||
name: "SURBL",
|
||||
zone: "multi.surbl.org",
|
||||
scope: Scope::Domain,
|
||||
lookup: "https://surbl.org/surbl-analysis",
|
||||
note: None,
|
||||
read: surbl,
|
||||
},
|
||||
BlockList {
|
||||
name: "URIBL",
|
||||
zone: "multi.uribl.com",
|
||||
scope: Scope::Domain,
|
||||
lookup: "https://admin.uribl.com/",
|
||||
note: None,
|
||||
read: uribl,
|
||||
},
|
||||
];
|
||||
|
||||
pub fn by_name(name: &str) -> Option<&'static BlockList> {
|
||||
LISTS.iter().find(|list| list.name == name)
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::*;
|
||||
|
||||
fn ip(s: &str) -> Ipv4Addr {
|
||||
s.parse().unwrap()
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn a_refusal_is_not_a_listing() {
|
||||
let zen = by_name("Spamhaus ZEN").unwrap();
|
||||
assert!(matches!(
|
||||
zen.read(ip("127.255.255.254")),
|
||||
Answer::Refused(_)
|
||||
));
|
||||
assert!(matches!(zen.read(ip("127.0.0.2")), Answer::Listed(_)));
|
||||
assert!(matches!(zen.read(ip("127.0.0.10")), Answer::Listed(_)));
|
||||
assert_eq!(zen.read(ip("127.0.0.200")), Answer::Unknown);
|
||||
|
||||
let uribl = by_name("URIBL").unwrap();
|
||||
assert!(matches!(uribl.read(ip("127.0.0.1")), Answer::Refused(_)));
|
||||
assert!(matches!(uribl.read(ip("127.0.0.2")), Answer::Listed(_)));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn queries_are_built_per_scope() {
|
||||
let zen = by_name("Spamhaus ZEN").unwrap();
|
||||
let dbl = by_name("Spamhaus DBL").unwrap();
|
||||
let v4 = Subject::Ip("192.0.2.10".parse().unwrap());
|
||||
let v6 = Subject::Ip("2001:db8::1".parse().unwrap());
|
||||
let domain = Subject::Domain("example.org");
|
||||
assert_eq!(
|
||||
zen.query(&v4).as_deref(),
|
||||
Some("10.2.0.192.zen.spamhaus.org.")
|
||||
);
|
||||
assert_eq!(zen.query(&v6), None);
|
||||
assert_eq!(zen.query(&domain), None);
|
||||
assert_eq!(
|
||||
dbl.query(&domain).as_deref(),
|
||||
Some("example.org.dbl.spamhaus.org.")
|
||||
);
|
||||
assert_eq!(dbl.query(&v4), None);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn names_are_unique() {
|
||||
for (i, a) in LISTS.iter().enumerate() {
|
||||
assert!(
|
||||
LISTS[i + 1..].iter().all(|b| b.name != a.name),
|
||||
"{}",
|
||||
a.name
|
||||
);
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,410 @@
|
||||
/*
|
||||
* SPDX-FileCopyrightText: 2026 Coffey Labs LLC
|
||||
*
|
||||
* SPDX-License-Identifier: AGPL-3.0-only
|
||||
*/
|
||||
|
||||
//! The deliverability check (deliverability spec): what other mail servers
|
||||
//! see when this one sends. Not a rebuild of anything upstream ships.
|
||||
//!
|
||||
//! Every node that sends mail checks itself, because only it knows which
|
||||
//! address it leaves from, and keeps one report. The report holds facts: an
|
||||
//! address's reverse DNS, what each blocklist answered, what SPF said for
|
||||
//! each address, whether a DKIM key in DNS matches the one signing. The
|
||||
//! console grades them, so its wording can change without a server release.
|
||||
//!
|
||||
//! Kept in the fork's subspace (`store::SUBSPACE_INBUXA`). Every key starts
|
||||
//! with `D`, then one byte for the kind:
|
||||
//!
|
||||
//! - `r` + node id (u64): that node's last report, as JSON.
|
||||
//! - `s`: the settings, as JSON.
|
||||
//!
|
||||
//! Numbers are big-endian.
|
||||
|
||||
pub mod lists;
|
||||
|
||||
use serde::{Deserialize as SerdeDeserialize, Serialize as SerdeSerialize};
|
||||
use store::{
|
||||
Deserialize, IterateParams, SUBSPACE_INBUXA, Serialize, Store, ValueKey,
|
||||
write::{AnyClass, BatchBuilder, ValueClass},
|
||||
};
|
||||
use trc::AddContext;
|
||||
|
||||
const FEATURE: u8 = b'D';
|
||||
const KIND_REPORT: u8 = b'r';
|
||||
const KIND_SETTINGS: u8 = b's';
|
||||
|
||||
/// DL-15: **Check now** runs a node again only this long after its last run.
|
||||
pub const MIN_INTERVAL_SECS: u64 = 600;
|
||||
|
||||
#[derive(Debug, Clone, Default, PartialEq, SerdeSerialize, SerdeDeserialize)]
|
||||
#[serde(rename_all = "camelCase", default)]
|
||||
pub struct Report {
|
||||
/// The node's cluster id, as metric samples carry it.
|
||||
pub node_id: u64,
|
||||
pub hostname: String,
|
||||
/// Seconds since the epoch.
|
||||
pub checked_at: u64,
|
||||
pub addresses: Vec<Address>,
|
||||
pub domains: Vec<DomainReport>,
|
||||
pub certificates: Vec<Certificate>,
|
||||
}
|
||||
|
||||
#[derive(Debug, Clone, Default, PartialEq, SerdeSerialize, SerdeDeserialize)]
|
||||
#[serde(rename_all = "camelCase", default)]
|
||||
pub struct Address {
|
||||
pub ip: String,
|
||||
/// DL-2: how the node came by the address.
|
||||
pub source: AddressSource,
|
||||
/// The connection strategy that sends from it.
|
||||
pub strategy: String,
|
||||
/// The name the node greets with from this address.
|
||||
pub ehlo: String,
|
||||
/// The PTR names, empty when there's none.
|
||||
pub ptr: Vec<String>,
|
||||
/// Some PTR name resolves back to the address.
|
||||
pub forward_confirmed: bool,
|
||||
/// The forward-confirmed name is the EHLO name.
|
||||
pub ehlo_matches: bool,
|
||||
/// Set when the reverse lookup itself failed, rather than found nothing.
|
||||
pub ptr_error: Option<String>,
|
||||
pub listings: Vec<Listing>,
|
||||
}
|
||||
|
||||
#[derive(Debug, Clone, Copy, Default, PartialEq, Eq, SerdeSerialize, SerdeDeserialize)]
|
||||
#[serde(rename_all = "camelCase")]
|
||||
pub enum AddressSource {
|
||||
/// Set in the connection strategy's source addresses.
|
||||
#[default]
|
||||
Configured,
|
||||
/// What the EHLO name resolves to.
|
||||
Ehlo,
|
||||
}
|
||||
|
||||
#[derive(Debug, Clone, Default, PartialEq, SerdeSerialize, SerdeDeserialize)]
|
||||
#[serde(rename_all = "camelCase", default)]
|
||||
pub struct Listing {
|
||||
/// The list's name, as in [`lists::LISTS`].
|
||||
pub list: String,
|
||||
pub state: ListingState,
|
||||
/// The address the list answered, when it answered one.
|
||||
pub code: Option<String>,
|
||||
/// What the list says the answer means.
|
||||
pub meaning: Option<String>,
|
||||
}
|
||||
|
||||
#[derive(Debug, Clone, Copy, Default, PartialEq, Eq, SerdeSerialize, SerdeDeserialize)]
|
||||
#[serde(rename_all = "camelCase")]
|
||||
pub enum ListingState {
|
||||
#[default]
|
||||
Clean,
|
||||
Listed,
|
||||
/// The list wouldn't answer, or the lookup failed: neither listed nor clean.
|
||||
Refused,
|
||||
Error,
|
||||
/// Switched off in the settings, so not asked.
|
||||
Off,
|
||||
}
|
||||
|
||||
#[derive(Debug, Clone, Default, PartialEq, SerdeSerialize, SerdeDeserialize)]
|
||||
#[serde(rename_all = "camelCase", default)]
|
||||
pub struct DomainReport {
|
||||
pub domain: String,
|
||||
/// DL-20: a tenant administrator sees only their tenant's domains.
|
||||
pub tenant_id: Option<u32>,
|
||||
/// DL-7: what SPF says for each of the node's addresses.
|
||||
pub spf: Vec<SpfResult>,
|
||||
/// DL-8: each DKIM key the domain signs with.
|
||||
pub dkim: Vec<DkimKey>,
|
||||
/// DL-9: the DMARC record, if there's one.
|
||||
pub dmarc: Option<Dmarc>,
|
||||
/// DL-10.
|
||||
pub mta_sts: MtaSts,
|
||||
/// DL-11: there's a `_smtp._tls` record.
|
||||
pub tls_rpt: bool,
|
||||
/// DL-12.
|
||||
pub listings: Vec<Listing>,
|
||||
}
|
||||
|
||||
#[derive(Debug, Clone, Default, PartialEq, SerdeSerialize, SerdeDeserialize)]
|
||||
#[serde(rename_all = "camelCase", default)]
|
||||
pub struct SpfResult {
|
||||
pub ip: String,
|
||||
/// `pass`, `fail`, `softFail`, `neutral`, `none`, `tempError` or `permError`.
|
||||
pub result: String,
|
||||
}
|
||||
|
||||
#[derive(Debug, Clone, Default, PartialEq, SerdeSerialize, SerdeDeserialize)]
|
||||
#[serde(rename_all = "camelCase", default)]
|
||||
pub struct DkimKey {
|
||||
pub selector: String,
|
||||
pub state: DkimState,
|
||||
}
|
||||
|
||||
#[derive(Debug, Clone, Copy, Default, PartialEq, Eq, SerdeSerialize, SerdeDeserialize)]
|
||||
#[serde(rename_all = "camelCase")]
|
||||
pub enum DkimState {
|
||||
#[default]
|
||||
Matches,
|
||||
/// Nothing published at `<selector>._domainkey.<domain>`.
|
||||
Missing,
|
||||
/// Published, but a different key.
|
||||
Different,
|
||||
/// The lookup failed.
|
||||
Error,
|
||||
}
|
||||
|
||||
#[derive(Debug, Clone, Default, PartialEq, SerdeSerialize, SerdeDeserialize)]
|
||||
#[serde(rename_all = "camelCase", default)]
|
||||
pub struct Dmarc {
|
||||
/// `none`, `quarantine` or `reject`.
|
||||
pub policy: String,
|
||||
/// DKIM alignment: `relaxed` or `strict`.
|
||||
pub adkim: String,
|
||||
/// SPF alignment: `relaxed` or `strict`.
|
||||
pub aspf: String,
|
||||
}
|
||||
|
||||
#[derive(Debug, Clone, Default, PartialEq, SerdeSerialize, SerdeDeserialize)]
|
||||
#[serde(rename_all = "camelCase", default)]
|
||||
pub struct MtaSts {
|
||||
/// The `_mta-sts` record's id; None when there's no record.
|
||||
pub record_id: Option<String>,
|
||||
/// The policy was fetched and parsed. False with a record means the
|
||||
/// fetch or the parse failed, and `error` says why.
|
||||
pub fetched: bool,
|
||||
pub error: Option<String>,
|
||||
/// `enforce`, `testing` or `none`.
|
||||
pub mode: Option<String>,
|
||||
pub max_age: Option<u64>,
|
||||
/// The domain's MX names no `mx:` line matches.
|
||||
pub mx_not_covered: Vec<String>,
|
||||
}
|
||||
|
||||
#[derive(Debug, Clone, Default, PartialEq, SerdeSerialize, SerdeDeserialize)]
|
||||
#[serde(rename_all = "camelCase", default)]
|
||||
pub struct Certificate {
|
||||
/// The EHLO name, or an MX name that points at this node.
|
||||
pub name: String,
|
||||
/// The node holds a certificate for the name.
|
||||
pub covered: bool,
|
||||
}
|
||||
|
||||
#[derive(Debug, Clone, Default, PartialEq, SerdeSerialize, SerdeDeserialize)]
|
||||
#[serde(rename_all = "camelCase", default)]
|
||||
pub struct Settings {
|
||||
/// DL-6: lists not to ask, by name.
|
||||
pub disabled_lists: Vec<String>,
|
||||
}
|
||||
|
||||
impl Settings {
|
||||
pub fn is_off(&self, list: &str) -> bool {
|
||||
self.disabled_lists.iter().any(|name| name == list)
|
||||
}
|
||||
|
||||
/// Only the built-in lists' names, once each.
|
||||
pub fn validate(&self) -> Result<(), String> {
|
||||
for (i, name) in self.disabled_lists.iter().enumerate() {
|
||||
if lists::by_name(name).is_none() {
|
||||
return Err(format!("There's no list called {name:?}."));
|
||||
}
|
||||
if self.disabled_lists[..i].contains(name) {
|
||||
return Err(format!("{name:?} is named twice."));
|
||||
}
|
||||
}
|
||||
Ok(())
|
||||
}
|
||||
}
|
||||
|
||||
impl Report {
|
||||
/// DL-20: what a tenant administrator may see: their tenant's domains
|
||||
/// and nothing about the node's addresses or certificates.
|
||||
pub fn for_tenant(&self, tenant_id: u32) -> Report {
|
||||
Report {
|
||||
node_id: self.node_id,
|
||||
hostname: self.hostname.clone(),
|
||||
checked_at: self.checked_at,
|
||||
addresses: Vec::new(),
|
||||
domains: self
|
||||
.domains
|
||||
.iter()
|
||||
.filter(|d| d.tenant_id == Some(tenant_id))
|
||||
.cloned()
|
||||
.collect(),
|
||||
certificates: Vec::new(),
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// --- Storage --------------------------------------------------------------
|
||||
|
||||
struct Json<T>(T);
|
||||
|
||||
impl<T: SerdeSerialize> Serialize for Json<T> {
|
||||
fn serialize(&self) -> trc::Result<Vec<u8>> {
|
||||
serde_json::to_vec(&self.0).map_err(|err| {
|
||||
trc::StoreEvent::UnexpectedError
|
||||
.into_err()
|
||||
.details("Failed to serialize deliverability data")
|
||||
.reason(err)
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
impl<T: for<'de> SerdeDeserialize<'de> + Send + Sync> Deserialize for Json<T> {
|
||||
fn deserialize(bytes: &[u8]) -> trc::Result<Self> {
|
||||
serde_json::from_slice(bytes).map(Json).map_err(|err| {
|
||||
trc::StoreEvent::DataCorruption
|
||||
.into_err()
|
||||
.details("Invalid deliverability data")
|
||||
.reason(err)
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
fn class(kind: u8, node_id: Option<u64>) -> ValueClass {
|
||||
let mut key = Vec::with_capacity(10);
|
||||
key.push(FEATURE);
|
||||
key.push(kind);
|
||||
if let Some(node_id) = node_id {
|
||||
key.extend_from_slice(&node_id.to_be_bytes());
|
||||
}
|
||||
ValueClass::Any(AnyClass {
|
||||
subspace: SUBSPACE_INBUXA,
|
||||
key,
|
||||
})
|
||||
}
|
||||
|
||||
pub async fn report(data: &Store, node_id: u64) -> trc::Result<Option<Report>> {
|
||||
Ok(data
|
||||
.get_value::<Json<Report>>(ValueKey::from(class(KIND_REPORT, Some(node_id))))
|
||||
.await
|
||||
.caused_by(trc::location!())?
|
||||
.map(|Json(report)| report))
|
||||
}
|
||||
|
||||
/// Every node's report, by node id.
|
||||
pub async fn reports(data: &Store) -> trc::Result<Vec<Report>> {
|
||||
let mut out = Vec::new();
|
||||
data.iterate(
|
||||
IterateParams::new(
|
||||
ValueKey::from(class(KIND_REPORT, Some(0))),
|
||||
ValueKey::from(class(KIND_REPORT, Some(u64::MAX))),
|
||||
),
|
||||
|_, value| {
|
||||
if let Ok(Json(report)) = Json::<Report>::deserialize(value) {
|
||||
out.push(report);
|
||||
}
|
||||
Ok(true)
|
||||
},
|
||||
)
|
||||
.await
|
||||
.caused_by(trc::location!())?;
|
||||
out.sort_by_key(|r| r.node_id);
|
||||
Ok(out)
|
||||
}
|
||||
|
||||
/// Replaces the node's report.
|
||||
pub async fn put_report(data: &Store, report: &Report) -> trc::Result<()> {
|
||||
let mut batch = BatchBuilder::new();
|
||||
batch.set(
|
||||
class(KIND_REPORT, Some(report.node_id)),
|
||||
Json(report).serialize()?,
|
||||
);
|
||||
data.write(batch.build_all())
|
||||
.await
|
||||
.caused_by(trc::location!())?;
|
||||
Ok(())
|
||||
}
|
||||
|
||||
pub async fn settings(data: &Store) -> trc::Result<Settings> {
|
||||
Ok(data
|
||||
.get_value::<Json<Settings>>(ValueKey::from(class(KIND_SETTINGS, None)))
|
||||
.await
|
||||
.caused_by(trc::location!())?
|
||||
.map(|Json(settings)| settings)
|
||||
.unwrap_or_default())
|
||||
}
|
||||
|
||||
pub async fn put_settings(data: &Store, settings: &Settings) -> trc::Result<()> {
|
||||
let mut batch = BatchBuilder::new();
|
||||
batch.set(class(KIND_SETTINGS, None), Json(settings).serialize()?);
|
||||
data.write(batch.build_all())
|
||||
.await
|
||||
.caused_by(trc::location!())?;
|
||||
Ok(())
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::*;
|
||||
|
||||
#[test]
|
||||
fn settings_name_only_built_in_lists_once() {
|
||||
let ok = Settings {
|
||||
disabled_lists: vec!["Barracuda".into(), "URIBL".into()],
|
||||
};
|
||||
assert!(ok.validate().is_ok());
|
||||
assert!(ok.is_off("Barracuda"));
|
||||
assert!(!ok.is_off("SpamCop"));
|
||||
let unknown = Settings {
|
||||
disabled_lists: vec!["My list".into()],
|
||||
};
|
||||
assert!(unknown.validate().is_err());
|
||||
let twice = Settings {
|
||||
disabled_lists: vec!["URIBL".into(), "URIBL".into()],
|
||||
};
|
||||
assert!(twice.validate().is_err());
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn a_tenant_sees_only_its_domains() {
|
||||
let report = Report {
|
||||
node_id: 2,
|
||||
hostname: "mx2.example.org".into(),
|
||||
checked_at: 1,
|
||||
addresses: vec![Address {
|
||||
ip: "192.0.2.10".into(),
|
||||
..Default::default()
|
||||
}],
|
||||
domains: vec![
|
||||
DomainReport {
|
||||
domain: "a.example".into(),
|
||||
tenant_id: Some(7),
|
||||
..Default::default()
|
||||
},
|
||||
DomainReport {
|
||||
domain: "b.example".into(),
|
||||
tenant_id: Some(8),
|
||||
..Default::default()
|
||||
},
|
||||
DomainReport {
|
||||
domain: "server.example".into(),
|
||||
tenant_id: None,
|
||||
..Default::default()
|
||||
},
|
||||
],
|
||||
certificates: vec![Certificate {
|
||||
name: "mx2.example.org".into(),
|
||||
covered: true,
|
||||
}],
|
||||
};
|
||||
let seen = report.for_tenant(7);
|
||||
assert!(seen.addresses.is_empty());
|
||||
assert!(seen.certificates.is_empty());
|
||||
assert_eq!(
|
||||
seen.domains
|
||||
.iter()
|
||||
.map(|d| d.domain.as_str())
|
||||
.collect::<Vec<_>>(),
|
||||
["a.example"]
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn a_report_reads_back_with_missing_fields() {
|
||||
let report: Report = serde_json::from_str(r#"{"nodeId": 3}"#).unwrap();
|
||||
assert_eq!(report.node_id, 3);
|
||||
assert!(report.domains.is_empty());
|
||||
}
|
||||
}
|
||||
@@ -1,5 +1,5 @@
|
||||
/*
|
||||
* SPDX-FileCopyrightText: 2026 Coffey Labs
|
||||
* SPDX-FileCopyrightText: 2026 Coffey Labs LLC
|
||||
*
|
||||
* SPDX-License-Identifier: AGPL-3.0-only
|
||||
*/
|
||||
|
||||
@@ -0,0 +1,120 @@
|
||||
/*
|
||||
* SPDX-FileCopyrightText: 2026 Coffey Labs LLC
|
||||
*
|
||||
* SPDX-License-Identifier: AGPL-3.0-only
|
||||
*/
|
||||
|
||||
//! Reports on their way to an outside archive (JR-7). Keys, after `J`:
|
||||
//!
|
||||
//! - `o` + the report's queue id: what goes into the built-in journal if
|
||||
//! the archive never takes the report, as JSON. Cleared once it's
|
||||
//! delivered or kept.
|
||||
//! - `w` + journal id (u32): how often that journal's archive didn't take a
|
||||
//! report, and the last time and reason, for the console's warning.
|
||||
|
||||
use super::{FEATURE, Json, entries::Entry};
|
||||
use serde::{Deserialize as SerdeDeserialize, Serialize as SerdeSerialize};
|
||||
use store::{
|
||||
SUBSPACE_INBUXA, Serialize, Store, ValueKey,
|
||||
write::{AnyClass, BatchBuilder, ValueClass},
|
||||
};
|
||||
use trc::AddContext;
|
||||
|
||||
const KIND_PENDING: u8 = b'o';
|
||||
const KIND_FAILURES: u8 = b'w';
|
||||
|
||||
/// A report queued to an archive.
|
||||
#[derive(Debug, Clone, PartialEq, Eq, SerdeSerialize, SerdeDeserialize)]
|
||||
#[serde(rename_all = "camelCase")]
|
||||
pub struct Pending {
|
||||
pub address: String,
|
||||
/// The entry, should the archive not take it: its own, with the
|
||||
/// sending journals' retention, whatever else the built-in journal has.
|
||||
pub entry: Entry,
|
||||
}
|
||||
|
||||
/// How a journal's archive has been taking its reports.
|
||||
#[derive(Debug, Clone, Default, PartialEq, Eq, SerdeSerialize, SerdeDeserialize)]
|
||||
#[serde(rename_all = "camelCase")]
|
||||
pub struct Failures {
|
||||
pub count: u64,
|
||||
/// Seconds.
|
||||
pub last_at: u64,
|
||||
pub last_reason: String,
|
||||
}
|
||||
|
||||
fn class(kind: u8, id: &[u8]) -> ValueClass {
|
||||
let mut key = Vec::with_capacity(2 + id.len());
|
||||
key.push(FEATURE);
|
||||
key.push(kind);
|
||||
key.extend_from_slice(id);
|
||||
ValueClass::Any(AnyClass {
|
||||
subspace: SUBSPACE_INBUXA,
|
||||
key,
|
||||
})
|
||||
}
|
||||
|
||||
pub async fn set_pending(data: &Store, queue_id: u64, pending: &Pending) -> trc::Result<()> {
|
||||
let mut batch = BatchBuilder::new();
|
||||
batch.set(
|
||||
class(KIND_PENDING, &queue_id.to_be_bytes()),
|
||||
Json(pending).serialize()?,
|
||||
);
|
||||
data.write(batch.build_all())
|
||||
.await
|
||||
.caused_by(trc::location!())?;
|
||||
Ok(())
|
||||
}
|
||||
|
||||
pub async fn pending(data: &Store, queue_id: u64) -> trc::Result<Option<Pending>> {
|
||||
Ok(data
|
||||
.get_value::<Json<Pending>>(ValueKey::from(class(KIND_PENDING, &queue_id.to_be_bytes())))
|
||||
.await
|
||||
.caused_by(trc::location!())?
|
||||
.map(|Json(pending)| pending))
|
||||
}
|
||||
|
||||
pub async fn clear_pending(data: &Store, queue_id: u64) -> trc::Result<()> {
|
||||
let mut batch = BatchBuilder::new();
|
||||
batch.clear(class(KIND_PENDING, &queue_id.to_be_bytes()));
|
||||
data.write(batch.build_all())
|
||||
.await
|
||||
.caused_by(trc::location!())?;
|
||||
Ok(())
|
||||
}
|
||||
|
||||
pub async fn failures(data: &Store, journal_id: u32) -> trc::Result<Failures> {
|
||||
Ok(data
|
||||
.get_value::<Json<Failures>>(ValueKey::from(class(
|
||||
KIND_FAILURES,
|
||||
&journal_id.to_be_bytes(),
|
||||
)))
|
||||
.await
|
||||
.caused_by(trc::location!())?
|
||||
.map(|Json(failures)| failures)
|
||||
.unwrap_or_default())
|
||||
}
|
||||
|
||||
/// Counts one report an archive didn't take, for each of `journals`.
|
||||
pub async fn record_failure(
|
||||
data: &Store,
|
||||
journals: &[u32],
|
||||
at: u64,
|
||||
reason: &str,
|
||||
) -> trc::Result<()> {
|
||||
for journal_id in journals {
|
||||
let mut failures = failures(data, *journal_id).await?;
|
||||
failures.count += 1;
|
||||
failures.last_at = at;
|
||||
failures.last_reason = reason.chars().take(500).collect();
|
||||
let mut batch = BatchBuilder::new();
|
||||
batch.set(
|
||||
class(KIND_FAILURES, &journal_id.to_be_bytes()),
|
||||
Json(&failures).serialize()?,
|
||||
);
|
||||
data.write(batch.build_all())
|
||||
.await
|
||||
.caused_by(trc::location!())?;
|
||||
}
|
||||
Ok(())
|
||||
}
|
||||
@@ -0,0 +1,869 @@
|
||||
/*
|
||||
* SPDX-FileCopyrightText: 2026 Coffey Labs LLC
|
||||
*
|
||||
* SPDX-License-Identifier: AGPL-3.0-only
|
||||
*/
|
||||
|
||||
//! The built-in journal (JR-5, JR-6, JR-13). Keys, after `J`:
|
||||
//!
|
||||
//! - `e` + node + seq: a chain link: its seq, the hash of the link before
|
||||
//! it, and the SHA-256 of its entry. One chain per node, as the audit log
|
||||
//! keeps (AU-6), but a link names its entry by hash instead of holding it,
|
||||
//! so an entry can go at the end of its own retention without breaking
|
||||
//! the chain: entries don't expire in chain order.
|
||||
//! - `c` + node + seq: the entry, as JSON; its bytes are what the link's
|
||||
//! hash names.
|
||||
//! - `p` + node + seq: when an entry past its retention was purged. A link
|
||||
//! whose entry is gone without this marker is a broken chain.
|
||||
//! - `t` + time + node + seq: the time index, for search.
|
||||
//! - `x` + expiry + node + seq: the expiry index, for purge.
|
||||
//! - `h` + node: the chain's head: its hash, then its seq as the last eight
|
||||
//! bytes, which each append asserts.
|
||||
//! - `f` + node: where the chain starts after purged links at its start
|
||||
//! were cleared, and the hash the first kept link names.
|
||||
//!
|
||||
//! The report itself is a blob, kept by a temporary link that lasts until
|
||||
//! its entry is purged. Nothing here changes or removes an entry before
|
||||
//! its time; nothing in JMAP can.
|
||||
|
||||
use super::{Direction, FEATURE, Json};
|
||||
use crate::hold::HELD_UNTIL;
|
||||
use serde::{Deserialize as SerdeDeserialize, Serialize as SerdeSerialize};
|
||||
use sha2::{Digest, Sha256};
|
||||
use std::fmt;
|
||||
use store::{
|
||||
BlobStore, Deserialize, IterateParams, SUBSPACE_INBUXA, Serialize, Store, ValueKey,
|
||||
write::{AnyClass, BatchBuilder, BlobLink, BlobOp, ValueClass, assert::AssertValue},
|
||||
};
|
||||
use tokio::sync::Mutex;
|
||||
use trc::AddContext;
|
||||
use types::blob_hash::BlobHash;
|
||||
|
||||
const KIND_LINK: u8 = b'e';
|
||||
const KIND_CONTENT: u8 = b'c';
|
||||
const KIND_PURGED: u8 = b'p';
|
||||
const KIND_TIME: u8 = b't';
|
||||
const KIND_EXPIRY: u8 = b'x';
|
||||
const KIND_HEAD: u8 = b'h';
|
||||
const KIND_FLOOR: u8 = b'f';
|
||||
|
||||
const APPEND_ATTEMPTS: usize = 5;
|
||||
/// Entries purged per batch.
|
||||
const PURGE_BATCH: usize = 100;
|
||||
|
||||
/// Where one entry sits: its node's chain and its place in it.
|
||||
#[derive(Debug, Clone, Copy, PartialEq, Eq, Hash, PartialOrd, Ord)]
|
||||
pub struct EntryId {
|
||||
pub node: u64,
|
||||
pub seq: u64,
|
||||
}
|
||||
|
||||
impl EntryId {
|
||||
/// As one number, for JMAP ids: the node in the top 16 bits.
|
||||
pub fn to_u64(&self) -> u64 {
|
||||
(self.node << 48) | (self.seq & ((1 << 48) - 1))
|
||||
}
|
||||
|
||||
pub fn from_u64(id: u64) -> Self {
|
||||
EntryId {
|
||||
node: id >> 48,
|
||||
seq: id & ((1 << 48) - 1),
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
impl fmt::Display for EntryId {
|
||||
fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result {
|
||||
write!(f, "{}-{}", self.node, self.seq)
|
||||
}
|
||||
}
|
||||
|
||||
/// One journaled message (JR-5).
|
||||
#[derive(Debug, Clone, PartialEq, Eq, SerdeSerialize, SerdeDeserialize)]
|
||||
#[serde(rename_all = "camelCase")]
|
||||
pub struct Entry {
|
||||
pub queue_id: u64,
|
||||
/// Seconds.
|
||||
pub at: u64,
|
||||
pub direction: Direction,
|
||||
pub sender: String,
|
||||
pub authenticated: bool,
|
||||
pub recipients: Vec<String>,
|
||||
pub subject: String,
|
||||
pub message_id: String,
|
||||
/// The people here on either side, whose holds keep the entry.
|
||||
pub accounts: Vec<u32>,
|
||||
pub tenants: Vec<u32>,
|
||||
/// The journals that took it.
|
||||
pub journals: Vec<u32>,
|
||||
pub held: bool,
|
||||
/// The report's blob, hex.
|
||||
pub blob: String,
|
||||
pub size: u64,
|
||||
/// SHA-256 of the report, hex.
|
||||
pub sha256: String,
|
||||
/// Seconds.
|
||||
pub expires_at: u64,
|
||||
}
|
||||
|
||||
impl Entry {
|
||||
pub fn blob_hash(&self) -> Option<BlobHash> {
|
||||
let bytes = unhex(&self.blob)?;
|
||||
BlobHash::try_from_hash_slice(&bytes).ok()
|
||||
}
|
||||
}
|
||||
|
||||
#[derive(Debug, Clone, SerdeSerialize, SerdeDeserialize)]
|
||||
#[serde(rename_all = "camelCase")]
|
||||
struct Link {
|
||||
seq: u64,
|
||||
prev: String,
|
||||
content: String,
|
||||
}
|
||||
|
||||
#[derive(Debug, Clone, Default, PartialEq, SerdeSerialize, SerdeDeserialize)]
|
||||
struct Floor {
|
||||
seq: u64,
|
||||
prev: String,
|
||||
}
|
||||
|
||||
#[derive(Debug, Clone, Default, PartialEq)]
|
||||
struct Head {
|
||||
seq: u64,
|
||||
hash: String,
|
||||
}
|
||||
|
||||
impl Head {
|
||||
fn to_bytes(&self) -> Vec<u8> {
|
||||
let mut bytes = self.hash.as_bytes().to_vec();
|
||||
bytes.extend_from_slice(&self.seq.to_be_bytes());
|
||||
bytes
|
||||
}
|
||||
}
|
||||
|
||||
impl Deserialize for Head {
|
||||
fn deserialize(bytes: &[u8]) -> trc::Result<Self> {
|
||||
let split = bytes.len().checked_sub(8).ok_or_else(|| {
|
||||
trc::StoreEvent::DataCorruption
|
||||
.into_err()
|
||||
.details("Invalid journal chain head")
|
||||
})?;
|
||||
Ok(Head {
|
||||
seq: u64::from_be_bytes(bytes[split..].try_into().unwrap()),
|
||||
hash: String::from_utf8_lossy(&bytes[..split]).into_owned(),
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
struct Raw(Vec<u8>);
|
||||
|
||||
impl Deserialize for Raw {
|
||||
fn deserialize(bytes: &[u8]) -> trc::Result<Self> {
|
||||
Ok(Raw(bytes.to_vec()))
|
||||
}
|
||||
}
|
||||
|
||||
fn class(kind: u8, parts: &[u64]) -> ValueClass {
|
||||
let mut key = Vec::with_capacity(2 + parts.len() * 8);
|
||||
key.push(FEATURE);
|
||||
key.push(kind);
|
||||
for part in parts {
|
||||
key.extend_from_slice(&part.to_be_bytes());
|
||||
}
|
||||
ValueClass::Any(AnyClass {
|
||||
subspace: SUBSPACE_INBUXA,
|
||||
key,
|
||||
})
|
||||
}
|
||||
|
||||
fn key(kind: u8, parts: &[u64]) -> ValueKey<ValueClass> {
|
||||
ValueKey::from(class(kind, parts))
|
||||
}
|
||||
|
||||
/// Where an entry's content is kept, for tests that check tampering shows.
|
||||
pub fn content_key(id: EntryId) -> ValueKey<ValueClass> {
|
||||
key(KIND_CONTENT, &[id.node, id.seq])
|
||||
}
|
||||
|
||||
/// The numbers after the kind byte, from the key's tail.
|
||||
fn parse_key(key: &[u8], kind: u8, parts: usize) -> Option<Vec<u64>> {
|
||||
let len = 2 + parts * 8;
|
||||
let tail = key.get(key.len().checked_sub(len)?..)?;
|
||||
(tail[0] == FEATURE && tail[1] == kind).then_some(())?;
|
||||
Some(
|
||||
tail[2..]
|
||||
.chunks_exact(8)
|
||||
.map(|chunk| u64::from_be_bytes(chunk.try_into().unwrap()))
|
||||
.collect(),
|
||||
)
|
||||
}
|
||||
|
||||
pub fn hex(bytes: &[u8]) -> String {
|
||||
bytes.iter().map(|b| format!("{b:02x}")).collect()
|
||||
}
|
||||
|
||||
fn unhex(value: &str) -> Option<Vec<u8>> {
|
||||
(value.len() % 2 == 0).then_some(())?;
|
||||
(0..value.len())
|
||||
.step_by(2)
|
||||
.map(|i| u8::from_str_radix(value.get(i..i + 2)?, 16).ok())
|
||||
.collect()
|
||||
}
|
||||
|
||||
pub fn sha256(bytes: &[u8]) -> String {
|
||||
hex(&Sha256::digest(bytes))
|
||||
}
|
||||
|
||||
async fn head(data: &Store, node: u64) -> trc::Result<Option<Head>> {
|
||||
data.get_value::<Head>(key(KIND_HEAD, &[node]))
|
||||
.await
|
||||
.caused_by(trc::location!())
|
||||
}
|
||||
|
||||
async fn floor(data: &Store, node: u64) -> trc::Result<Floor> {
|
||||
Ok(data
|
||||
.get_value::<Json<Floor>>(key(KIND_FLOOR, &[node]))
|
||||
.await
|
||||
.caused_by(trc::location!())?
|
||||
.map(|Json(floor)| floor)
|
||||
.unwrap_or(Floor {
|
||||
seq: 1,
|
||||
prev: String::new(),
|
||||
}))
|
||||
}
|
||||
|
||||
async fn nodes(data: &Store) -> trc::Result<Vec<u64>> {
|
||||
let mut nodes = Vec::new();
|
||||
data.iterate(
|
||||
IterateParams::new(key(KIND_HEAD, &[0]), key(KIND_HEAD, &[u64::MAX])).no_values(),
|
||||
|key, _| {
|
||||
if let Some(parts) = parse_key(key, KIND_HEAD, 1) {
|
||||
nodes.push(parts[0]);
|
||||
}
|
||||
Ok(true)
|
||||
},
|
||||
)
|
||||
.await
|
||||
.caused_by(trc::location!())?;
|
||||
Ok(nodes)
|
||||
}
|
||||
|
||||
/// Lines up this process's appends; the store's assert settles the rest.
|
||||
static APPENDING: Mutex<()> = Mutex::const_new(());
|
||||
|
||||
/// Adds an entry to this node's chain, and links its report's blob (already
|
||||
/// written) until the entry is purged. An error means nothing was written.
|
||||
pub async fn append(data: &Store, node: u64, entry: &Entry) -> trc::Result<EntryId> {
|
||||
let blob = entry.blob_hash().ok_or_else(|| {
|
||||
trc::StoreEvent::UnexpectedError
|
||||
.into_err()
|
||||
.details("Journal entry without a blob")
|
||||
})?;
|
||||
let content = Json(entry).serialize()?;
|
||||
let content_hash = sha256(&content);
|
||||
let _appending = APPENDING.lock().await;
|
||||
let mut attempt = 0;
|
||||
loop {
|
||||
attempt += 1;
|
||||
let current = head(data, node).await?;
|
||||
let (seq, prev) = current
|
||||
.as_ref()
|
||||
.map_or((1, String::new()), |head| (head.seq + 1, head.hash.clone()));
|
||||
let link = Json(&Link {
|
||||
seq,
|
||||
prev,
|
||||
content: content_hash.clone(),
|
||||
})
|
||||
.serialize()?;
|
||||
let new_head = Head {
|
||||
seq,
|
||||
hash: sha256(&link),
|
||||
};
|
||||
|
||||
let mut batch = BatchBuilder::new();
|
||||
batch.assert_value(
|
||||
class(KIND_HEAD, &[node]),
|
||||
current.map_or(AssertValue::None, |head| AssertValue::U64(head.seq)),
|
||||
);
|
||||
batch
|
||||
.set(class(KIND_LINK, &[node, seq]), link)
|
||||
.set(class(KIND_CONTENT, &[node, seq]), content.clone())
|
||||
.set(class(KIND_TIME, &[entry.at, node, seq]), vec![])
|
||||
.set(class(KIND_EXPIRY, &[entry.expires_at, node, seq]), vec![])
|
||||
.set(class(KIND_HEAD, &[node]), new_head.to_bytes())
|
||||
.set(
|
||||
BlobOp::Link {
|
||||
hash: blob.clone(),
|
||||
to: BlobLink::Temporary { until: HELD_UNTIL },
|
||||
},
|
||||
vec![],
|
||||
)
|
||||
.set(BlobOp::Commit { hash: blob.clone() }, vec![]);
|
||||
match data.write(batch.build_all()).await {
|
||||
Ok(_) => return Ok(EntryId { node, seq }),
|
||||
Err(err)
|
||||
if attempt < APPEND_ATTEMPTS
|
||||
&& matches!(
|
||||
err.as_ref(),
|
||||
trc::EventType::Store(trc::StoreEvent::AssertValueFailed)
|
||||
) => {}
|
||||
Err(err) => return Err(err.caused_by(trc::location!())),
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// One entry, unless it was purged.
|
||||
pub async fn get(data: &Store, id: EntryId) -> trc::Result<Option<Entry>> {
|
||||
Ok(data
|
||||
.get_value::<Json<Entry>>(key(KIND_CONTENT, &[id.node, id.seq]))
|
||||
.await
|
||||
.caused_by(trc::location!())?
|
||||
.map(|Json(entry)| entry))
|
||||
}
|
||||
|
||||
/// Entries written in `[after, before)` (seconds), newest first, up to
|
||||
/// `limit`.
|
||||
pub async fn list(
|
||||
data: &Store,
|
||||
after: u64,
|
||||
before: u64,
|
||||
limit: usize,
|
||||
) -> trc::Result<Vec<(EntryId, Entry)>> {
|
||||
let mut ids = Vec::new();
|
||||
data.iterate(
|
||||
IterateParams::new(
|
||||
key(KIND_TIME, &[after, 0, 0]),
|
||||
key(KIND_TIME, &[before.saturating_sub(1), u64::MAX, u64::MAX]),
|
||||
)
|
||||
.descending()
|
||||
.no_values(),
|
||||
|key, _| {
|
||||
if let Some(parts) = parse_key(key, KIND_TIME, 3) {
|
||||
ids.push(EntryId {
|
||||
node: parts[1],
|
||||
seq: parts[2],
|
||||
});
|
||||
}
|
||||
Ok(ids.len() < limit)
|
||||
},
|
||||
)
|
||||
.await
|
||||
.caused_by(trc::location!())?;
|
||||
let mut out = Vec::with_capacity(ids.len());
|
||||
for id in ids {
|
||||
if let Some(entry) = get(data, id).await? {
|
||||
out.push((id, entry));
|
||||
}
|
||||
}
|
||||
Ok(out)
|
||||
}
|
||||
|
||||
/// Most results one search page returns.
|
||||
pub const MAX_QUERY_LIMIT: usize = 500;
|
||||
|
||||
/// A search of the journal (JR-15): conditions that must all hold.
|
||||
#[derive(Debug, Clone, Default, PartialEq, Eq, SerdeSerialize)]
|
||||
#[serde(rename_all = "camelCase")]
|
||||
pub struct Filter {
|
||||
/// From this time on, in seconds.
|
||||
#[serde(skip_serializing_if = "Option::is_none")]
|
||||
pub after: Option<u64>,
|
||||
/// Before this time, in seconds.
|
||||
#[serde(skip_serializing_if = "Option::is_none")]
|
||||
pub before: Option<u64>,
|
||||
/// Part of the sender's address, ignoring case.
|
||||
#[serde(skip_serializing_if = "Option::is_none")]
|
||||
pub sender: Option<String>,
|
||||
/// Part of any recipient's address, ignoring case.
|
||||
#[serde(skip_serializing_if = "Option::is_none")]
|
||||
pub recipient: Option<String>,
|
||||
/// Part of the sender's or any recipient's address.
|
||||
#[serde(skip_serializing_if = "Option::is_none")]
|
||||
pub address: Option<String>,
|
||||
#[serde(skip_serializing_if = "Option::is_none")]
|
||||
pub direction: Option<Direction>,
|
||||
/// Words that must all appear in the subject, ignoring case.
|
||||
#[serde(skip_serializing_if = "Option::is_none")]
|
||||
pub text: Option<String>,
|
||||
#[serde(skip_serializing_if = "Option::is_none")]
|
||||
pub message_id: Option<String>,
|
||||
#[serde(skip_serializing_if = "Option::is_none")]
|
||||
pub journal_id: Option<u32>,
|
||||
}
|
||||
|
||||
impl Filter {
|
||||
pub fn matches(&self, entry: &Entry) -> bool {
|
||||
let has = |value: &str, part: &str| value.to_lowercase().contains(&part.to_lowercase());
|
||||
self.after.is_none_or(|after| entry.at >= after)
|
||||
&& self.before.is_none_or(|before| entry.at < before)
|
||||
&& self.sender.as_deref().is_none_or(|s| has(&entry.sender, s))
|
||||
&& self
|
||||
.recipient
|
||||
.as_deref()
|
||||
.is_none_or(|r| entry.recipients.iter().any(|a| has(a, r)))
|
||||
&& self
|
||||
.address
|
||||
.as_deref()
|
||||
.is_none_or(|a| has(&entry.sender, a) || entry.recipients.iter().any(|r| has(r, a)))
|
||||
&& self
|
||||
.direction
|
||||
.is_none_or(|d| d == Direction::Any || d == entry.direction)
|
||||
&& self.text.as_deref().is_none_or(|text| {
|
||||
let subject = entry.subject.to_lowercase();
|
||||
text.to_lowercase()
|
||||
.split_whitespace()
|
||||
.all(|word| subject.contains(word))
|
||||
})
|
||||
&& self.message_id.as_deref().is_none_or(|id| {
|
||||
entry.message_id.trim_matches(['<', '>']) == id.trim_matches(['<', '>'])
|
||||
})
|
||||
&& self.journal_id.is_none_or(|j| entry.journals.contains(&j))
|
||||
}
|
||||
}
|
||||
|
||||
/// Entries matching `filter`, newest first: a page from `position`, up to
|
||||
/// `limit`, and, when asked, how many match in all.
|
||||
pub async fn query(
|
||||
data: &Store,
|
||||
filter: &Filter,
|
||||
position: usize,
|
||||
limit: usize,
|
||||
count_all: bool,
|
||||
) -> trc::Result<(Vec<EntryId>, usize)> {
|
||||
let after = filter.after.unwrap_or(0);
|
||||
let before = filter.before.unwrap_or(u64::MAX);
|
||||
let mut ids = Vec::new();
|
||||
data.iterate(
|
||||
IterateParams::new(
|
||||
key(KIND_TIME, &[after, 0, 0]),
|
||||
key(KIND_TIME, &[before.saturating_sub(1), u64::MAX, u64::MAX]),
|
||||
)
|
||||
.descending()
|
||||
.no_values(),
|
||||
|key, _| {
|
||||
if let Some(parts) = parse_key(key, KIND_TIME, 3) {
|
||||
ids.push(EntryId {
|
||||
node: parts[1],
|
||||
seq: parts[2],
|
||||
});
|
||||
}
|
||||
Ok(true)
|
||||
},
|
||||
)
|
||||
.await
|
||||
.caused_by(trc::location!())?;
|
||||
let mut page = Vec::new();
|
||||
let mut total = 0;
|
||||
for id in ids {
|
||||
let Some(entry) = get(data, id).await? else {
|
||||
continue;
|
||||
};
|
||||
if !filter.matches(&entry) {
|
||||
continue;
|
||||
}
|
||||
if total >= position && page.len() < limit {
|
||||
page.push(id);
|
||||
}
|
||||
total += 1;
|
||||
if !count_all && page.len() >= limit {
|
||||
break;
|
||||
}
|
||||
}
|
||||
Ok((page, total))
|
||||
}
|
||||
|
||||
/// What a purge did.
|
||||
#[derive(Debug, Clone, Default, PartialEq, Eq)]
|
||||
pub struct Purged {
|
||||
pub removed: usize,
|
||||
/// Past their time, kept for a legal hold.
|
||||
pub kept_for_hold: usize,
|
||||
}
|
||||
|
||||
/// Removes entries past their retention (JR-13), except those `held` keeps:
|
||||
/// the entry, its indexes and its blob's link go; the chain link stays,
|
||||
/// with a purge marker. Then each chain's start moves past purged links.
|
||||
pub async fn purge(
|
||||
data: &Store,
|
||||
now: u64,
|
||||
held: impl Fn(&Entry) -> bool + Sync + Send,
|
||||
) -> trc::Result<Purged> {
|
||||
let mut due = Vec::new();
|
||||
data.iterate(
|
||||
IterateParams::new(
|
||||
key(KIND_EXPIRY, &[0, 0, 0]),
|
||||
key(KIND_EXPIRY, &[now, u64::MAX, u64::MAX]),
|
||||
)
|
||||
.ascending()
|
||||
.no_values(),
|
||||
|key, _| {
|
||||
if let Some(parts) = parse_key(key, KIND_EXPIRY, 3) {
|
||||
due.push((
|
||||
parts[0],
|
||||
EntryId {
|
||||
node: parts[1],
|
||||
seq: parts[2],
|
||||
},
|
||||
));
|
||||
}
|
||||
Ok(due.len() < 100_000)
|
||||
},
|
||||
)
|
||||
.await
|
||||
.caused_by(trc::location!())?;
|
||||
|
||||
let mut purged = Purged::default();
|
||||
for chunk in due.chunks(PURGE_BATCH) {
|
||||
let mut batch = BatchBuilder::new();
|
||||
for (expires_at, id) in chunk {
|
||||
let parts = [id.node, id.seq];
|
||||
let Some(entry) = get(data, *id).await? else {
|
||||
// Its entry is already gone: only the index is left
|
||||
batch.clear(class(KIND_EXPIRY, &[*expires_at, id.node, id.seq]));
|
||||
continue;
|
||||
};
|
||||
if held(&entry) {
|
||||
purged.kept_for_hold += 1;
|
||||
continue;
|
||||
}
|
||||
batch
|
||||
.clear(class(KIND_CONTENT, &parts))
|
||||
.clear(class(KIND_TIME, &[entry.at, id.node, id.seq]))
|
||||
.clear(class(KIND_EXPIRY, &[*expires_at, id.node, id.seq]))
|
||||
.set(class(KIND_PURGED, &parts), now.to_be_bytes().to_vec());
|
||||
if let Some(blob) = entry.blob_hash() {
|
||||
batch.clear(BlobOp::Link {
|
||||
hash: blob,
|
||||
to: BlobLink::Temporary { until: HELD_UNTIL },
|
||||
});
|
||||
}
|
||||
purged.removed += 1;
|
||||
}
|
||||
if !batch.is_empty() {
|
||||
data.write(batch.build_all())
|
||||
.await
|
||||
.caused_by(trc::location!())?;
|
||||
}
|
||||
}
|
||||
|
||||
for node in nodes(data).await? {
|
||||
advance_floor(data, node).await?;
|
||||
}
|
||||
Ok(purged)
|
||||
}
|
||||
|
||||
/// Clears the purged links at the start of a node's chain, recording where
|
||||
/// it now starts and the hash that start names.
|
||||
async fn advance_floor(data: &Store, node: u64) -> trc::Result<()> {
|
||||
let start = floor(data, node).await?;
|
||||
let mut cleared: Vec<u64> = Vec::new();
|
||||
let mut next = start.clone();
|
||||
let mut purged_seqs = Vec::new();
|
||||
data.iterate(
|
||||
IterateParams::new(
|
||||
key(KIND_PURGED, &[node, start.seq]),
|
||||
key(KIND_PURGED, &[node, u64::MAX]),
|
||||
)
|
||||
.ascending()
|
||||
.no_values(),
|
||||
|key, _| {
|
||||
if let Some(parts) = parse_key(key, KIND_PURGED, 2) {
|
||||
purged_seqs.push(parts[1]);
|
||||
}
|
||||
Ok(purged_seqs.len() < 100_000)
|
||||
},
|
||||
)
|
||||
.await
|
||||
.caused_by(trc::location!())?;
|
||||
for seq in purged_seqs {
|
||||
if seq != next.seq {
|
||||
break;
|
||||
}
|
||||
let Some(Raw(link)) = data
|
||||
.get_value::<Raw>(key(KIND_LINK, &[node, seq]))
|
||||
.await
|
||||
.caused_by(trc::location!())?
|
||||
else {
|
||||
break;
|
||||
};
|
||||
next = Floor {
|
||||
seq: seq + 1,
|
||||
prev: sha256(&link),
|
||||
};
|
||||
cleared.push(seq);
|
||||
}
|
||||
if cleared.is_empty() {
|
||||
return Ok(());
|
||||
}
|
||||
// The floor moves first: a run cut short leaves links before it, which
|
||||
// the next run clears, never a chain that looks broken
|
||||
let mut batch = BatchBuilder::new();
|
||||
batch.set(class(KIND_FLOOR, &[node]), Json(&next).serialize()?);
|
||||
data.write(batch.build_all())
|
||||
.await
|
||||
.caused_by(trc::location!())?;
|
||||
for chunk in cleared.chunks(PURGE_BATCH) {
|
||||
let mut batch = BatchBuilder::new();
|
||||
for seq in chunk {
|
||||
batch
|
||||
.clear(class(KIND_LINK, &[node, *seq]))
|
||||
.clear(class(KIND_PURGED, &[node, *seq]));
|
||||
}
|
||||
data.write(batch.build_all())
|
||||
.await
|
||||
.caused_by(trc::location!())?;
|
||||
}
|
||||
Ok(())
|
||||
}
|
||||
|
||||
/// One node's chain, as [`verify`] found it.
|
||||
#[derive(Debug, Clone, PartialEq, Eq, SerdeSerialize)]
|
||||
#[serde(rename_all = "camelCase")]
|
||||
pub struct ChainReport {
|
||||
pub node: u64,
|
||||
pub entries: u64,
|
||||
pub purged: u64,
|
||||
pub first_seq: u64,
|
||||
pub last_seq: u64,
|
||||
#[serde(skip_serializing_if = "Option::is_none")]
|
||||
pub broken_at: Option<String>,
|
||||
#[serde(skip_serializing_if = "Option::is_none")]
|
||||
pub reason: Option<String>,
|
||||
}
|
||||
|
||||
/// Rechecks every node's chain (JR-6): each link names the hash of the one
|
||||
/// before it, seqs run without gaps, the head matches the last link, each
|
||||
/// entry hashes to what its link names or was purged, and, with `blobs`,
|
||||
/// each report is there and hashes to what its entry names.
|
||||
pub async fn verify(data: &Store, blobs: Option<&BlobStore>) -> trc::Result<Vec<ChainReport>> {
|
||||
let mut reports = Vec::new();
|
||||
for node in nodes(data).await? {
|
||||
let start = floor(data, node).await?;
|
||||
let head = head(data, node).await?.unwrap_or_default();
|
||||
let mut report = ChainReport {
|
||||
node,
|
||||
entries: 0,
|
||||
purged: 0,
|
||||
first_seq: start.seq,
|
||||
last_seq: start.seq.saturating_sub(1),
|
||||
broken_at: None,
|
||||
reason: None,
|
||||
};
|
||||
let mut links = Vec::new();
|
||||
data.iterate(
|
||||
IterateParams::new(
|
||||
key(KIND_LINK, &[node, start.seq]),
|
||||
key(KIND_LINK, &[node, u64::MAX]),
|
||||
)
|
||||
.ascending(),
|
||||
|key, value| {
|
||||
if let Some(parts) = parse_key(key, KIND_LINK, 2) {
|
||||
links.push((parts[1], value.to_vec()));
|
||||
}
|
||||
Ok(true)
|
||||
},
|
||||
)
|
||||
.await
|
||||
.caused_by(trc::location!())?;
|
||||
|
||||
let mut expected_seq = start.seq;
|
||||
let mut expected_prev = start.prev.clone();
|
||||
for (seq, bytes) in links {
|
||||
let broken = |report: &mut ChainReport, reason: &str| {
|
||||
report.broken_at = Some(EntryId { node, seq }.to_string());
|
||||
report.reason = Some(reason.to_string());
|
||||
};
|
||||
let Ok(Json(link)) = Json::<Link>::deserialize(&bytes) else {
|
||||
broken(&mut report, "The link can't be read.");
|
||||
break;
|
||||
};
|
||||
if seq != expected_seq || link.seq != seq {
|
||||
report.broken_at = Some(EntryId { node, seq }.to_string());
|
||||
report.reason = Some(format!(
|
||||
"Entry {expected_seq} is missing; the next one found is {seq}."
|
||||
));
|
||||
break;
|
||||
}
|
||||
if link.prev != expected_prev {
|
||||
broken(
|
||||
&mut report,
|
||||
"The link doesn't follow from the one before it: one of them was changed.",
|
||||
);
|
||||
break;
|
||||
}
|
||||
match data
|
||||
.get_value::<Raw>(key(KIND_CONTENT, &[node, seq]))
|
||||
.await
|
||||
.caused_by(trc::location!())?
|
||||
{
|
||||
Some(Raw(content)) => {
|
||||
if sha256(&content) != link.content {
|
||||
broken(&mut report, "The entry was changed after it was written.");
|
||||
break;
|
||||
}
|
||||
if let Some(blobs) = blobs {
|
||||
let Ok(Json(entry)) = Json::<Entry>::deserialize(&content) else {
|
||||
broken(&mut report, "The entry can't be read.");
|
||||
break;
|
||||
};
|
||||
let report_bytes = match entry.blob_hash() {
|
||||
Some(hash) => blobs
|
||||
.get_blob(hash.as_slice(), 0..usize::MAX)
|
||||
.await
|
||||
.caused_by(trc::location!())?,
|
||||
None => None,
|
||||
};
|
||||
match report_bytes {
|
||||
Some(bytes) if sha256(&bytes) == entry.sha256 => {}
|
||||
Some(_) => {
|
||||
broken(&mut report, "The report doesn't match its entry.");
|
||||
break;
|
||||
}
|
||||
None => {
|
||||
broken(&mut report, "The report is missing.");
|
||||
break;
|
||||
}
|
||||
}
|
||||
}
|
||||
report.entries += 1;
|
||||
}
|
||||
None => {
|
||||
if data
|
||||
.get_value::<Raw>(key(KIND_PURGED, &[node, seq]))
|
||||
.await
|
||||
.caused_by(trc::location!())?
|
||||
.is_none()
|
||||
{
|
||||
broken(&mut report, "The entry was removed before its time.");
|
||||
break;
|
||||
}
|
||||
report.purged += 1;
|
||||
}
|
||||
}
|
||||
expected_prev = sha256(&bytes);
|
||||
expected_seq = seq + 1;
|
||||
report.last_seq = seq;
|
||||
}
|
||||
|
||||
if report.broken_at.is_none()
|
||||
&& (head.seq != report.last_seq
|
||||
|| (report.last_seq >= report.first_seq && head.hash != expected_prev))
|
||||
{
|
||||
report.broken_at = Some(
|
||||
EntryId {
|
||||
node,
|
||||
seq: report.last_seq,
|
||||
}
|
||||
.to_string(),
|
||||
);
|
||||
report.reason = Some(
|
||||
"The chain's recorded end doesn't match its last link: entries were removed \
|
||||
or changed at the end."
|
||||
.into(),
|
||||
);
|
||||
}
|
||||
reports.push(report);
|
||||
}
|
||||
Ok(reports)
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::*;
|
||||
|
||||
#[test]
|
||||
fn keys_read_back() {
|
||||
let ValueClass::Any(any) = class(KIND_EXPIRY, &[5, 3, 9]) else {
|
||||
panic!()
|
||||
};
|
||||
assert_eq!(parse_key(&any.key, KIND_EXPIRY, 3), Some(vec![5, 3, 9]));
|
||||
let mut with_subspace = vec![SUBSPACE_INBUXA];
|
||||
with_subspace.extend_from_slice(&any.key);
|
||||
assert_eq!(
|
||||
parse_key(&with_subspace, KIND_EXPIRY, 3),
|
||||
Some(vec![5, 3, 9])
|
||||
);
|
||||
assert_eq!(parse_key(&any.key, KIND_TIME, 3), None);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn filters_match() {
|
||||
let entry = Entry {
|
||||
queue_id: 1,
|
||||
at: 100,
|
||||
direction: Direction::Outgoing,
|
||||
sender: "[email protected]".into(),
|
||||
authenticated: true,
|
||||
recipients: vec!["[email protected]".into()],
|
||||
subject: "Q3 figures, final".into(),
|
||||
message_id: "<[email protected]>".into(),
|
||||
accounts: vec![3],
|
||||
tenants: vec![],
|
||||
journals: vec![2],
|
||||
held: false,
|
||||
blob: String::new(),
|
||||
size: 0,
|
||||
sha256: String::new(),
|
||||
expires_at: 0,
|
||||
};
|
||||
let yes = |f: Filter| assert!(f.matches(&entry), "{f:?}");
|
||||
let no = |f: Filter| assert!(!f.matches(&entry), "{f:?}");
|
||||
yes(Filter::default());
|
||||
yes(Filter {
|
||||
sender: Some("alice@".into()),
|
||||
..Default::default()
|
||||
});
|
||||
yes(Filter {
|
||||
address: Some("BANK".into()),
|
||||
..Default::default()
|
||||
});
|
||||
yes(Filter {
|
||||
text: Some("final q3".into()),
|
||||
..Default::default()
|
||||
});
|
||||
yes(Filter {
|
||||
message_id: Some("[email protected]".into()),
|
||||
..Default::default()
|
||||
});
|
||||
yes(Filter {
|
||||
direction: Some(Direction::Any),
|
||||
..Default::default()
|
||||
});
|
||||
no(Filter {
|
||||
direction: Some(Direction::Incoming),
|
||||
..Default::default()
|
||||
});
|
||||
no(Filter {
|
||||
recipient: Some("alice".into()),
|
||||
..Default::default()
|
||||
});
|
||||
no(Filter {
|
||||
before: Some(100),
|
||||
..Default::default()
|
||||
});
|
||||
yes(Filter {
|
||||
after: Some(100),
|
||||
journal_id: Some(2),
|
||||
..Default::default()
|
||||
});
|
||||
no(Filter {
|
||||
journal_id: Some(5),
|
||||
..Default::default()
|
||||
});
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn hex_round_trips() {
|
||||
let bytes = [0u8, 1, 0xab, 0xff];
|
||||
assert_eq!(unhex(&hex(&bytes)), Some(bytes.to_vec()));
|
||||
assert_eq!(unhex("abc"), None);
|
||||
assert_eq!(unhex("zz"), None);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn ids_read_back() {
|
||||
let id = EntryId { node: 3, seq: 77 };
|
||||
assert_eq!(EntryId::from_u64(id.to_u64()), id);
|
||||
assert_eq!(id.to_string(), "3-77");
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,512 @@
|
||||
/*
|
||||
* SPDX-FileCopyrightText: 2026 Coffey Labs LLC
|
||||
*
|
||||
* SPDX-License-Identifier: AGPL-3.0-only
|
||||
*/
|
||||
|
||||
//! Journaling (journaling spec, JR-1 to JR-18): a copy of each message the
|
||||
//! server queues, with its envelope, kept where nothing in the product
|
||||
//! changes or removes it before its retention ends.
|
||||
//!
|
||||
//! - this module: journals, what makes one valid, and where they're kept;
|
||||
//! - [`report`]: the journal report around the untouched message (JR-3);
|
||||
//! - [`entries`]: the built-in journal and its chain (JR-5, JR-6, JR-13).
|
||||
//!
|
||||
//! Kept in the fork's subspace (`store::SUBSPACE_INBUXA`). Every key starts
|
||||
//! with `J`; journals are `j` + id (u32), as JSON. There are few, so they're
|
||||
//! read whole.
|
||||
|
||||
pub mod archive;
|
||||
pub mod entries;
|
||||
pub mod report;
|
||||
|
||||
use crate::{hold::Member, mailflow::rules::jmap_ids};
|
||||
use serde::{Deserialize as SerdeDeserialize, Serialize as SerdeSerialize, de::DeserializeOwned};
|
||||
use std::{
|
||||
sync::{Arc, RwLock},
|
||||
time::{Duration, Instant},
|
||||
};
|
||||
use store::{
|
||||
Deserialize, IterateParams, SUBSPACE_INBUXA, Serialize, Store, ValueKey,
|
||||
write::{AnyClass, BatchBuilder, ValueClass, assert::AssertValue},
|
||||
};
|
||||
use trc::AddContext;
|
||||
|
||||
pub(crate) const FEATURE: u8 = b'J';
|
||||
const KIND_JOURNAL: u8 = b'j';
|
||||
const CREATE_ATTEMPTS: usize = 5;
|
||||
|
||||
/// Retention a journal may be given, in days (settled answer 3).
|
||||
pub const MIN_RETENTION_DAYS: u32 = 30;
|
||||
pub const MAX_RETENTION_DAYS: u32 = 3650;
|
||||
/// Most entries in one scope list.
|
||||
const MAX_LIST: usize = 5_000;
|
||||
|
||||
/// Which way a message goes, from this server's side (JR-9).
|
||||
#[derive(Debug, Clone, Copy, PartialEq, Eq, SerdeSerialize, SerdeDeserialize)]
|
||||
#[serde(rename_all = "camelCase")]
|
||||
pub enum Direction {
|
||||
/// From someone here to at least one recipient elsewhere.
|
||||
Outgoing,
|
||||
/// From elsewhere to someone here.
|
||||
Incoming,
|
||||
/// From someone here, to people here only.
|
||||
Internal,
|
||||
Any,
|
||||
}
|
||||
|
||||
impl Direction {
|
||||
pub fn as_str(&self) -> &'static str {
|
||||
match self {
|
||||
Direction::Outgoing => "outgoing",
|
||||
Direction::Incoming => "incoming",
|
||||
Direction::Internal => "internal",
|
||||
Direction::Any => "any",
|
||||
}
|
||||
}
|
||||
|
||||
/// A message's direction: `Any` is never one.
|
||||
pub fn of(sender_local: bool, any_remote: bool, any_local: bool) -> Direction {
|
||||
match (sender_local, any_remote) {
|
||||
(true, true) => Direction::Outgoing,
|
||||
(true, false) => Direction::Internal,
|
||||
(false, _) if any_local => Direction::Incoming,
|
||||
// Nobody here on either side: relayed mail counts as outgoing
|
||||
(false, _) => Direction::Outgoing,
|
||||
}
|
||||
}
|
||||
|
||||
fn includes(&self, direction: Direction) -> bool {
|
||||
*self == Direction::Any || *self == direction
|
||||
}
|
||||
}
|
||||
|
||||
/// Whose mail a journal takes (JR-9): everyone, or people reached through
|
||||
/// their account, domain, group or tenant. Ids are in the JMAP form.
|
||||
#[derive(Debug, Clone, Default, PartialEq, Eq, SerdeSerialize, SerdeDeserialize)]
|
||||
#[serde(rename_all = "camelCase")]
|
||||
pub struct Scope {
|
||||
#[serde(default)]
|
||||
pub everyone: bool,
|
||||
#[serde(default, with = "jmap_ids")]
|
||||
pub accounts: Vec<u32>,
|
||||
#[serde(default, with = "jmap_ids")]
|
||||
pub groups: Vec<u32>,
|
||||
#[serde(default, with = "jmap_ids")]
|
||||
pub domains: Vec<u32>,
|
||||
#[serde(default, with = "jmap_ids")]
|
||||
pub tenants: Vec<u32>,
|
||||
}
|
||||
|
||||
impl Scope {
|
||||
fn lists(&self) -> [&Vec<u32>; 4] {
|
||||
[&self.accounts, &self.groups, &self.domains, &self.tenants]
|
||||
}
|
||||
|
||||
/// Whether this scope reaches one person here.
|
||||
pub fn covers(&self, member: &Member) -> bool {
|
||||
self.everyone
|
||||
|| self.accounts.contains(&member.account)
|
||||
|| member.domains.iter().any(|d| self.domains.contains(d))
|
||||
|| member.groups.iter().any(|g| self.groups.contains(g))
|
||||
|| member.tenant.is_some_and(|t| self.tenants.contains(&t))
|
||||
}
|
||||
}
|
||||
|
||||
/// A journal (JR-9): what it takes, and how long its entries are kept.
|
||||
#[derive(Debug, Clone, PartialEq, Eq, SerdeSerialize, SerdeDeserialize)]
|
||||
#[serde(rename_all = "camelCase")]
|
||||
pub struct Journal {
|
||||
#[serde(default)]
|
||||
pub id: u32,
|
||||
pub name: String,
|
||||
#[serde(default)]
|
||||
pub description: String,
|
||||
#[serde(default)]
|
||||
pub enabled: bool,
|
||||
pub direction: Direction,
|
||||
pub scope: Scope,
|
||||
/// How long an entry this journal writes is kept. An entry keeps the
|
||||
/// retention it was written with (JR-12).
|
||||
pub retention_days: u32,
|
||||
/// Whether entries go into the built-in journal (JR-5).
|
||||
#[serde(default = "yes")]
|
||||
pub built_in: bool,
|
||||
/// An outside archive's journal address, sent each report (JR-7).
|
||||
#[serde(default, skip_serializing_if = "Option::is_none")]
|
||||
pub archive_address: Option<String>,
|
||||
#[serde(default)]
|
||||
pub created_by: String,
|
||||
#[serde(default)]
|
||||
pub created_at: u64,
|
||||
#[serde(default)]
|
||||
pub updated_at: u64,
|
||||
}
|
||||
|
||||
/// Why a journal was refused: the property, and what to do.
|
||||
#[derive(Debug, Clone, PartialEq, Eq)]
|
||||
pub struct Invalid {
|
||||
pub property: &'static str,
|
||||
pub reason: String,
|
||||
}
|
||||
|
||||
fn invalid(property: &'static str, reason: impl Into<String>) -> Result<(), Invalid> {
|
||||
Err(Invalid {
|
||||
property,
|
||||
reason: reason.into(),
|
||||
})
|
||||
}
|
||||
|
||||
impl Journal {
|
||||
pub fn validate(&self) -> Result<(), Invalid> {
|
||||
if self.name.trim().is_empty() {
|
||||
return invalid("name", "Give the journal a name.");
|
||||
}
|
||||
if self.name.len() > 200 || self.description.len() > 2_000 {
|
||||
return invalid("name", "The name or description is too long.");
|
||||
}
|
||||
if !(MIN_RETENTION_DAYS..=MAX_RETENTION_DAYS).contains(&self.retention_days) {
|
||||
return invalid(
|
||||
"retentionDays",
|
||||
format!("Keep entries between {MIN_RETENTION_DAYS} and {MAX_RETENTION_DAYS} days."),
|
||||
);
|
||||
}
|
||||
// Neither is a journal only rules send mail to (JR-10)
|
||||
let chosen = self.scope.lists().iter().any(|list| !list.is_empty());
|
||||
if self.scope.everyone && chosen {
|
||||
return invalid(
|
||||
"scope",
|
||||
"Journal everyone, or choose accounts, groups, domains or tenants; not both.",
|
||||
);
|
||||
}
|
||||
if !self.built_in && self.archive_address.is_none() {
|
||||
return invalid(
|
||||
"builtIn",
|
||||
"Keep entries in the built-in journal, send them to an archive, or both.",
|
||||
);
|
||||
}
|
||||
if let Some(address) = &self.archive_address
|
||||
&& !is_address(address)
|
||||
{
|
||||
return invalid(
|
||||
"archiveAddress",
|
||||
format!("\"{address}\" isn't an email address."),
|
||||
);
|
||||
}
|
||||
if self.scope.lists().iter().any(|list| list.len() > MAX_LIST) {
|
||||
return invalid("scope", format!("Choose at most {MAX_LIST} of each."));
|
||||
}
|
||||
Ok(())
|
||||
}
|
||||
|
||||
/// Whether this journal takes a message going `direction` with these
|
||||
/// people here on either side.
|
||||
/// Whether only rules send this journal mail (JR-10).
|
||||
pub fn rules_only(&self) -> bool {
|
||||
!self.scope.everyone && self.scope.lists().iter().all(|list| list.is_empty())
|
||||
}
|
||||
|
||||
pub fn takes(&self, direction: Direction, members: &[Member]) -> bool {
|
||||
self.enabled
|
||||
&& self.direction.includes(direction)
|
||||
&& (self.scope.everyone || members.iter().any(|m| self.scope.covers(m)))
|
||||
}
|
||||
}
|
||||
|
||||
fn yes() -> bool {
|
||||
true
|
||||
}
|
||||
|
||||
/// An address an archive can be sent to: one `@`, something either side,
|
||||
/// nothing that would break an envelope.
|
||||
fn is_address(address: &str) -> bool {
|
||||
address.len() <= 320
|
||||
&& address.split_once('@').is_some_and(|(local, domain)| {
|
||||
!local.is_empty() && domain.contains('.') && !domain.contains('@')
|
||||
})
|
||||
&& !address
|
||||
.chars()
|
||||
.any(|c| c.is_whitespace() || c.is_control() || matches!(c, '<' | '>' | ',' | ';'))
|
||||
}
|
||||
|
||||
/// A value stored as JSON.
|
||||
pub(crate) struct Json<T>(pub T);
|
||||
|
||||
impl<T: SerdeSerialize> Serialize for Json<T> {
|
||||
fn serialize(&self) -> trc::Result<Vec<u8>> {
|
||||
serde_json::to_vec(&self.0).map_err(|err| {
|
||||
trc::StoreEvent::UnexpectedError
|
||||
.into_err()
|
||||
.details("Failed to serialize a journal record")
|
||||
.reason(err)
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
impl<T: DeserializeOwned + Sync + Send> Deserialize for Json<T> {
|
||||
fn deserialize(bytes: &[u8]) -> trc::Result<Self> {
|
||||
serde_json::from_slice(bytes).map(Json).map_err(|err| {
|
||||
trc::StoreEvent::DataCorruption
|
||||
.into_err()
|
||||
.details("Invalid journal record")
|
||||
.reason(err)
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
fn class(id: u32) -> ValueClass {
|
||||
let mut key = Vec::with_capacity(6);
|
||||
key.push(FEATURE);
|
||||
key.push(KIND_JOURNAL);
|
||||
key.extend_from_slice(&id.to_be_bytes());
|
||||
ValueClass::Any(AnyClass {
|
||||
subspace: SUBSPACE_INBUXA,
|
||||
key,
|
||||
})
|
||||
}
|
||||
|
||||
fn key(id: u32) -> ValueKey<ValueClass> {
|
||||
ValueKey::from(class(id))
|
||||
}
|
||||
|
||||
pub async fn get(data: &Store, id: u32) -> trc::Result<Option<Journal>> {
|
||||
Ok(data
|
||||
.get_value::<Json<Journal>>(key(id))
|
||||
.await
|
||||
.caused_by(trc::location!())?
|
||||
.map(|Json(journal)| journal))
|
||||
}
|
||||
|
||||
/// Every journal, oldest first.
|
||||
pub async fn all(data: &Store) -> trc::Result<Vec<Journal>> {
|
||||
let mut journals = Vec::new();
|
||||
data.iterate(IterateParams::new(key(0), key(u32::MAX)), |_, value| {
|
||||
if let Ok(Json(journal)) = Json::<Journal>::deserialize(value) {
|
||||
journals.push(journal);
|
||||
}
|
||||
Ok(true)
|
||||
})
|
||||
.await
|
||||
.caused_by(trc::location!())?;
|
||||
journals.sort_by_key(|journal| journal.id);
|
||||
Ok(journals)
|
||||
}
|
||||
|
||||
/// Writes a new journal under the next free id, which it returns.
|
||||
pub async fn create(data: &Store, journal: &Journal) -> trc::Result<u32> {
|
||||
let mut attempt = 0;
|
||||
loop {
|
||||
attempt += 1;
|
||||
let id = all(data).await?.iter().map(|j| j.id).max().unwrap_or(0) + 1;
|
||||
let stored = Journal {
|
||||
id,
|
||||
..journal.clone()
|
||||
};
|
||||
let mut batch = BatchBuilder::new();
|
||||
batch.assert_value(class(id), AssertValue::None);
|
||||
batch.set(class(id), Json(&stored).serialize()?);
|
||||
match data.write(batch.build_all()).await {
|
||||
Ok(_) => {
|
||||
invalidate();
|
||||
return Ok(id);
|
||||
}
|
||||
Err(err)
|
||||
if attempt < CREATE_ATTEMPTS
|
||||
&& matches!(
|
||||
err.as_ref(),
|
||||
trc::EventType::Store(trc::StoreEvent::AssertValueFailed)
|
||||
) => {}
|
||||
Err(err) => return Err(err.caused_by(trc::location!())),
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// Replaces a stored journal (same id).
|
||||
pub async fn update(data: &Store, journal: &Journal) -> trc::Result<()> {
|
||||
let mut batch = BatchBuilder::new();
|
||||
batch.set(class(journal.id), Json(journal).serialize()?);
|
||||
data.write(batch.build_all())
|
||||
.await
|
||||
.caused_by(trc::location!())?;
|
||||
invalidate();
|
||||
Ok(())
|
||||
}
|
||||
|
||||
/// Removes a journal. Its entries stay, each until its own time.
|
||||
pub async fn delete(data: &Store, id: u32) -> trc::Result<()> {
|
||||
let mut batch = BatchBuilder::new();
|
||||
batch.clear(class(id));
|
||||
data.write(batch.build_all())
|
||||
.await
|
||||
.caused_by(trc::location!())?;
|
||||
invalidate();
|
||||
Ok(())
|
||||
}
|
||||
|
||||
/// How long a node keeps its copy of the journals before reading them again.
|
||||
pub const TTL: Duration = Duration::from_secs(30);
|
||||
|
||||
type Cached = Option<(Instant, Arc<Vec<Journal>>)>;
|
||||
static CACHE: RwLock<Cached> = RwLock::new(None);
|
||||
|
||||
/// Forgets this node's copy, so the next message reads the journals again.
|
||||
pub fn invalidate() {
|
||||
if let Ok(mut cache) = CACHE.write() {
|
||||
*cache = None;
|
||||
}
|
||||
}
|
||||
|
||||
/// The enabled journals, from this node's copy (refreshed every [`TTL`]).
|
||||
pub async fn enabled(data: &Store) -> trc::Result<Arc<Vec<Journal>>> {
|
||||
if let Ok(cache) = CACHE.read()
|
||||
&& let Some((at, journals)) = cache.as_ref()
|
||||
&& at.elapsed() < TTL
|
||||
{
|
||||
return Ok(journals.clone());
|
||||
}
|
||||
let journals = Arc::new(
|
||||
all(data)
|
||||
.await?
|
||||
.into_iter()
|
||||
.filter(|journal| journal.enabled)
|
||||
.collect::<Vec<_>>(),
|
||||
);
|
||||
if let Ok(mut cache) = CACHE.write() {
|
||||
*cache = Some((Instant::now(), journals.clone()));
|
||||
}
|
||||
Ok(journals)
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::*;
|
||||
|
||||
fn journal(scope: Scope) -> Journal {
|
||||
Journal {
|
||||
id: 1,
|
||||
name: "Finance".into(),
|
||||
description: String::new(),
|
||||
enabled: true,
|
||||
direction: Direction::Any,
|
||||
scope,
|
||||
retention_days: 365,
|
||||
built_in: true,
|
||||
archive_address: None,
|
||||
created_by: String::new(),
|
||||
created_at: 0,
|
||||
updated_at: 0,
|
||||
}
|
||||
}
|
||||
|
||||
fn member(account: u32, groups: Vec<u32>) -> Member {
|
||||
Member {
|
||||
account,
|
||||
domains: vec![1],
|
||||
groups,
|
||||
tenant: None,
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn scope_is_everyone_or_chosen() {
|
||||
assert!(
|
||||
journal(Scope {
|
||||
everyone: true,
|
||||
..Default::default()
|
||||
})
|
||||
.validate()
|
||||
.is_ok()
|
||||
);
|
||||
// Nobody chosen: only rules send it mail
|
||||
let rules_only = journal(Scope::default());
|
||||
assert!(rules_only.validate().is_ok());
|
||||
assert!(rules_only.rules_only());
|
||||
assert!(!rules_only.takes(Direction::Any, &[member(3, vec![7])]));
|
||||
let both = Scope {
|
||||
everyone: true,
|
||||
groups: vec![4],
|
||||
..Default::default()
|
||||
};
|
||||
assert_eq!(journal(both).validate().unwrap_err().property, "scope");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn destinations() {
|
||||
let mut j = journal(Scope {
|
||||
everyone: true,
|
||||
..Default::default()
|
||||
});
|
||||
j.built_in = false;
|
||||
assert_eq!(j.validate().unwrap_err().property, "builtIn");
|
||||
j.archive_address = Some("[email protected]".into());
|
||||
assert!(j.validate().is_ok());
|
||||
for bad in [
|
||||
"archive",
|
||||
"a@b",
|
||||
"a [email protected]",
|
||||
"<[email protected]>",
|
||||
"a@[email protected]",
|
||||
] {
|
||||
j.archive_address = Some(bad.into());
|
||||
assert_eq!(
|
||||
j.validate().unwrap_err().property,
|
||||
"archiveAddress",
|
||||
"{bad}"
|
||||
);
|
||||
}
|
||||
// Stored before destinations existed: the built-in journal
|
||||
let old: Journal = serde_json::from_str(
|
||||
r#"{"name":"Old","direction":"any","scope":{"everyone":true},"retentionDays":30}"#,
|
||||
)
|
||||
.unwrap();
|
||||
assert!(old.built_in && old.archive_address.is_none());
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn retention_has_bounds() {
|
||||
let mut j = journal(Scope {
|
||||
everyone: true,
|
||||
..Default::default()
|
||||
});
|
||||
j.retention_days = 29;
|
||||
assert_eq!(j.validate().unwrap_err().property, "retentionDays");
|
||||
j.retention_days = 3651;
|
||||
assert!(j.validate().is_err());
|
||||
j.retention_days = 3650;
|
||||
assert!(j.validate().is_ok());
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn takes_by_direction_and_member() {
|
||||
let mut j = journal(Scope {
|
||||
groups: vec![7],
|
||||
..Default::default()
|
||||
});
|
||||
assert!(j.takes(Direction::Outgoing, &[member(3, vec![7])]));
|
||||
assert!(!j.takes(Direction::Outgoing, &[member(3, vec![8])]));
|
||||
assert!(!j.takes(Direction::Outgoing, &[]));
|
||||
j.direction = Direction::Incoming;
|
||||
assert!(!j.takes(Direction::Outgoing, &[member(3, vec![7])]));
|
||||
j.enabled = false;
|
||||
assert!(!j.takes(Direction::Incoming, &[member(3, vec![7])]));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn directions() {
|
||||
assert_eq!(Direction::of(true, true, true), Direction::Outgoing);
|
||||
assert_eq!(Direction::of(true, false, true), Direction::Internal);
|
||||
assert_eq!(Direction::of(false, false, true), Direction::Incoming);
|
||||
assert_eq!(Direction::of(false, true, true), Direction::Incoming);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn scope_ids_are_jmap_ids() {
|
||||
let scope: Scope = serde_json::from_str(r#"{"groups":["b"],"tenants":[7]}"#).unwrap();
|
||||
assert_eq!(scope.groups, vec![1]);
|
||||
assert_eq!(scope.tenants, vec![7]);
|
||||
assert_eq!(
|
||||
serde_json::to_value(&scope).unwrap()["tenants"],
|
||||
serde_json::json!(["h"])
|
||||
);
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,385 @@
|
||||
/*
|
||||
* SPDX-FileCopyrightText: 2026 Coffey Labs LLC
|
||||
*
|
||||
* SPDX-License-Identifier: AGPL-3.0-only
|
||||
*/
|
||||
|
||||
//! The journal report (JR-3, JR-4): a message whose first part lists the
|
||||
//! envelope, one field a line, and whose second part is the message as it
|
||||
//! was queued, byte for byte, as `message/rfc822`. Field names are fixed
|
||||
//! English: a report is a record, and scripts read it.
|
||||
|
||||
use super::Direction;
|
||||
use mail_builder::headers::{Header, date::Date, text::Text};
|
||||
use mail_parser::MessageParser;
|
||||
use sha2::{Digest, Sha256};
|
||||
|
||||
/// One envelope recipient, with the address it was given as (a list's, for
|
||||
/// the list's members).
|
||||
#[derive(Debug, Clone, PartialEq, Eq)]
|
||||
pub struct Recipient {
|
||||
pub address: String,
|
||||
pub orcpt: Option<String>,
|
||||
/// The mail flow rule that added or redirected to it.
|
||||
pub added_by: Option<String>,
|
||||
}
|
||||
|
||||
/// What the queue knows about a message.
|
||||
#[derive(Debug, Clone)]
|
||||
pub struct Envelope<'x> {
|
||||
pub sender: &'x str,
|
||||
pub authenticated: bool,
|
||||
pub recipients: &'x [Recipient],
|
||||
pub queue_id: u64,
|
||||
/// Seconds.
|
||||
pub received: u64,
|
||||
pub direction: Direction,
|
||||
pub held: bool,
|
||||
}
|
||||
|
||||
/// What a report says, besides the envelope's own fields.
|
||||
#[derive(Debug, Clone, Default, PartialEq, Eq)]
|
||||
pub struct Fields {
|
||||
pub subject: String,
|
||||
pub message_id: String,
|
||||
pub to: Vec<String>,
|
||||
pub cc: Vec<String>,
|
||||
/// Envelope recipients in neither To nor Cc, nor reached through a list.
|
||||
pub bcc: Vec<String>,
|
||||
/// A list's address, and its members among the recipients.
|
||||
pub expanded: Vec<(String, Vec<String>)>,
|
||||
/// A rule's name, and the recipients it added.
|
||||
pub added: Vec<(String, Vec<String>)>,
|
||||
}
|
||||
|
||||
/// One line's worth of a value: no line breaks, no control characters.
|
||||
fn line(value: &str) -> String {
|
||||
value
|
||||
.chars()
|
||||
.map(|c| if c.is_control() { ' ' } else { c })
|
||||
.collect::<String>()
|
||||
.trim()
|
||||
.to_string()
|
||||
}
|
||||
|
||||
/// The address an ORCPT names, without its `rfc822;` type.
|
||||
fn orcpt_address(orcpt: &str) -> String {
|
||||
let orcpt = orcpt.trim();
|
||||
let bare = match orcpt.split_once(';') {
|
||||
Some((kind, address)) if kind.eq_ignore_ascii_case("rfc822") => address,
|
||||
_ => orcpt,
|
||||
};
|
||||
bare.trim().to_lowercase()
|
||||
}
|
||||
|
||||
/// Sorts the envelope's recipients by how they were addressed.
|
||||
pub fn fields(envelope: &Envelope<'_>, original: &[u8]) -> Fields {
|
||||
let parsed = MessageParser::default().parse_headers(original);
|
||||
let headed = |which: Option<&mail_parser::Address<'_>>| -> Vec<String> {
|
||||
which
|
||||
.map(|list| {
|
||||
list.iter()
|
||||
.filter_map(|addr| addr.address())
|
||||
.map(|address| address.to_lowercase())
|
||||
.collect()
|
||||
})
|
||||
.unwrap_or_default()
|
||||
};
|
||||
let (subject, message_id, header_to, header_cc) = match &parsed {
|
||||
Some(message) => (
|
||||
message.subject().map(line).unwrap_or_default(),
|
||||
message
|
||||
.message_id()
|
||||
.map(|id| format!("<{}>", line(id)))
|
||||
.unwrap_or_default(),
|
||||
headed(message.to()),
|
||||
headed(message.cc()),
|
||||
),
|
||||
None => Default::default(),
|
||||
};
|
||||
|
||||
let mut fields = Fields {
|
||||
subject,
|
||||
message_id,
|
||||
..Default::default()
|
||||
};
|
||||
for rcpt in envelope.recipients {
|
||||
let address = rcpt.address.to_lowercase();
|
||||
let via = rcpt
|
||||
.orcpt
|
||||
.as_deref()
|
||||
.map(orcpt_address)
|
||||
.filter(|via| !via.is_empty() && *via != address);
|
||||
if let Some(rule) = &rcpt.added_by {
|
||||
match fields.added.iter_mut().find(|(name, _)| name == rule) {
|
||||
Some((_, added)) => added.push(line(&rcpt.address)),
|
||||
None => fields.added.push((line(rule), vec![line(&rcpt.address)])),
|
||||
}
|
||||
} else if header_to.contains(&address) {
|
||||
fields.to.push(line(&rcpt.address));
|
||||
} else if header_cc.contains(&address) {
|
||||
fields.cc.push(line(&rcpt.address));
|
||||
} else if let Some(via) = via {
|
||||
match fields.expanded.iter_mut().find(|(list, _)| *list == via) {
|
||||
Some((_, members)) => members.push(line(&rcpt.address)),
|
||||
None => fields
|
||||
.expanded
|
||||
.push((line(&via), vec![line(&rcpt.address)])),
|
||||
}
|
||||
} else {
|
||||
fields.bcc.push(line(&rcpt.address));
|
||||
}
|
||||
}
|
||||
fields
|
||||
}
|
||||
|
||||
/// The report's first part.
|
||||
pub fn text(envelope: &Envelope<'_>, fields: &Fields) -> String {
|
||||
let mut out = String::new();
|
||||
let mut field = |name: &str, value: &str| {
|
||||
if !value.is_empty() {
|
||||
out.push_str(name);
|
||||
out.push_str(": ");
|
||||
out.push_str(value);
|
||||
out.push_str("\r\n");
|
||||
}
|
||||
};
|
||||
let sender = if envelope.sender.is_empty() {
|
||||
"<>".to_string()
|
||||
} else {
|
||||
line(envelope.sender)
|
||||
};
|
||||
field("Sender", &sender);
|
||||
field(
|
||||
"Authenticated",
|
||||
if envelope.authenticated { "yes" } else { "no" },
|
||||
);
|
||||
field("Subject", &fields.subject);
|
||||
field("Message-ID", &fields.message_id);
|
||||
field("Queue ID", &format!("{:x}", envelope.queue_id));
|
||||
field(
|
||||
"Received",
|
||||
&mail_parser::DateTime::from_timestamp(envelope.received as i64).to_rfc3339(),
|
||||
);
|
||||
field("Direction", envelope.direction.as_str());
|
||||
field("To", &fields.to.join(", "));
|
||||
field("Cc", &fields.cc.join(", "));
|
||||
field("Bcc", &fields.bcc.join(", "));
|
||||
for (list, members) in &fields.expanded {
|
||||
field("Expanded", &format!("{list} -> {}", members.join(", ")));
|
||||
}
|
||||
for (rule, added) in &fields.added {
|
||||
field("Added by rule", &format!("{rule} -> {}", added.join(", ")));
|
||||
}
|
||||
if envelope.held {
|
||||
field("Held for review", "yes");
|
||||
}
|
||||
out
|
||||
}
|
||||
|
||||
fn hex(bytes: &[u8]) -> String {
|
||||
bytes.iter().map(|b| format!("{b:02x}")).collect()
|
||||
}
|
||||
|
||||
/// Whether a message can travel as 8bit: no NULs, no line past 998 bytes.
|
||||
fn fits_8bit(message: &[u8]) -> bool {
|
||||
!message.contains(&0) && message.split(|b| *b == b'\n').all(|l| l.len() <= 998)
|
||||
}
|
||||
|
||||
/// The whole report: headers, the fields, then the original untouched.
|
||||
/// `from` is the address the report is from; `host` names the server in its
|
||||
/// Message-ID.
|
||||
pub fn build(
|
||||
envelope: &Envelope<'_>,
|
||||
original: &[u8],
|
||||
from: &str,
|
||||
host: &str,
|
||||
) -> (Vec<u8>, Fields) {
|
||||
let fields = fields(envelope, original);
|
||||
let body = text(envelope, &fields);
|
||||
// A boundary that can't occur in the original
|
||||
let mut boundary = format!("journal-{}", &hex(&Sha256::digest(original))[..32]);
|
||||
while original
|
||||
.windows(boundary.len())
|
||||
.any(|window| window == boundary.as_bytes())
|
||||
{
|
||||
boundary.push('x');
|
||||
}
|
||||
|
||||
let mut out: Vec<u8> = Vec::with_capacity(original.len() + body.len() + 1024);
|
||||
out.extend_from_slice(format!("From: Journal <{}>\r\n", line(from)).as_bytes());
|
||||
out.extend_from_slice(b"Date: ");
|
||||
out.extend_from_slice(Date::new(envelope.received as i64).to_rfc822().as_bytes());
|
||||
out.extend_from_slice(b"\r\n");
|
||||
out.extend_from_slice(b"Subject: ");
|
||||
let subject = if fields.subject.is_empty() {
|
||||
"Journal report".to_string()
|
||||
} else {
|
||||
format!("Journal report: {}", fields.subject)
|
||||
};
|
||||
Text::new(subject).write_header(&mut out, "Subject: ".len());
|
||||
out.extend_from_slice(
|
||||
format!(
|
||||
"Message-ID: <journal.{:x}.{}@{}>\r\n",
|
||||
envelope.queue_id,
|
||||
envelope.received,
|
||||
line(host)
|
||||
)
|
||||
.as_bytes(),
|
||||
);
|
||||
out.extend_from_slice(format!("X-Inbuxa-Journal: {:x}\r\n", envelope.queue_id).as_bytes());
|
||||
out.extend_from_slice(b"MIME-Version: 1.0\r\n");
|
||||
out.extend_from_slice(
|
||||
format!("Content-Type: multipart/mixed; boundary=\"{boundary}\"\r\n\r\n").as_bytes(),
|
||||
);
|
||||
out.extend_from_slice(format!("--{boundary}\r\n").as_bytes());
|
||||
out.extend_from_slice(
|
||||
b"Content-Type: text/plain; charset=utf-8\r\nContent-Transfer-Encoding: 8bit\r\n\r\n",
|
||||
);
|
||||
out.extend_from_slice(body.as_bytes());
|
||||
out.extend_from_slice(format!("\r\n--{boundary}\r\n").as_bytes());
|
||||
out.extend_from_slice(b"Content-Type: message/rfc822\r\n");
|
||||
out.extend_from_slice(b"Content-Disposition: attachment; filename=\"original.eml\"\r\n");
|
||||
out.extend_from_slice(if fits_8bit(original) {
|
||||
b"Content-Transfer-Encoding: 8bit\r\n\r\n".as_slice()
|
||||
} else {
|
||||
b"Content-Transfer-Encoding: binary\r\n\r\n".as_slice()
|
||||
});
|
||||
out.extend_from_slice(original);
|
||||
// The line break before a boundary belongs to the boundary: the
|
||||
// original keeps its own last one
|
||||
out.extend_from_slice(format!("\r\n--{boundary}--\r\n").as_bytes());
|
||||
(out, fields)
|
||||
}
|
||||
|
||||
/// Where the original starts and ends inside a report [`build`] made.
|
||||
pub fn original(report: &[u8]) -> Option<&[u8]> {
|
||||
let parsed = MessageParser::default().parse(report)?;
|
||||
let part = parsed.attachment(0)?;
|
||||
let start = part.raw_body_offset() as usize;
|
||||
let end = part.raw_end_offset() as usize;
|
||||
report.get(start..end)
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::*;
|
||||
|
||||
const ORIGINAL: &[u8] = b"From: [email protected]\r\n\
|
||||
To: Bank <pay@bank.example>\r\n\
|
||||
Cc: bob@example.com\r\n\
|
||||
Subject: Q3 figures\r\n\
|
||||
Message-ID: <abc@example.com>\r\n\
|
||||
\r\n\
|
||||
The figures.\r\n";
|
||||
|
||||
fn rcpt(address: &str, orcpt: Option<&str>) -> Recipient {
|
||||
Recipient {
|
||||
address: address.into(),
|
||||
orcpt: orcpt.map(Into::into),
|
||||
added_by: None,
|
||||
}
|
||||
}
|
||||
|
||||
fn envelope(recipients: &[Recipient]) -> Envelope<'_> {
|
||||
Envelope {
|
||||
sender: "[email protected]",
|
||||
authenticated: true,
|
||||
recipients,
|
||||
queue_id: 0x1a2b,
|
||||
received: 1_790_000_000,
|
||||
direction: Direction::Outgoing,
|
||||
held: false,
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn recipients_sorted_by_how_they_were_addressed() {
|
||||
let recipients = [
|
||||
rcpt("[email protected]", None),
|
||||
rcpt("[email protected]", Some("rfc822;[email protected]")),
|
||||
rcpt("[email protected]", None),
|
||||
rcpt("[email protected]", Some("[email protected]")),
|
||||
rcpt("[email protected]", Some("rfc822;[email protected]")),
|
||||
];
|
||||
let fields = fields(&envelope(&recipients), ORIGINAL);
|
||||
assert_eq!(fields.subject, "Q3 figures");
|
||||
assert_eq!(fields.message_id, "<[email protected]>");
|
||||
assert_eq!(fields.to, vec!["[email protected]"]);
|
||||
assert_eq!(fields.cc, vec!["[email protected]"]);
|
||||
assert_eq!(fields.bcc, vec!["[email protected]"]);
|
||||
assert_eq!(
|
||||
fields.expanded,
|
||||
vec![(
|
||||
"[email protected]".to_string(),
|
||||
vec![
|
||||
"[email protected]".to_string(),
|
||||
"[email protected]".to_string()
|
||||
]
|
||||
)]
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn report_carries_the_original_untouched() {
|
||||
let recipients = [
|
||||
rcpt("[email protected]", None),
|
||||
rcpt("[email protected]", None),
|
||||
];
|
||||
let (report, _) = build(
|
||||
&envelope(&recipients),
|
||||
ORIGINAL,
|
||||
"[email protected]",
|
||||
"mx.example.com",
|
||||
);
|
||||
let text = String::from_utf8_lossy(&report);
|
||||
assert!(text.contains("Sender: [email protected]\r\n"));
|
||||
assert!(text.contains("Bcc: [email protected]\r\n"));
|
||||
assert!(text.contains("Queue ID: 1a2b\r\n"));
|
||||
assert!(text.contains("Direction: outgoing\r\n"));
|
||||
assert!(text.contains("Subject: Journal report: Q3 figures\r\n"));
|
||||
assert!(!text.contains("Held for review"));
|
||||
assert_eq!(original(&report), Some(ORIGINAL));
|
||||
let unterminated = &ORIGINAL[..ORIGINAL.len() - 2];
|
||||
let (report, _) = build(
|
||||
&envelope(&recipients),
|
||||
unterminated,
|
||||
"[email protected]",
|
||||
"mx.example.com",
|
||||
);
|
||||
assert_eq!(original(&report), Some(unterminated));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn rule_added_recipients_say_so() {
|
||||
let mut copied = rcpt("[email protected]", None);
|
||||
copied.added_by = Some("Copy finance".into());
|
||||
let recipients = [rcpt("[email protected]", None), copied];
|
||||
let env = envelope(&recipients);
|
||||
let fields = fields(&env, ORIGINAL);
|
||||
assert!(fields.bcc.is_empty(), "{fields:?}");
|
||||
assert!(
|
||||
text(&env, &fields).contains("Added by rule: Copy finance -> [email protected]\r\n")
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn values_stay_on_one_line() {
|
||||
let recipients = [rcpt("[email protected]", None)];
|
||||
let mut env = envelope(&recipients);
|
||||
env.sender = "[email protected]\r\nBcc: [email protected]";
|
||||
env.held = true;
|
||||
let body = text(&env, &Fields::default());
|
||||
assert_eq!(body.matches("\r\n").count(), body.lines().count());
|
||||
assert!(body.contains("Sender: [email protected] Bcc: [email protected]\r\n"));
|
||||
assert!(body.contains("Held for review: yes\r\n"));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn an_empty_sender_is_shown_as_such() {
|
||||
let recipients = [rcpt("[email protected]", None)];
|
||||
let mut env = envelope(&recipients);
|
||||
env.sender = "";
|
||||
assert!(text(&env, &Fields::default()).starts_with("Sender: <>\r\n"));
|
||||
}
|
||||
}
|
||||
@@ -1,5 +1,5 @@
|
||||
/*
|
||||
* SPDX-FileCopyrightText: 2026 Coffey Labs
|
||||
* SPDX-FileCopyrightText: 2026 Coffey Labs LLC
|
||||
*
|
||||
* SPDX-License-Identifier: AGPL-3.0-only
|
||||
*/
|
||||
@@ -21,8 +21,11 @@
|
||||
pub mod ai;
|
||||
pub mod audit;
|
||||
pub mod branding;
|
||||
pub mod deliverability; // inbuxa: the deliverability check (not a rebuild)
|
||||
pub mod hold;
|
||||
pub mod journal;
|
||||
pub mod lock;
|
||||
pub mod mailflow;
|
||||
pub mod masked_email;
|
||||
pub mod privacy;
|
||||
pub mod security;
|
||||
|
||||
@@ -1,5 +1,5 @@
|
||||
/*
|
||||
* SPDX-FileCopyrightText: 2026 Coffey Labs
|
||||
* SPDX-FileCopyrightText: 2026 Coffey Labs LLC
|
||||
*
|
||||
* SPDX-License-Identifier: AGPL-3.0-only
|
||||
*/
|
||||
@@ -66,6 +66,53 @@ const KIND_DELEGATE: u8 = b'd';
|
||||
/// Most delegates one lock may have (AL-5).
|
||||
pub const MAX_DELEGATES: usize = 10;
|
||||
|
||||
/// Most people one shared mailbox may have (MA-S): a help desk is bigger
|
||||
/// than the handful a departed colleague's mail is handed to.
|
||||
pub const MAX_SHARED_MAILBOX_DELEGATES: usize = 100;
|
||||
|
||||
/// What a lock is for (multi-account spec, MA-S).
|
||||
///
|
||||
/// Both kinds keep receiving mail, can't be signed in to, and are opened by
|
||||
/// delegates through real grants. A shared mailbox is a role address such
|
||||
/// as support@: it needs no reason, holds more people, runs its own Sieve
|
||||
/// replies (an automatic acknowledgement), records only what is sent as it,
|
||||
/// and may only send as its own addresses.
|
||||
#[derive(Debug, Clone, Copy, Default, PartialEq, Eq, Hash, SerdeSerialize, SerdeDeserialize)]
|
||||
#[serde(rename_all = "camelCase")]
|
||||
pub enum Kind {
|
||||
#[default]
|
||||
Lock,
|
||||
SharedMailbox,
|
||||
}
|
||||
|
||||
impl Kind {
|
||||
pub fn as_str(&self) -> &'static str {
|
||||
match self {
|
||||
Kind::Lock => "lock",
|
||||
Kind::SharedMailbox => "sharedMailbox",
|
||||
}
|
||||
}
|
||||
|
||||
pub fn parse(value: &str) -> Option<Self> {
|
||||
match value {
|
||||
"lock" => Some(Kind::Lock),
|
||||
"sharedMailbox" => Some(Kind::SharedMailbox),
|
||||
_ => None,
|
||||
}
|
||||
}
|
||||
|
||||
pub fn is_lock(&self) -> bool {
|
||||
matches!(self, Kind::Lock)
|
||||
}
|
||||
|
||||
pub fn max_delegates(&self) -> usize {
|
||||
match self {
|
||||
Kind::Lock => MAX_DELEGATES,
|
||||
Kind::SharedMailbox => MAX_SHARED_MAILBOX_DELEGATES,
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// What a delegate may do in the locked account (AL-6).
|
||||
#[derive(Debug, Clone, Copy, PartialEq, Eq, Hash, SerdeSerialize, SerdeDeserialize)]
|
||||
#[serde(rename_all = "camelCase")]
|
||||
@@ -189,6 +236,9 @@ pub struct Replaced {
|
||||
#[serde(rename_all = "camelCase")]
|
||||
pub struct Lock {
|
||||
pub account_id: u32,
|
||||
/// Absent on locks written before shared mailboxes existed: a lock.
|
||||
#[serde(default, skip_serializing_if = "Kind::is_lock")]
|
||||
pub kind: Kind,
|
||||
pub reason: String,
|
||||
/// Seconds since the epoch.
|
||||
pub locked_at: u64,
|
||||
@@ -401,8 +451,9 @@ pub async fn all(data: &Store) -> trc::Result<Vec<Lock>> {
|
||||
Ok(locks)
|
||||
}
|
||||
|
||||
/// The accounts delegated to `delegate`, with its delegation in each.
|
||||
pub async fn delegated_to(data: &Store, delegate: u32) -> trc::Result<Vec<(u32, Delegate)>> {
|
||||
/// The accounts delegated to `delegate`, with its delegation in each and
|
||||
/// the kind of lock it is in.
|
||||
pub async fn delegated_to(data: &Store, delegate: u32) -> trc::Result<Vec<(u32, Delegate, Kind)>> {
|
||||
let mut locked = Vec::new();
|
||||
data.iterate(
|
||||
IterateParams::new(
|
||||
@@ -425,7 +476,7 @@ pub async fn delegated_to(data: &Store, delegate: u32) -> trc::Result<Vec<(u32,
|
||||
if let Some(lock) = get(data, account_id).await?
|
||||
&& let Some(delegation) = lock.delegate(delegate)
|
||||
{
|
||||
delegations.push((account_id, delegation.clone()));
|
||||
delegations.push((account_id, delegation.clone(), lock.kind));
|
||||
}
|
||||
}
|
||||
Ok(delegations)
|
||||
@@ -472,6 +523,22 @@ pub async fn remove(data: &Store, lock: &Lock) -> trc::Result<()> {
|
||||
mod tests {
|
||||
use super::*;
|
||||
|
||||
#[test]
|
||||
fn kind_reads_back_and_defaults_to_lock() {
|
||||
// MA-S: a lock stored before shared mailboxes existed has no kind
|
||||
let stored = r#"{"accountId":1,"reason":"r","lockedAt":0,"lockedBy":"admin","delegates":[]}"#;
|
||||
let lock: Lock = serde_json::from_str(stored).unwrap();
|
||||
assert_eq!(lock.kind, Kind::Lock);
|
||||
assert!(!serde_json::to_string(&lock).unwrap().contains("kind"), "a lock is written as before");
|
||||
|
||||
let shared = Lock { kind: Kind::SharedMailbox, ..lock };
|
||||
let written = serde_json::to_string(&shared).unwrap();
|
||||
assert!(written.contains(r#""kind":"sharedMailbox""#), "{written}");
|
||||
assert_eq!(serde_json::from_str::<Lock>(&written).unwrap().kind, Kind::SharedMailbox);
|
||||
assert_eq!(Kind::parse("sharedMailbox"), Some(Kind::SharedMailbox));
|
||||
assert_eq!(Kind::SharedMailbox.max_delegates(), MAX_SHARED_MAILBOX_DELEGATES);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn keys_read_back() {
|
||||
let ValueClass::Any(any) = class(KIND_DELEGATE, &[7, 9]) else {
|
||||
@@ -509,6 +576,7 @@ mod tests {
|
||||
fn lock_with(delegates: Vec<Delegate>, replaced: Vec<Replaced>) -> Lock {
|
||||
Lock {
|
||||
account_id: 1,
|
||||
kind: Kind::Lock,
|
||||
reason: "r".into(),
|
||||
locked_at: 0,
|
||||
locked_by: "admin".into(),
|
||||
@@ -623,6 +691,7 @@ mod tests {
|
||||
fn expired_delegations_grant_nothing() {
|
||||
let lock = Lock {
|
||||
account_id: 1,
|
||||
kind: Kind::Lock,
|
||||
reason: "Left the company".into(),
|
||||
locked_at: 100,
|
||||
locked_by: "admin".into(),
|
||||
|
||||
@@ -0,0 +1,53 @@
|
||||
/*
|
||||
* SPDX-FileCopyrightText: 2026 Coffey Labs LLC
|
||||
*
|
||||
* SPDX-License-Identifier: AGPL-3.0-only
|
||||
*/
|
||||
|
||||
//! The compiled rules, kept per node so a message doesn't read the store.
|
||||
//! A change made on this node applies at once; one made on another node
|
||||
//! within [`TTL`], when the copy here is next refreshed.
|
||||
|
||||
use super::{engine::Compiled, rules};
|
||||
use std::{
|
||||
sync::{Arc, RwLock},
|
||||
time::{Duration, Instant},
|
||||
};
|
||||
use store::Store;
|
||||
|
||||
/// How long a node keeps its copy before reading the rules again.
|
||||
pub const TTL: Duration = Duration::from_secs(30);
|
||||
|
||||
static CACHE: RwLock<Option<(Instant, Arc<Compiled>)>> = RwLock::new(None);
|
||||
|
||||
/// Forgets the copy, so the next message reads the rules again.
|
||||
pub fn invalidate() {
|
||||
if let Ok(mut cache) = CACHE.write() {
|
||||
*cache = None;
|
||||
}
|
||||
}
|
||||
|
||||
/// The enabled rules, compiled. A rule that no longer compiles is left out
|
||||
/// and reported, once per refresh.
|
||||
pub async fn compiled(data: &Store) -> trc::Result<Arc<Compiled>> {
|
||||
if let Ok(cache) = CACHE.read()
|
||||
&& let Some((at, compiled)) = cache.as_ref()
|
||||
&& at.elapsed() < TTL
|
||||
{
|
||||
return Ok(compiled.clone());
|
||||
}
|
||||
let (compiled, skipped) = Compiled::new(&rules::all(data).await?);
|
||||
for (id, reason) in skipped {
|
||||
trc::event!(
|
||||
Store(trc::StoreEvent::DataCorruption),
|
||||
Id = u64::from(id),
|
||||
Reason = reason,
|
||||
Details = "Mail rule skipped: it no longer compiles"
|
||||
);
|
||||
}
|
||||
let compiled = Arc::new(compiled);
|
||||
if let Ok(mut cache) = CACHE.write() {
|
||||
*cache = Some((Instant::now(), compiled.clone()));
|
||||
}
|
||||
Ok(compiled)
|
||||
}
|
||||
@@ -0,0 +1,49 @@
|
||||
/*
|
||||
* SPDX-FileCopyrightText: 2026 Coffey Labs LLC
|
||||
*
|
||||
* SPDX-License-Identifier: AGPL-3.0-only
|
||||
*/
|
||||
|
||||
//! African identifiers (§2.3): South Africa's ID number.
|
||||
|
||||
use super::{Detector, Findings, Region, Strength, checks, valid_short_date};
|
||||
use regex::Regex;
|
||||
use std::sync::LazyLock;
|
||||
|
||||
pub static DETECTORS: &[Detector] = &[Detector::new(
|
||||
"za-id",
|
||||
"South Africa: ID number",
|
||||
Region::Africa,
|
||||
Strength::Checked,
|
||||
za_id,
|
||||
)];
|
||||
|
||||
/// Birth date `YYMMDD`, four digits, citizenship (0, 1 or 2), 8 or 9, a Luhn
|
||||
/// check digit. The date and the two fixed digits make it strong enough to
|
||||
/// count alone.
|
||||
static ZA_ID: LazyLock<Regex> = LazyLock::new(|| {
|
||||
Regex::new(r"\b(\d{2})(\d{2})(\d{2})\d{4}[012][89]\d\b").expect("detector pattern")
|
||||
});
|
||||
|
||||
fn za_id(text: &str, findings: &mut Findings) {
|
||||
for c in ZA_ID.captures_iter(text) {
|
||||
let n = &c[0];
|
||||
let num = |s: &str| s.parse::<u32>().unwrap_or(0);
|
||||
if valid_short_date(num(&c[1]), num(&c[2]), num(&c[3])) && checks::luhn(n) {
|
||||
findings.insert(n);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use crate::mailflow::detectors::by_id;
|
||||
|
||||
#[test]
|
||||
fn south_africa() {
|
||||
let detector = by_id("za-id").unwrap();
|
||||
assert_eq!(detector.count("ID 8001015009087"), 1);
|
||||
assert_eq!(detector.count("8001015009088"), 0);
|
||||
assert_eq!(detector.count("8013015009087"), 0);
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,172 @@
|
||||
/*
|
||||
* SPDX-FileCopyrightText: 2026 Coffey Labs LLC
|
||||
*
|
||||
* SPDX-License-Identifier: AGPL-3.0-only
|
||||
*/
|
||||
|
||||
//! Identifiers from the Americas outside the US and Canada (§2.3): Brazil's
|
||||
//! CPF and CNPJ, and Mexico's CURP.
|
||||
|
||||
use super::{Detector, Findings, Region, Strength, digit_values, valid_short_date, word_near};
|
||||
use regex::Regex;
|
||||
use std::sync::LazyLock;
|
||||
|
||||
pub static DETECTORS: &[Detector] = &[
|
||||
Detector::new(
|
||||
"br-cpf",
|
||||
"Brazil: CPF",
|
||||
Region::Americas,
|
||||
Strength::Checked,
|
||||
br_cpf,
|
||||
),
|
||||
Detector::new(
|
||||
"br-cnpj",
|
||||
"Brazil: CNPJ",
|
||||
Region::Americas,
|
||||
Strength::Checked,
|
||||
br_cnpj,
|
||||
),
|
||||
Detector::new(
|
||||
"mx-curp",
|
||||
"Mexico: CURP",
|
||||
Region::Americas,
|
||||
Strength::Checked,
|
||||
mx_curp,
|
||||
),
|
||||
];
|
||||
|
||||
fn re(pattern: &str) -> Regex {
|
||||
Regex::new(pattern).expect("detector pattern")
|
||||
}
|
||||
|
||||
/// Brazil's mod 11 check digit over `digits` with `weights`.
|
||||
fn br_check(digits: &[u32], weights: &[u32]) -> u32 {
|
||||
match digits.iter().zip(weights).map(|(a, w)| a * w).sum::<u32>() % 11 {
|
||||
0 | 1 => 0,
|
||||
r => 11 - r,
|
||||
}
|
||||
}
|
||||
|
||||
/// `111.444.777-35`, or eleven bare digits.
|
||||
static CPF: LazyLock<Regex> = LazyLock::new(|| re(r"\b\d{3}(\.?)\d{3}(\.?)\d{3}(-?)\d{2}\b"));
|
||||
|
||||
pub fn cpf_valid(n: &str) -> bool {
|
||||
let d = digit_values(n);
|
||||
// A run of one digit passes the arithmetic but is never issued
|
||||
d.len() == 11
|
||||
&& d.iter().any(|x| *x != d[0])
|
||||
&& br_check(&d[..9], &[10, 9, 8, 7, 6, 5, 4, 3, 2]) == d[9]
|
||||
&& br_check(&d[..10], &[11, 10, 9, 8, 7, 6, 5, 4, 3, 2]) == d[10]
|
||||
}
|
||||
|
||||
const CPF_WORDS: &[&str] = &[
|
||||
"cpf",
|
||||
"cadastro de pessoas físicas",
|
||||
"cadastro de pessoa física",
|
||||
];
|
||||
|
||||
fn br_cpf(text: &str, findings: &mut Findings) {
|
||||
for c in CPF.captures_iter(text) {
|
||||
let whole = c.get(0).unwrap();
|
||||
let written = &c[1] == "." && &c[2] == "." && &c[3] == "-";
|
||||
let n: String = whole
|
||||
.as_str()
|
||||
.chars()
|
||||
.filter(char::is_ascii_digit)
|
||||
.collect();
|
||||
if cpf_valid(&n) && (written || word_near(text, whole.start(), whole.end(), CPF_WORDS)) {
|
||||
findings.insert(n);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// `11.222.333/0001-81`, or fourteen bare digits.
|
||||
static CNPJ: LazyLock<Regex> =
|
||||
LazyLock::new(|| re(r"\b\d{2}(\.?)\d{3}(\.?)\d{3}(/?)\d{4}(-?)\d{2}\b"));
|
||||
|
||||
pub fn cnpj_valid(n: &str) -> bool {
|
||||
let d = digit_values(n);
|
||||
d.len() == 14
|
||||
&& d.iter().any(|x| *x != d[0])
|
||||
&& br_check(&d[..12], &[5, 4, 3, 2, 9, 8, 7, 6, 5, 4, 3, 2]) == d[12]
|
||||
&& br_check(&d[..13], &[6, 5, 4, 3, 2, 9, 8, 7, 6, 5, 4, 3, 2]) == d[13]
|
||||
}
|
||||
|
||||
const CNPJ_WORDS: &[&str] = &["cnpj", "cadastro nacional da pessoa jurídica"];
|
||||
|
||||
fn br_cnpj(text: &str, findings: &mut Findings) {
|
||||
for c in CNPJ.captures_iter(text) {
|
||||
let whole = c.get(0).unwrap();
|
||||
let written = &c[1] == "." && &c[2] == "." && &c[3] == "/" && &c[4] == "-";
|
||||
let n: String = whole
|
||||
.as_str()
|
||||
.chars()
|
||||
.filter(char::is_ascii_digit)
|
||||
.collect();
|
||||
if cnpj_valid(&n) && (written || word_near(text, whole.start(), whole.end(), CNPJ_WORDS)) {
|
||||
findings.insert(n);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// Four letters, the birth date, sex (H, M or X), the state, three
|
||||
/// consonants, a character that tells the century apart, the check digit.
|
||||
static CURP: LazyLock<Regex> = LazyLock::new(|| {
|
||||
re(r"(?i)\b[A-Z]{4}(\d{2})(\d{2})(\d{2})[HMX][A-Z]{2}[B-DF-HJ-NP-TV-Z]{3}[A-Z0-9]\d\b")
|
||||
});
|
||||
|
||||
/// RENAPO's check: each character's place in `0-9 A-N Ñ O-Z`, weighted 18
|
||||
/// down to 2; the digit is 10 minus the sum mod 10 (10 becomes 0).
|
||||
pub fn curp_valid(curp: &str) -> bool {
|
||||
const ALPHABET: &str = "0123456789ABCDEFGHIJKLMNÑOPQRSTUVWXYZ";
|
||||
let mut sum = 0u32;
|
||||
for (i, c) in curp.chars().take(17).enumerate() {
|
||||
let Some(value) = ALPHABET.chars().position(|a| a == c) else {
|
||||
return false;
|
||||
};
|
||||
sum += value as u32 * (18 - i as u32);
|
||||
}
|
||||
curp.chars().nth(17).and_then(|c| c.to_digit(10)) == Some((10 - sum % 10) % 10)
|
||||
}
|
||||
|
||||
fn mx_curp(text: &str, findings: &mut Findings) {
|
||||
for c in CURP.captures_iter(text) {
|
||||
let curp = c[0].to_ascii_uppercase();
|
||||
if valid_short_date(num(&c[1]), num(&c[2]), num(&c[3])) && curp_valid(&curp) {
|
||||
findings.insert(curp);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
fn num(s: &str) -> u32 {
|
||||
s.parse().unwrap_or(0)
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use crate::mailflow::detectors::by_id;
|
||||
|
||||
fn count(id: &str, text: &str) -> usize {
|
||||
by_id(id).unwrap().count(text)
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn brazil() {
|
||||
assert_eq!(count("br-cpf", "CPF 111.444.777-35"), 1);
|
||||
assert_eq!(count("br-cpf", "111.444.777-36"), 0);
|
||||
assert_eq!(count("br-cpf", "pedido 11144477735"), 0);
|
||||
assert_eq!(count("br-cpf", "cpf: 11144477735"), 1);
|
||||
assert_eq!(count("br-cpf", "CPF 111.111.111-11"), 0);
|
||||
assert_eq!(count("br-cnpj", "11.222.333/0001-81"), 1);
|
||||
assert_eq!(count("br-cnpj", "11.222.333/0001-82"), 0);
|
||||
assert_eq!(count("br-cnpj", "CNPJ 11222333000181"), 1);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn mexico() {
|
||||
// python-stdnum's documented example
|
||||
assert_eq!(count("mx-curp", "CURP BOXW310820HNERXN09"), 1);
|
||||
assert_eq!(count("mx-curp", "BOXW310820HNERXN08"), 0);
|
||||
assert_eq!(count("mx-curp", "BOXW311320HNERXN09"), 0);
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,511 @@
|
||||
/*
|
||||
* SPDX-FileCopyrightText: 2026 Coffey Labs LLC
|
||||
*
|
||||
* SPDX-License-Identifier: AGPL-3.0-only
|
||||
*/
|
||||
|
||||
//! Detectors that aren't tied to one country (§2.3, region "Any").
|
||||
|
||||
use super::{
|
||||
Detector, Findings, Region, Strength, checks, digits, stands_alone, valid_date, word_near,
|
||||
};
|
||||
use regex::Regex;
|
||||
use std::sync::LazyLock;
|
||||
|
||||
pub static DETECTORS: &[Detector] = &[
|
||||
Detector::new(
|
||||
"payment-card",
|
||||
"Payment card number",
|
||||
Region::Any,
|
||||
Strength::Checked,
|
||||
payment_card,
|
||||
),
|
||||
Detector::new("iban", "IBAN", Region::Any, Strength::Checked, iban),
|
||||
Detector::new(
|
||||
"swift-bic",
|
||||
"SWIFT/BIC code",
|
||||
Region::Any,
|
||||
Strength::NeedsWord,
|
||||
swift_bic,
|
||||
),
|
||||
Detector::new(
|
||||
"email-addresses",
|
||||
"Email addresses",
|
||||
Region::Any,
|
||||
Strength::Checked,
|
||||
email_addresses,
|
||||
),
|
||||
Detector::new(
|
||||
"phone-numbers",
|
||||
"Phone numbers",
|
||||
Region::Any,
|
||||
Strength::NeedsWord,
|
||||
phone_numbers,
|
||||
),
|
||||
Detector::new(
|
||||
"date-of-birth",
|
||||
"Date of birth",
|
||||
Region::Any,
|
||||
Strength::NeedsWord,
|
||||
date_of_birth,
|
||||
),
|
||||
Detector::new(
|
||||
"passport",
|
||||
"Passport number",
|
||||
Region::Any,
|
||||
Strength::NeedsWord,
|
||||
passport,
|
||||
),
|
||||
Detector::new(
|
||||
"private-key",
|
||||
"Private key",
|
||||
Region::Any,
|
||||
Strength::Checked,
|
||||
private_key,
|
||||
),
|
||||
Detector::new(
|
||||
"credentials",
|
||||
"Cloud and service credentials",
|
||||
Region::Any,
|
||||
Strength::Checked,
|
||||
credentials,
|
||||
),
|
||||
];
|
||||
|
||||
fn re(pattern: &str) -> Regex {
|
||||
Regex::new(pattern).expect("detector pattern")
|
||||
}
|
||||
|
||||
// --- Payment cards --------------------------------------------------------
|
||||
|
||||
/// Issuer prefixes (ISO/IEC 7812 IINs) and the lengths each network issues.
|
||||
fn card_network(number: &str) -> bool {
|
||||
let len = number.len();
|
||||
let prefix = |n: usize| number[..n].parse::<u32>().unwrap_or(0);
|
||||
match number.as_bytes()[0] {
|
||||
// Visa
|
||||
b'4' => matches!(len, 13 | 16 | 19),
|
||||
b'5' => {
|
||||
// Mastercard 51–55; Maestro 50, 56–58
|
||||
(51..=55).contains(&prefix(2)) && len == 16
|
||||
|| matches!(prefix(2), 50 | 56..=58) && (12..=19).contains(&len)
|
||||
}
|
||||
// Mastercard 2221–2720
|
||||
b'2' => (2221..=2720).contains(&prefix(4)) && len == 16,
|
||||
b'3' => {
|
||||
// American Express 34, 37; JCB 3528–3589; Diners 300–305, 36, 38, 39
|
||||
matches!(prefix(2), 34 | 37) && len == 15
|
||||
|| (3528..=3589).contains(&prefix(4)) && (16..=19).contains(&len)
|
||||
|| ((300..=305).contains(&prefix(3)) || matches!(prefix(2), 36 | 38 | 39))
|
||||
&& (14..=19).contains(&len)
|
||||
}
|
||||
// Discover 6011, 644–649, 65; UnionPay 62; Maestro 6x
|
||||
b'6' => (12..=19).contains(&len),
|
||||
_ => false,
|
||||
}
|
||||
}
|
||||
|
||||
fn is_card(number: &str) -> bool {
|
||||
(12..=19).contains(&number.len()) && card_network(number) && checks::luhn(number)
|
||||
}
|
||||
|
||||
static CARD: LazyLock<Regex> = LazyLock::new(|| re(r"\b\d(?:[ -]?\d){11,18}\b"));
|
||||
|
||||
fn payment_card(text: &str, findings: &mut Findings) {
|
||||
for m in CARD.find_iter(text) {
|
||||
if !stands_alone(text, m.start(), m.end()) {
|
||||
continue;
|
||||
}
|
||||
let whole = digits(m.as_str());
|
||||
if is_card(&whole) {
|
||||
findings.insert(whole);
|
||||
continue;
|
||||
}
|
||||
// Two numbers side by side ("4242 4242 4242 4242 2031"): try each
|
||||
// run of whole groups
|
||||
let groups: Vec<String> = m.as_str().split([' ', '-']).map(digits).collect();
|
||||
'runs: for from in 0..groups.len() {
|
||||
let mut number = String::new();
|
||||
for group in &groups[from..] {
|
||||
number.push_str(group);
|
||||
if is_card(&number) {
|
||||
findings.insert(number);
|
||||
break 'runs;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// --- IBAN -----------------------------------------------------------------
|
||||
|
||||
static IBAN: LazyLock<Regex> =
|
||||
LazyLock::new(|| re(r"\b[A-Za-z]{2}\d{2}(?:[ ]?[A-Za-z0-9]){11,30}"));
|
||||
|
||||
fn iban(text: &str, findings: &mut Findings) {
|
||||
// The pattern can run on into the next words, even the next IBAN: after
|
||||
// each hit, look again from where that IBAN ended
|
||||
let mut from = 0;
|
||||
while let Some(m) = IBAN.find_at(text, from) {
|
||||
from = m.start() + 1;
|
||||
let compact = m.as_str().replace(' ', "").to_ascii_uppercase();
|
||||
let Some(len) = checks::iban_length(&compact[..2]) else {
|
||||
continue;
|
||||
};
|
||||
if compact.len() < len {
|
||||
continue;
|
||||
}
|
||||
// Where the country's length ends in the text, spaces counted
|
||||
let mut seen = 0;
|
||||
let Some(end) = m
|
||||
.as_str()
|
||||
.char_indices()
|
||||
.find(|(_, c)| {
|
||||
if *c != ' ' {
|
||||
seen += 1;
|
||||
}
|
||||
seen == len
|
||||
})
|
||||
.map(|(i, c)| m.start() + i + c.len_utf8())
|
||||
else {
|
||||
continue;
|
||||
};
|
||||
let candidate = &compact[..len];
|
||||
if stands_alone(text, m.start(), end) && checks::iban(candidate) {
|
||||
findings.insert(candidate);
|
||||
from = end;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// --- SWIFT/BIC ------------------------------------------------------------
|
||||
|
||||
static BIC: LazyLock<Regex> =
|
||||
LazyLock::new(|| re(r"\b[A-Z]{4}[A-Z]{2}[A-Z0-9]{2}(?:[A-Z0-9]{3})?\b"));
|
||||
|
||||
const BIC_WORDS: &[&str] = &[
|
||||
"swift",
|
||||
"bic",
|
||||
"swift/bic",
|
||||
"bank",
|
||||
"banque",
|
||||
"bankverbindung",
|
||||
];
|
||||
|
||||
fn swift_bic(text: &str, findings: &mut Findings) {
|
||||
for m in BIC.find_iter(text) {
|
||||
let code = m.as_str();
|
||||
if checks::is_country(&code[4..6]) && word_near(text, m.start(), m.end(), BIC_WORDS) {
|
||||
findings.insert(code);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// --- Contact lists --------------------------------------------------------
|
||||
|
||||
static EMAIL: LazyLock<Regex> =
|
||||
LazyLock::new(|| re(r"(?i)\b[a-z0-9._%+-]+@[a-z0-9-]+(?:\.[a-z0-9-]+)*\.[a-z]{2,}\b"));
|
||||
|
||||
fn email_addresses(text: &str, findings: &mut Findings) {
|
||||
for m in EMAIL.find_iter(text) {
|
||||
findings.insert(m.as_str().to_lowercase());
|
||||
}
|
||||
}
|
||||
|
||||
/// International form: found alone. National form: only with a word.
|
||||
static PHONE_INTL: LazyLock<Regex> = LazyLock::new(|| re(r"\+\d{1,3}(?:[ .-]?\(?\d{1,4}\)?){2,5}"));
|
||||
static PHONE_NATIONAL: LazyLock<Regex> =
|
||||
LazyLock::new(|| re(r"\(?\d{2,4}\)?[ .-]\d{3,4}[ .-]\d{3,4}"));
|
||||
|
||||
const PHONE_WORDS: &[&str] = &[
|
||||
"phone",
|
||||
"tel",
|
||||
"telephone",
|
||||
"mobile",
|
||||
"cell",
|
||||
"fax",
|
||||
"telefon",
|
||||
"téléphone",
|
||||
"teléfono",
|
||||
"telefono",
|
||||
"handy",
|
||||
"portable",
|
||||
"móvil",
|
||||
"cellulare",
|
||||
"mobiel",
|
||||
];
|
||||
|
||||
fn phone_numbers(text: &str, findings: &mut Findings) {
|
||||
let mut international = Vec::new();
|
||||
for m in PHONE_INTL.find_iter(text) {
|
||||
let number = digits(m.as_str());
|
||||
if (8..=15).contains(&number.len()) && stands_alone(text, m.start() + 1, m.end()) {
|
||||
findings.insert(number);
|
||||
international.push(m.range());
|
||||
}
|
||||
}
|
||||
for m in PHONE_NATIONAL.find_iter(text) {
|
||||
let number = digits(m.as_str());
|
||||
// Not the tail of an international number already counted
|
||||
if international.iter().any(|r| r.contains(&m.start())) {
|
||||
continue;
|
||||
}
|
||||
if (9..=11).contains(&number.len())
|
||||
&& stands_alone(text, m.start(), m.end())
|
||||
&& !text[..m.start()].ends_with('+')
|
||||
&& word_near(text, m.start(), m.end(), PHONE_WORDS)
|
||||
{
|
||||
findings.insert(number);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// --- Date of birth --------------------------------------------------------
|
||||
|
||||
static DATE_ISO: LazyLock<Regex> = LazyLock::new(|| re(r"\b(\d{4})-(\d{2})-(\d{2})\b"));
|
||||
static DATE_NUMERIC: LazyLock<Regex> =
|
||||
LazyLock::new(|| re(r"\b(\d{1,2})[./-](\d{1,2})[./-](\d{4})\b"));
|
||||
static DATE_WORDS: LazyLock<Regex> = LazyLock::new(|| {
|
||||
re(
|
||||
r"(?i)\b(?:(\d{1,2})\s+(jan|feb|mar|apr|may|jun|jul|aug|sep|oct|nov|dec)[a-z]*\.?,?\s+(\d{4})|(jan|feb|mar|apr|may|jun|jul|aug|sep|oct|nov|dec)[a-z]*\.?\s+(\d{1,2}),?\s+(\d{4}))\b",
|
||||
)
|
||||
});
|
||||
|
||||
const BIRTH_WORDS: &[&str] = &[
|
||||
"born",
|
||||
"birth",
|
||||
"dob",
|
||||
"d.o.b",
|
||||
"birthday",
|
||||
"birthdate",
|
||||
"geburtsdatum",
|
||||
"geboren",
|
||||
"naissance",
|
||||
"né le",
|
||||
"née le",
|
||||
"nacimiento",
|
||||
"nacido",
|
||||
"nacida",
|
||||
"nascita",
|
||||
"nato il",
|
||||
"nata il",
|
||||
"geboortedatum",
|
||||
"födelsedatum",
|
||||
"fødselsdato",
|
||||
"syntymäaika",
|
||||
"urodzenia",
|
||||
"nascimento",
|
||||
];
|
||||
|
||||
fn month_number(name: &str) -> u32 {
|
||||
const MONTHS: [&str; 12] = [
|
||||
"jan", "feb", "mar", "apr", "may", "jun", "jul", "aug", "sep", "oct", "nov", "dec",
|
||||
];
|
||||
let name = name.to_lowercase();
|
||||
MONTHS
|
||||
.iter()
|
||||
.position(|m| *m == name)
|
||||
.map_or(0, |i| i as u32 + 1)
|
||||
}
|
||||
|
||||
fn date_of_birth(text: &str, findings: &mut Findings) {
|
||||
let mut add = |start: usize, end: usize, key: String| {
|
||||
if word_near(text, start, end, BIRTH_WORDS) {
|
||||
findings.insert(key);
|
||||
}
|
||||
};
|
||||
let num = |s: &str| s.parse::<u32>().unwrap_or(0);
|
||||
for c in DATE_ISO.captures_iter(text) {
|
||||
let (y, m, d) = (num(&c[1]), num(&c[2]), num(&c[3]));
|
||||
let whole = c.get(0).unwrap();
|
||||
if valid_date(y, m, d) {
|
||||
add(whole.start(), whole.end(), format!("{y:04}{m:02}{d:02}"));
|
||||
}
|
||||
}
|
||||
for c in DATE_NUMERIC.captures_iter(text) {
|
||||
let (a, b, y) = (num(&c[1]), num(&c[2]), num(&c[3]));
|
||||
let whole = c.get(0).unwrap();
|
||||
// Day first or month first: either reading that is a real date
|
||||
if valid_date(y, b, a) || valid_date(y, a, b) {
|
||||
add(whole.start(), whole.end(), whole.as_str().to_string());
|
||||
}
|
||||
}
|
||||
for c in DATE_WORDS.captures_iter(text) {
|
||||
let whole = c.get(0).unwrap();
|
||||
let (d, m, y) = match (c.get(1), c.get(4)) {
|
||||
(Some(d), _) => (num(d.as_str()), month_number(&c[2]), num(&c[3])),
|
||||
(_, Some(m)) => (num(&c[5]), month_number(m.as_str()), num(&c[6])),
|
||||
_ => continue,
|
||||
};
|
||||
if valid_date(y, m, d) {
|
||||
add(whole.start(), whole.end(), format!("{y:04}{m:02}{d:02}"));
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// --- Passport -------------------------------------------------------------
|
||||
|
||||
static PASSPORT: LazyLock<Regex> = LazyLock::new(|| re(r"\b[A-Z0-9]{6,9}\b"));
|
||||
|
||||
const PASSPORT_WORDS: &[&str] = &[
|
||||
"passport",
|
||||
"passeport",
|
||||
"reisepass",
|
||||
"pasaporte",
|
||||
"passaporto",
|
||||
"paspoort",
|
||||
"passnummer",
|
||||
"pass-nr",
|
||||
"passport no",
|
||||
"pasaporte n.º",
|
||||
"passaporte",
|
||||
];
|
||||
|
||||
fn passport(text: &str, findings: &mut Findings) {
|
||||
for m in PASSPORT.find_iter(text) {
|
||||
let value = m.as_str();
|
||||
if value.bytes().filter(u8::is_ascii_digit).count() >= 5
|
||||
&& word_near(text, m.start(), m.end(), PASSPORT_WORDS)
|
||||
{
|
||||
findings.insert(value);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// --- Keys and credentials -------------------------------------------------
|
||||
|
||||
static PRIVATE_KEY: LazyLock<Regex> = LazyLock::new(|| {
|
||||
re(
|
||||
r"-----BEGIN (?:(?:RSA|EC|DSA|OPENSSH|ENCRYPTED|PGP) )?PRIVATE KEY(?: BLOCK)?-----\s*([A-Za-z0-9+/=:\s-]{0,64})",
|
||||
)
|
||||
});
|
||||
|
||||
fn private_key(text: &str, findings: &mut Findings) {
|
||||
for c in PRIVATE_KEY.captures_iter(text) {
|
||||
// Each key once, by the start of its body
|
||||
let body: String = c[1].chars().filter(|c| !c.is_whitespace()).collect();
|
||||
let whole = c.get(0).unwrap();
|
||||
findings.insert(if body.is_empty() {
|
||||
format!("@{}", whole.start())
|
||||
} else {
|
||||
body
|
||||
});
|
||||
}
|
||||
}
|
||||
|
||||
/// Published token formats: AWS access key IDs, GitHub tokens, Slack
|
||||
/// tokens, Stripe live secret and restricted keys, Google API keys.
|
||||
static CREDENTIAL: LazyLock<Regex> = LazyLock::new(|| {
|
||||
re(concat!(
|
||||
r"\b(?:",
|
||||
r"(?:AKIA|ASIA|ABIA|ACCA)[A-Z0-9]{16}",
|
||||
r"|gh[pousr]_[A-Za-z0-9]{36}",
|
||||
r"|github_pat_[A-Za-z0-9_]{82}",
|
||||
r"|xox[abposr]-[A-Za-z0-9-]{10,72}",
|
||||
r"|(?:sk|rk)_live_[A-Za-z0-9]{24,99}",
|
||||
r"|AIza[0-9A-Za-z_-]{35}",
|
||||
r")\b"
|
||||
))
|
||||
});
|
||||
|
||||
fn credentials(text: &str, findings: &mut Findings) {
|
||||
for m in CREDENTIAL.find_iter(text) {
|
||||
findings.insert(m.as_str());
|
||||
}
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use crate::mailflow::detectors::by_id;
|
||||
|
||||
fn count(id: &str, text: &str) -> usize {
|
||||
by_id(id).unwrap().count(text)
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn payment_cards() {
|
||||
// Networks' and processors' published test numbers
|
||||
let text = "Visa 4242 4242 4242 4242, MC 5555-5555-5555-4444, Amex 378282246310005, \
|
||||
Discover 6011111111111117, JCB 3566002020360505, Diners 30569309025904, \
|
||||
UnionPay 6200000000000005, Mastercard 2-series 2223003122003222";
|
||||
assert_eq!(count("payment-card", text), 8);
|
||||
// Luhn fails, wrong network length, inside a longer number
|
||||
assert_eq!(count("payment-card", "4242424242424241"), 0);
|
||||
assert_eq!(count("payment-card", "378282246310005 0"), 1);
|
||||
assert_eq!(count("payment-card", "order 94242424242424242 shipped"), 0);
|
||||
// The same number twice counts once
|
||||
assert_eq!(
|
||||
count("payment-card", "4242424242424242 and 4242-4242-4242-4242"),
|
||||
1
|
||||
);
|
||||
// A card followed by a year
|
||||
assert_eq!(count("payment-card", "card 4242 4242 4242 4242 2031"), 1);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn ibans() {
|
||||
let text =
|
||||
"Pay GB29 NWBK 6016 1331 9268 19 or de89370400440532013000 (NL91ABNA0417164300).";
|
||||
assert_eq!(count("iban", text), 3);
|
||||
assert_eq!(count("iban", "GB29 NWBK 6016 1331 9268 18"), 0);
|
||||
// Runs into the next word: still found at the country's length
|
||||
assert_eq!(count("iban", "IBAN NL91ABNA0417164300 BIC ABNANL2A"), 1);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn swift_codes_need_a_word() {
|
||||
assert_eq!(count("swift-bic", "SWIFT: DEUTDEFF500"), 1);
|
||||
assert_eq!(count("swift-bic", "BIC NWBKGB2L"), 1);
|
||||
assert_eq!(count("swift-bic", "HAPPYDAYS DEUTDEFF"), 0);
|
||||
// Not a country in positions 5–6
|
||||
assert_eq!(count("swift-bic", "BIC DEUTZZFF"), 0);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn email_and_phone_lists() {
|
||||
let list = "[email protected], [email protected], [email protected], [email protected]";
|
||||
assert_eq!(count("email-addresses", list), 3);
|
||||
assert_eq!(
|
||||
count("phone-numbers", "+44 20 7946 0958, +1 (415) 555-2671"),
|
||||
2
|
||||
);
|
||||
assert_eq!(count("phone-numbers", "call 020 7946 0958"), 0);
|
||||
assert_eq!(count("phone-numbers", "Tel: 020 7946 0958"), 1);
|
||||
assert_eq!(count("phone-numbers", "invoice 020 7946 0958"), 0);
|
||||
// One number, not also its national tail
|
||||
assert_eq!(count("phone-numbers", "Tel: +44 20 7946 0958"), 1);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn dates_of_birth() {
|
||||
assert_eq!(count("date-of-birth", "DOB: 1984-02-29"), 1);
|
||||
assert_eq!(count("date-of-birth", "Geburtsdatum 31.12.1970"), 1);
|
||||
assert_eq!(count("date-of-birth", "born on March 3, 1962"), 1);
|
||||
assert_eq!(count("date-of-birth", "date of birth 3 Mar 1962"), 1);
|
||||
// Not a real date, no word, a meeting
|
||||
assert_eq!(count("date-of-birth", "DOB: 1985-02-29"), 0);
|
||||
assert_eq!(count("date-of-birth", "invoice 1984-02-29"), 0);
|
||||
assert_eq!(count("date-of-birth", "Meeting on 12/05/2026"), 0);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn passports_need_a_word() {
|
||||
assert_eq!(count("passport", "Passport number: 533380006"), 1);
|
||||
assert_eq!(count("passport", "Reisepass C01X00T47"), 1);
|
||||
assert_eq!(count("passport", "Order 533380006 shipped"), 0);
|
||||
// Mostly letters: a word, not a number
|
||||
assert_eq!(count("passport", "passport PASSWORD"), 0);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn keys_and_credentials() {
|
||||
let key = "-----BEGIN OPENSSH PRIVATE KEY-----\nb3BlbnNzaC1rZXktdjEAAAAABG5vbmUAAAAEbm9uZQ\n-----END OPENSSH PRIVATE KEY-----";
|
||||
assert_eq!(count("private-key", key), 1);
|
||||
assert_eq!(count("private-key", "-----BEGIN PUBLIC KEY-----\nMFkw"), 0);
|
||||
// Documentation examples of each format
|
||||
let tokens = "AKIAIOSFODNN7EXAMPLE ghp_0123456789abcdefghijklmnopqrstuvwxyz \
|
||||
AIzaSyA-0123456789abcdefghijklmnopqrstu";
|
||||
assert_eq!(count("credentials", tokens), 3);
|
||||
assert_eq!(count("credentials", "AKIA123 ghp_short"), 0);
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,281 @@
|
||||
/*
|
||||
* SPDX-FileCopyrightText: 2026 Coffey Labs LLC
|
||||
*
|
||||
* SPDX-License-Identifier: AGPL-3.0-only
|
||||
*/
|
||||
|
||||
//! Asian identifiers (§2.3): India's Aadhaar and PAN, China's resident ID,
|
||||
//! Japan's My Number, Singapore's NRIC and FIN, and South Korea's resident
|
||||
//! registration number.
|
||||
|
||||
use super::{
|
||||
Detector, Findings, Region, Strength, digit_values, valid_date, valid_short_date, word_near,
|
||||
};
|
||||
use regex::Regex;
|
||||
use std::sync::LazyLock;
|
||||
|
||||
pub static DETECTORS: &[Detector] = &[
|
||||
Detector::new(
|
||||
"in-aadhaar",
|
||||
"India: Aadhaar",
|
||||
Region::Asia,
|
||||
Strength::Checked,
|
||||
in_aadhaar,
|
||||
),
|
||||
Detector::new(
|
||||
"in-pan",
|
||||
"India: PAN",
|
||||
Region::Asia,
|
||||
Strength::NeedsWord,
|
||||
in_pan,
|
||||
),
|
||||
Detector::new(
|
||||
"cn-resident-id",
|
||||
"China: resident ID",
|
||||
Region::Asia,
|
||||
Strength::Checked,
|
||||
cn_resident_id,
|
||||
),
|
||||
Detector::new(
|
||||
"jp-my-number",
|
||||
"Japan: My Number",
|
||||
Region::Asia,
|
||||
Strength::Checked,
|
||||
jp_my_number,
|
||||
),
|
||||
Detector::new(
|
||||
"sg-nric",
|
||||
"Singapore: NRIC and FIN",
|
||||
Region::Asia,
|
||||
Strength::Checked,
|
||||
sg_nric,
|
||||
),
|
||||
Detector::new(
|
||||
"kr-rrn",
|
||||
"South Korea: resident registration number",
|
||||
Region::Asia,
|
||||
Strength::NeedsWord,
|
||||
kr_rrn,
|
||||
),
|
||||
];
|
||||
|
||||
fn re(pattern: &str) -> Regex {
|
||||
Regex::new(pattern).expect("detector pattern")
|
||||
}
|
||||
|
||||
/// Twelve digits written in fours, or bare.
|
||||
static TWELVE: LazyLock<Regex> = LazyLock::new(|| re(r"\b(\d{4})( ?)(\d{4})( ?)(\d{4})\b"));
|
||||
|
||||
const VERHOEFF_D: [[u8; 10]; 10] = [
|
||||
[0, 1, 2, 3, 4, 5, 6, 7, 8, 9],
|
||||
[1, 2, 3, 4, 0, 6, 7, 8, 9, 5],
|
||||
[2, 3, 4, 0, 1, 7, 8, 9, 5, 6],
|
||||
[3, 4, 0, 1, 2, 8, 9, 5, 6, 7],
|
||||
[4, 0, 1, 2, 3, 9, 5, 6, 7, 8],
|
||||
[5, 9, 8, 7, 6, 0, 4, 3, 2, 1],
|
||||
[6, 5, 9, 8, 7, 1, 0, 4, 3, 2],
|
||||
[7, 6, 5, 9, 8, 2, 1, 0, 4, 3],
|
||||
[8, 7, 6, 5, 9, 3, 2, 1, 0, 4],
|
||||
[9, 8, 7, 6, 5, 4, 3, 2, 1, 0],
|
||||
];
|
||||
const VERHOEFF_P: [[u8; 10]; 8] = [
|
||||
[0, 1, 2, 3, 4, 5, 6, 7, 8, 9],
|
||||
[1, 5, 7, 6, 2, 8, 3, 0, 9, 4],
|
||||
[5, 8, 0, 3, 7, 9, 6, 1, 4, 2],
|
||||
[8, 9, 1, 6, 0, 4, 3, 5, 2, 7],
|
||||
[9, 4, 5, 3, 1, 2, 6, 8, 7, 0],
|
||||
[4, 2, 8, 6, 5, 7, 3, 9, 0, 1],
|
||||
[2, 7, 9, 3, 8, 0, 6, 4, 1, 5],
|
||||
[7, 0, 4, 6, 9, 1, 3, 2, 5, 8],
|
||||
];
|
||||
|
||||
/// The Verhoeff check (dihedral group D5).
|
||||
pub fn verhoeff(n: &str) -> bool {
|
||||
let mut c = 0u8;
|
||||
for (i, b) in n.bytes().rev().enumerate() {
|
||||
c = VERHOEFF_D[c as usize][VERHOEFF_P[i % 8][(b - b'0') as usize] as usize];
|
||||
}
|
||||
c == 0
|
||||
}
|
||||
|
||||
const AADHAAR_WORDS: &[&str] = &["aadhaar", "aadhar", "uidai", "uid"];
|
||||
|
||||
fn in_aadhaar(text: &str, findings: &mut Findings) {
|
||||
for c in TWELVE.captures_iter(text) {
|
||||
let whole = c.get(0).unwrap();
|
||||
let n = format!("{}{}{}", &c[1], &c[3], &c[5]);
|
||||
let written = &c[2] == " " && &c[4] == " ";
|
||||
// Never starts with 0 or 1
|
||||
if !n.starts_with(['0', '1'])
|
||||
&& verhoeff(&n)
|
||||
&& (written || word_near(text, whole.start(), whole.end(), AADHAAR_WORDS))
|
||||
{
|
||||
findings.insert(n);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// Five letters (the fourth names the holder's type), four digits, a letter.
|
||||
static PAN: LazyLock<Regex> = LazyLock::new(|| re(r"\b[A-Z]{3}[ABCFGHLJPTK][A-Z]\d{4}[A-Z]\b"));
|
||||
|
||||
const PAN_WORDS: &[&str] = &["pan", "pan card", "permanent account number", "income tax"];
|
||||
|
||||
fn in_pan(text: &str, findings: &mut Findings) {
|
||||
for m in PAN.find_iter(text) {
|
||||
if word_near(text, m.start(), m.end(), PAN_WORDS) {
|
||||
findings.insert(m.as_str());
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// Region, birth date `YYYYMMDD`, sequence, then the ISO 7064 MOD 11-2
|
||||
/// check (0–9 or X).
|
||||
static CN_ID: LazyLock<Regex> =
|
||||
LazyLock::new(|| re(r"(?i)\b[1-8]\d{5}(\d{4})(\d{2})(\d{2})\d{3}[\dX]\b"));
|
||||
|
||||
pub fn cn_id_valid(id: &str) -> bool {
|
||||
const WEIGHTS: [u32; 17] = [7, 9, 10, 5, 8, 4, 2, 1, 6, 3, 7, 9, 10, 5, 8, 4, 2];
|
||||
const CHECKS: &[u8] = b"10X98765432";
|
||||
let sum: u32 = digit_values(&id[..17])
|
||||
.iter()
|
||||
.zip(WEIGHTS)
|
||||
.map(|(a, w)| a * w)
|
||||
.sum();
|
||||
CHECKS[(sum % 11) as usize] == id.as_bytes()[17].to_ascii_uppercase()
|
||||
}
|
||||
|
||||
fn cn_resident_id(text: &str, findings: &mut Findings) {
|
||||
for c in CN_ID.captures_iter(text) {
|
||||
let id = c[0].to_ascii_uppercase();
|
||||
let (y, m, d) = (num(&c[1]), num(&c[2]), num(&c[3]));
|
||||
if valid_date(y, m, d) && cn_id_valid(&id) {
|
||||
findings.insert(id);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// My Number: weights 2–7 then 2–6 from the right; a remainder of 0 or 1
|
||||
/// gives 0, else 11 minus it.
|
||||
pub fn my_number_valid(n: &str) -> bool {
|
||||
let d = digit_values(n);
|
||||
let sum: u32 = (1..=11)
|
||||
.map(|i| d[11 - i] * if i <= 6 { i as u32 + 1 } else { i as u32 - 5 })
|
||||
.sum();
|
||||
let check = match sum % 11 {
|
||||
0 | 1 => 0,
|
||||
r => 11 - r,
|
||||
};
|
||||
check == d[11]
|
||||
}
|
||||
|
||||
const MY_NUMBER_WORDS: &[&str] = &[
|
||||
"my number",
|
||||
"mynumber",
|
||||
"マイナンバー",
|
||||
"個人番号",
|
||||
"kojin bango",
|
||||
];
|
||||
|
||||
fn jp_my_number(text: &str, findings: &mut Findings) {
|
||||
for c in TWELVE.captures_iter(text) {
|
||||
let whole = c.get(0).unwrap();
|
||||
let n = format!("{}{}{}", &c[1], &c[3], &c[5]);
|
||||
let written = &c[2] == " " && &c[4] == " ";
|
||||
if my_number_valid(&n)
|
||||
&& (written || word_near(text, whole.start(), whole.end(), MY_NUMBER_WORDS))
|
||||
{
|
||||
findings.insert(n);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
static NRIC: LazyLock<Regex> = LazyLock::new(|| re(r"(?i)\b([STFGM])(\d{7})([A-Z])\b"));
|
||||
|
||||
/// Weights 2, 7, 6, 5, 4, 3, 2; T and G add 4, M adds 3; each series has its
|
||||
/// own table of check letters.
|
||||
fn nric_valid(prefix: u8, digits: &str, check: u8) -> bool {
|
||||
let sum: u32 = digit_values(digits)
|
||||
.iter()
|
||||
.zip([2, 7, 6, 5, 4, 3, 2])
|
||||
.map(|(a, w)| a * w)
|
||||
.sum::<u32>()
|
||||
+ match prefix {
|
||||
b'T' | b'G' => 4,
|
||||
b'M' => 3,
|
||||
_ => 0,
|
||||
};
|
||||
let table: &[u8] = match prefix {
|
||||
b'S' | b'T' => b"JZIHGFEDCBA",
|
||||
b'F' | b'G' => b"XWUTRQPNMLK",
|
||||
_ => b"KLJNPQRTUWX",
|
||||
};
|
||||
table[(sum % 11) as usize] == check
|
||||
}
|
||||
|
||||
fn sg_nric(text: &str, findings: &mut Findings) {
|
||||
for c in NRIC.captures_iter(text) {
|
||||
let id = c[0].to_ascii_uppercase();
|
||||
let bytes = id.as_bytes();
|
||||
if nric_valid(bytes[0], &c[2], bytes[8]) {
|
||||
findings.insert(id);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// `YYMMDD-GNNNNNN`, the seventh digit giving sex and century.
|
||||
static RRN: LazyLock<Regex> = LazyLock::new(|| re(r"\b(\d{2})(\d{2})(\d{2})-?([1-8])\d{6}\b"));
|
||||
|
||||
const RRN_WORDS: &[&str] = &["주민등록번호", "주민번호", "resident registration", "rrn"];
|
||||
|
||||
fn kr_rrn(text: &str, findings: &mut Findings) {
|
||||
for c in RRN.captures_iter(text) {
|
||||
let whole = c.get(0).unwrap();
|
||||
if valid_short_date(num(&c[1]), num(&c[2]), num(&c[3]))
|
||||
&& word_near(text, whole.start(), whole.end(), RRN_WORDS)
|
||||
{
|
||||
findings.insert(whole.as_str().replace('-', ""));
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
fn num(s: &str) -> u32 {
|
||||
s.parse().unwrap_or(0)
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use crate::mailflow::detectors::by_id;
|
||||
|
||||
fn count(id: &str, text: &str) -> usize {
|
||||
by_id(id).unwrap().count(text)
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn india() {
|
||||
assert_eq!(count("in-aadhaar", "2345 6789 0124"), 1);
|
||||
assert_eq!(count("in-aadhaar", "2345 6789 0125"), 0);
|
||||
assert_eq!(count("in-aadhaar", "order 234567890124"), 0);
|
||||
assert_eq!(count("in-aadhaar", "Aadhaar 234567890124"), 1);
|
||||
assert_eq!(count("in-pan", "PAN: ABCPE1234F"), 1);
|
||||
assert_eq!(count("in-pan", "ABCPE1234F"), 0);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn china_japan() {
|
||||
assert_eq!(count("cn-resident-id", "11010519491231002X"), 1);
|
||||
assert_eq!(count("cn-resident-id", "110105194912310021"), 0);
|
||||
assert_eq!(count("cn-resident-id", "11010519491331002X"), 0);
|
||||
assert_eq!(count("jp-my-number", "1234 5678 9018"), 1);
|
||||
assert_eq!(count("jp-my-number", "1234 5678 9017"), 0);
|
||||
assert_eq!(count("jp-my-number", "マイナンバー 123456789018"), 1);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn singapore_korea() {
|
||||
assert_eq!(count("sg-nric", "S1234567D and T1234567J"), 2);
|
||||
assert_eq!(count("sg-nric", "S1234567E"), 0);
|
||||
assert_eq!(count("kr-rrn", "주민등록번호 800101-1234567"), 1);
|
||||
assert_eq!(count("kr-rrn", "800101-1234567"), 0);
|
||||
assert_eq!(count("kr-rrn", "RRN 801301-1234567"), 0);
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,128 @@
|
||||
/*
|
||||
* SPDX-FileCopyrightText: 2026 Coffey Labs LLC
|
||||
*
|
||||
* SPDX-License-Identifier: AGPL-3.0-only
|
||||
*/
|
||||
|
||||
//! Australian identifiers (§2.3): the ATO's Tax File Number and the Medicare
|
||||
//! card number.
|
||||
|
||||
use super::{Detector, Findings, Region, Strength, word_near};
|
||||
use regex::Regex;
|
||||
use std::sync::LazyLock;
|
||||
|
||||
pub static DETECTORS: &[Detector] = &[
|
||||
Detector::new(
|
||||
"au-tfn",
|
||||
"Australian Tax File Number",
|
||||
Region::Australia,
|
||||
Strength::Checked,
|
||||
tfn,
|
||||
),
|
||||
Detector::new(
|
||||
"au-medicare",
|
||||
"Australian Medicare number",
|
||||
Region::Australia,
|
||||
Strength::Checked,
|
||||
medicare,
|
||||
),
|
||||
];
|
||||
|
||||
fn re(pattern: &str) -> Regex {
|
||||
Regex::new(pattern).expect("detector pattern")
|
||||
}
|
||||
|
||||
/// `NNN NNN NNN` stands alone; bare digits (eight or nine) need a word.
|
||||
static TFN: LazyLock<Regex> = LazyLock::new(|| re(r"\b(\d{3})( ?)(\d{3})( ?)(\d{2,3})\b"));
|
||||
|
||||
/// Weighted sum mod 11, with the ATO's weights for 9- and 8-digit numbers.
|
||||
pub fn tfn_valid(n: &str) -> bool {
|
||||
let weights: &[u32] = match n.len() {
|
||||
9 => &[1, 4, 3, 7, 5, 8, 6, 9, 10],
|
||||
8 => &[10, 7, 8, 4, 6, 3, 5, 1],
|
||||
_ => return false,
|
||||
};
|
||||
n.bytes()
|
||||
.zip(weights)
|
||||
.map(|(b, w)| u32::from(b - b'0') * w)
|
||||
.sum::<u32>()
|
||||
% 11
|
||||
== 0
|
||||
}
|
||||
|
||||
const TFN_WORDS: &[&str] = &["tfn", "tax file number", "tax file no"];
|
||||
|
||||
fn tfn(text: &str, findings: &mut Findings) {
|
||||
for c in TFN.captures_iter(text) {
|
||||
let whole = c.get(0).unwrap();
|
||||
let n = format!("{}{}{}", &c[1], &c[3], &c[5]);
|
||||
let written = n.len() == 9 && c[2] == *" " && c[4] == *" ";
|
||||
if tfn_valid(&n) && (written || word_near(text, whole.start(), whole.end(), TFN_WORDS)) {
|
||||
findings.insert(n);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// `NNNN NNNNN N` (and an optional issue number) stands alone; bare digits
|
||||
/// need a word.
|
||||
static MEDICARE: LazyLock<Regex> =
|
||||
LazyLock::new(|| re(r"\b([2-6]\d{3})( ?)(\d{5})( ?)(\d)(?:[ -]?\d)?\b"));
|
||||
|
||||
/// The ninth digit is the weighted sum (1, 3, 7, 9, 1, 3, 7, 9) of the first
|
||||
/// eight, mod 10.
|
||||
pub fn medicare_valid(n: &str) -> bool {
|
||||
let d: Vec<u32> = n.bytes().map(|b| u32::from(b - b'0')).collect();
|
||||
d.len() >= 9
|
||||
&& d[..8]
|
||||
.iter()
|
||||
.zip([1, 3, 7, 9, 1, 3, 7, 9])
|
||||
.map(|(a, w)| a * w)
|
||||
.sum::<u32>()
|
||||
% 10
|
||||
== d[8]
|
||||
}
|
||||
|
||||
const MEDICARE_WORDS: &[&str] = &[
|
||||
"medicare",
|
||||
"medicare card",
|
||||
"medicare no",
|
||||
"medicare number",
|
||||
];
|
||||
|
||||
fn medicare(text: &str, findings: &mut Findings) {
|
||||
for c in MEDICARE.captures_iter(text) {
|
||||
let whole = c.get(0).unwrap();
|
||||
let n = format!("{}{}{}", &c[1], &c[3], &c[5]);
|
||||
let written = c[2] == *" " && c[4] == *" ";
|
||||
if medicare_valid(&n)
|
||||
&& (written || word_near(text, whole.start(), whole.end(), MEDICARE_WORDS))
|
||||
{
|
||||
findings.insert(n);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use crate::mailflow::detectors::by_id;
|
||||
|
||||
fn count(id: &str, text: &str) -> usize {
|
||||
by_id(id).unwrap().count(text)
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn tax_file_numbers() {
|
||||
assert_eq!(count("au-tfn", "TFN 123 456 782"), 1);
|
||||
assert_eq!(count("au-tfn", "123 456 789"), 0);
|
||||
assert_eq!(count("au-tfn", "order 123456782"), 0);
|
||||
assert_eq!(count("au-tfn", "tax file number 123456782"), 1);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn medicare_numbers() {
|
||||
assert_eq!(count("au-medicare", "2123 45670 1"), 1);
|
||||
assert_eq!(count("au-medicare", "2123 45671 1"), 0);
|
||||
assert_eq!(count("au-medicare", "ref 2123456701"), 0);
|
||||
assert_eq!(count("au-medicare", "Medicare 2123456701"), 1);
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,66 @@
|
||||
/*
|
||||
* SPDX-FileCopyrightText: 2026 Coffey Labs LLC
|
||||
*
|
||||
* SPDX-License-Identifier: AGPL-3.0-only
|
||||
*/
|
||||
|
||||
//! Canadian identifiers (§2.3): the Social Insurance Number.
|
||||
|
||||
use super::{Detector, Findings, Region, Strength, checks, word_near};
|
||||
use regex::Regex;
|
||||
use std::sync::LazyLock;
|
||||
|
||||
pub static DETECTORS: &[Detector] = &[Detector::new(
|
||||
"ca-sin",
|
||||
"Canadian Social Insurance Number",
|
||||
Region::Canada,
|
||||
Strength::Checked,
|
||||
sin,
|
||||
)];
|
||||
|
||||
/// `NNN NNN NNN` or `NNN-NNN-NNN` stands alone; nine bare digits need a word.
|
||||
static SIN: LazyLock<Regex> = LazyLock::new(|| {
|
||||
Regex::new(r"\b(\d{3})([ -]?)(\d{3})([ -]?)(\d{3})\b").expect("detector pattern")
|
||||
});
|
||||
|
||||
const SIN_WORDS: &[&str] = &[
|
||||
"sin",
|
||||
"social insurance",
|
||||
"nas",
|
||||
"numéro d'assurance sociale",
|
||||
"assurance sociale",
|
||||
];
|
||||
|
||||
fn sin(text: &str, findings: &mut Findings) {
|
||||
for c in SIN.captures_iter(text) {
|
||||
let whole = c.get(0).unwrap();
|
||||
let n = format!("{}{}{}", &c[1], &c[3], &c[5]);
|
||||
let written = !c[2].is_empty() && c[2] == c[4];
|
||||
// 0 and 8 are never issued as a first digit
|
||||
if !n.starts_with(['0', '8'])
|
||||
&& checks::luhn(&n)
|
||||
&& (written || word_near(text, whole.start(), whole.end(), SIN_WORDS))
|
||||
{
|
||||
findings.insert(n);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use crate::mailflow::detectors::by_id;
|
||||
|
||||
fn count(text: &str) -> usize {
|
||||
by_id("ca-sin").unwrap().count(text)
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn social_insurance_numbers() {
|
||||
assert_eq!(count("130 692 544 and 193-456-787"), 2);
|
||||
assert_eq!(count("130 692 545"), 0);
|
||||
// The government's printed example starts with 0, never issued
|
||||
assert_eq!(count("046 454 286"), 0);
|
||||
assert_eq!(count("order 130692544"), 0);
|
||||
assert_eq!(count("SIN: 130692544"), 1);
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,227 @@
|
||||
/*
|
||||
* SPDX-FileCopyrightText: 2026 Coffey Labs LLC
|
||||
*
|
||||
* SPDX-License-Identifier: AGPL-3.0-only
|
||||
*/
|
||||
|
||||
//! Check-digit algorithms, each from its public definition.
|
||||
|
||||
/// The Luhn check (ISO/IEC 7812-1, Annex B) over a string of ASCII digits.
|
||||
pub fn luhn(digits: &str) -> bool {
|
||||
if digits.len() < 2 || !digits.bytes().all(|b| b.is_ascii_digit()) {
|
||||
return false;
|
||||
}
|
||||
let sum: u32 = digits
|
||||
.bytes()
|
||||
.rev()
|
||||
.enumerate()
|
||||
.map(|(i, b)| {
|
||||
let d = u32::from(b - b'0');
|
||||
if i % 2 == 1 {
|
||||
let d = d * 2;
|
||||
if d > 9 { d - 9 } else { d }
|
||||
} else {
|
||||
d
|
||||
}
|
||||
})
|
||||
.sum();
|
||||
sum.is_multiple_of(10)
|
||||
}
|
||||
|
||||
/// ISO 13616 IBAN lengths, by country, from the IBAN registry.
|
||||
const IBAN_LENGTHS: &[(&str, usize)] = &[
|
||||
("AD", 24),
|
||||
("AE", 23),
|
||||
("AL", 28),
|
||||
("AT", 20),
|
||||
("AZ", 28),
|
||||
("BA", 20),
|
||||
("BE", 16),
|
||||
("BG", 22),
|
||||
("BH", 22),
|
||||
("BI", 27),
|
||||
("BR", 29),
|
||||
("BY", 28),
|
||||
("CH", 21),
|
||||
("CR", 22),
|
||||
("CY", 28),
|
||||
("CZ", 24),
|
||||
("DE", 22),
|
||||
("DJ", 27),
|
||||
("DK", 18),
|
||||
("DO", 28),
|
||||
("EE", 20),
|
||||
("EG", 29),
|
||||
("ES", 24),
|
||||
("FI", 18),
|
||||
("FK", 18),
|
||||
("FO", 18),
|
||||
("FR", 27),
|
||||
("GB", 22),
|
||||
("GE", 22),
|
||||
("GI", 23),
|
||||
("GL", 18),
|
||||
("GR", 27),
|
||||
("GT", 28),
|
||||
("HN", 28),
|
||||
("HR", 21),
|
||||
("HU", 28),
|
||||
("IE", 22),
|
||||
("IL", 23),
|
||||
("IQ", 23),
|
||||
("IS", 26),
|
||||
("IT", 27),
|
||||
("JO", 30),
|
||||
("KW", 30),
|
||||
("KZ", 20),
|
||||
("LB", 28),
|
||||
("LC", 32),
|
||||
("LI", 21),
|
||||
("LT", 20),
|
||||
("LU", 20),
|
||||
("LV", 21),
|
||||
("LY", 25),
|
||||
("MC", 27),
|
||||
("MD", 24),
|
||||
("ME", 22),
|
||||
("MK", 19),
|
||||
("MN", 20),
|
||||
("MR", 27),
|
||||
("MT", 31),
|
||||
("MU", 30),
|
||||
("NI", 28),
|
||||
("NL", 18),
|
||||
("NO", 15),
|
||||
("OM", 23),
|
||||
("PK", 24),
|
||||
("PL", 28),
|
||||
("PS", 29),
|
||||
("PT", 25),
|
||||
("QA", 29),
|
||||
("RO", 24),
|
||||
("RS", 22),
|
||||
("RU", 33),
|
||||
("SA", 24),
|
||||
("SC", 31),
|
||||
("SD", 18),
|
||||
("SE", 24),
|
||||
("SI", 19),
|
||||
("SK", 24),
|
||||
("SM", 27),
|
||||
("SO", 23),
|
||||
("ST", 25),
|
||||
("SV", 28),
|
||||
("TL", 23),
|
||||
("TN", 24),
|
||||
("TR", 26),
|
||||
("UA", 29),
|
||||
("VA", 22),
|
||||
("VG", 24),
|
||||
("XK", 20),
|
||||
("YE", 30),
|
||||
];
|
||||
|
||||
/// The IBAN length for a country code, if the country uses IBANs.
|
||||
pub fn iban_length(country: &str) -> Option<usize> {
|
||||
IBAN_LENGTHS
|
||||
.iter()
|
||||
.find(|(code, _)| *code == country)
|
||||
.map(|(_, len)| *len)
|
||||
}
|
||||
|
||||
/// ISO 13616 / ISO 7064 MOD 97-10 over an IBAN with no spaces, upper case:
|
||||
/// move the first four characters to the end, turn letters into 10–35, and
|
||||
/// the number mod 97 must be 1. Also checks the country's length.
|
||||
pub fn iban(iban: &str) -> bool {
|
||||
if iban.len() < 5
|
||||
|| !iban
|
||||
.bytes()
|
||||
.all(|b| b.is_ascii_uppercase() || b.is_ascii_digit())
|
||||
{
|
||||
return false;
|
||||
}
|
||||
if iban_length(&iban[..2]) != Some(iban.len())
|
||||
|| !iban[2..4].bytes().all(|b| b.is_ascii_digit())
|
||||
{
|
||||
return false;
|
||||
}
|
||||
let mut remainder: u32 = 0;
|
||||
for b in iban[4..].bytes().chain(iban[..4].bytes()) {
|
||||
let value = if b.is_ascii_digit() {
|
||||
u32::from(b - b'0')
|
||||
} else {
|
||||
u32::from(b - b'A') + 10
|
||||
};
|
||||
remainder = if value >= 10 {
|
||||
(remainder * 100 + value) % 97
|
||||
} else {
|
||||
(remainder * 10 + value) % 97
|
||||
};
|
||||
}
|
||||
remainder == 1
|
||||
}
|
||||
|
||||
/// ISO 3166-1 alpha-2 country codes, for SWIFT/BIC positions 5–6.
|
||||
const COUNTRIES: &str = "AD AE AF AG AI AL AM AO AQ AR AS AT AU AW AX AZ BA BB BD BE BF BG BH BI BJ \
|
||||
BL BM BN BO BQ BR BS BT BV BW BY BZ CA CC CD CF CG CH CI CK CL CM CN CO CR CU CV CW CX CY CZ DE DJ \
|
||||
DK DM DO DZ EC EE EG EH ER ES ET FI FJ FK FM FO FR GA GB GD GE GF GG GH GI GL GM GN GP GQ GR GS GT \
|
||||
GU GW GY HK HM HN HR HT HU ID IE IL IM IN IO IQ IR IS IT JE JM JO JP KE KG KH KI KM KN KP KR KW KY \
|
||||
KZ LA LB LC LI LK LR LS LT LU LV LY MA MC MD ME MF MG MH MK ML MM MN MO MP MQ MR MS MT MU MV MW MX \
|
||||
MY MZ NA NC NE NF NG NI NL NO NP NR NU NZ OM PA PE PF PG PH PK PL PM PN PR PS PT PW PY QA RE RO RS \
|
||||
RU RW SA SB SC SD SE SG SH SI SJ SK SL SM SN SO SR SS ST SV SX SY SZ TC TD TF TG TH TJ TK TL TM TN \
|
||||
TO TR TT TV TW TZ UA UG UM US UY UZ VA VC VE VG VI VN VU WF WS XK YE YT ZA ZM ZW";
|
||||
|
||||
pub fn is_country(code: &str) -> bool {
|
||||
code.len() == 2 && COUNTRIES.split(' ').any(|c| c == code)
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::*;
|
||||
|
||||
#[test]
|
||||
fn luhn_known_numbers() {
|
||||
// Published test card numbers
|
||||
for good in [
|
||||
"4242424242424242",
|
||||
"5555555555554444",
|
||||
"378282246310005",
|
||||
"79927398713",
|
||||
] {
|
||||
assert!(luhn(good), "{good}");
|
||||
}
|
||||
for bad in ["4242424242424241", "79927398710", "1", "12a4"] {
|
||||
assert!(!luhn(bad), "{bad}");
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn iban_registry_examples() {
|
||||
// The IBAN registry's own examples
|
||||
for good in [
|
||||
"GB29NWBK60161331926819",
|
||||
"DE89370400440532013000",
|
||||
"FR1420041010050500013M02606",
|
||||
"NL91ABNA0417164300",
|
||||
"BE68539007547034",
|
||||
"NO9386011117947",
|
||||
"CH9300762011623852957",
|
||||
] {
|
||||
assert!(iban(good), "{good}");
|
||||
}
|
||||
for bad in [
|
||||
"GB29NWBK60161331926818", // check fails
|
||||
"GB29NWBK6016133192681", // too short for GB
|
||||
"ZZ29NWBK60161331926819", // no such country
|
||||
"DE8937040044053201300A", // letters where DE has none still fail mod 97
|
||||
] {
|
||||
assert!(!iban(bad), "{bad}");
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn countries() {
|
||||
assert!(is_country("DE") && is_country("US") && is_country("XK"));
|
||||
assert!(!is_country("ZZ") && !is_country("D"));
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,646 @@
|
||||
/*
|
||||
* SPDX-FileCopyrightText: 2026 Coffey Labs LLC
|
||||
*
|
||||
* SPDX-License-Identifier: AGPL-3.0-only
|
||||
*/
|
||||
|
||||
//! European Union national identifiers (§2.3), each from its issuer's
|
||||
//! published rules. An identifier that is only digits and whose check a
|
||||
//! random number passes often (mod 10, mod 11) counts alone only in its
|
||||
//! written form, and as bare digits only beside a word.
|
||||
|
||||
use super::{
|
||||
Detector, Findings, Region, Strength, checks, digit_values, stands_alone, valid_short_date,
|
||||
word_near,
|
||||
};
|
||||
use regex::Regex;
|
||||
use std::sync::LazyLock;
|
||||
|
||||
pub static DETECTORS: &[Detector] = &[
|
||||
Detector::new(
|
||||
"de-tax-id",
|
||||
"Germany: tax ID (Steuer-ID)",
|
||||
Region::Eu,
|
||||
Strength::Checked,
|
||||
de_tax_id,
|
||||
),
|
||||
Detector::new(
|
||||
"de-id-card",
|
||||
"Germany: ID card number",
|
||||
Region::Eu,
|
||||
Strength::Checked,
|
||||
de_id_card,
|
||||
),
|
||||
Detector::new(
|
||||
"fr-nir",
|
||||
"France: social security number (NIR)",
|
||||
Region::Eu,
|
||||
Strength::Checked,
|
||||
fr_nir,
|
||||
),
|
||||
Detector::new(
|
||||
"es-dni-nie",
|
||||
"Spain: DNI and NIE",
|
||||
Region::Eu,
|
||||
Strength::Checked,
|
||||
es_dni_nie,
|
||||
),
|
||||
Detector::new(
|
||||
"it-codice-fiscale",
|
||||
"Italy: codice fiscale",
|
||||
Region::Eu,
|
||||
Strength::Checked,
|
||||
it_codice_fiscale,
|
||||
),
|
||||
Detector::new(
|
||||
"nl-bsn",
|
||||
"Netherlands: BSN",
|
||||
Region::Eu,
|
||||
Strength::Checked,
|
||||
nl_bsn,
|
||||
),
|
||||
Detector::new(
|
||||
"be-national-number",
|
||||
"Belgium: national number",
|
||||
Region::Eu,
|
||||
Strength::Checked,
|
||||
be_national_number,
|
||||
),
|
||||
Detector::new(
|
||||
"pl-pesel",
|
||||
"Poland: PESEL",
|
||||
Region::Eu,
|
||||
Strength::Checked,
|
||||
pl_pesel,
|
||||
),
|
||||
Detector::new(
|
||||
"se-personnummer",
|
||||
"Sweden: personnummer",
|
||||
Region::Eu,
|
||||
Strength::Checked,
|
||||
se_personnummer,
|
||||
),
|
||||
Detector::new(
|
||||
"dk-cpr",
|
||||
"Denmark: CPR number",
|
||||
Region::Eu,
|
||||
Strength::NeedsWord,
|
||||
dk_cpr,
|
||||
),
|
||||
Detector::new(
|
||||
"fi-hetu",
|
||||
"Finland: personal identity code",
|
||||
Region::Eu,
|
||||
Strength::Checked,
|
||||
fi_hetu,
|
||||
),
|
||||
Detector::new(
|
||||
"ie-pps",
|
||||
"Ireland: PPS number",
|
||||
Region::Eu,
|
||||
Strength::Checked,
|
||||
ie_pps,
|
||||
),
|
||||
Detector::new(
|
||||
"pt-nif",
|
||||
"Portugal: NIF",
|
||||
Region::Eu,
|
||||
Strength::Checked,
|
||||
pt_nif,
|
||||
),
|
||||
Detector::new(
|
||||
"at-svnr",
|
||||
"Austria: social insurance number",
|
||||
Region::Eu,
|
||||
Strength::Checked,
|
||||
at_svnr,
|
||||
),
|
||||
];
|
||||
|
||||
fn re(pattern: &str) -> Regex {
|
||||
Regex::new(pattern).expect("detector pattern")
|
||||
}
|
||||
|
||||
fn num(s: &str) -> u32 {
|
||||
s.parse().unwrap_or(0)
|
||||
}
|
||||
|
||||
// --- Germany --------------------------------------------------------------
|
||||
|
||||
/// Eleven digits, written `86 095 742 719` on the BZSt's letters.
|
||||
static DE_TAX: LazyLock<Regex> = LazyLock::new(|| re(r"\b\d{2}( ?)\d{3}( ?)\d{3}( ?)\d{3}\b"));
|
||||
|
||||
/// ISO 7064 MOD 11,10; no leading zero; in the first ten digits one digit
|
||||
/// appears two or three times and every other at most once.
|
||||
pub fn de_tax_id_valid(n: &str) -> bool {
|
||||
let d = digit_values(n);
|
||||
if d.len() != 11 || d[0] == 0 {
|
||||
return false;
|
||||
}
|
||||
let mut counts = [0u8; 10];
|
||||
for &x in &d[..10] {
|
||||
counts[x as usize] += 1;
|
||||
}
|
||||
let repeated = counts.iter().filter(|&&c| c >= 2).count();
|
||||
if repeated != 1 || counts.iter().any(|&c| c > 3) {
|
||||
return false;
|
||||
}
|
||||
let mut product = 10;
|
||||
for &x in &d[..10] {
|
||||
let mut sum = (x + product) % 10;
|
||||
if sum == 0 {
|
||||
sum = 10;
|
||||
}
|
||||
product = (2 * sum) % 11;
|
||||
}
|
||||
let check = match 11 - product {
|
||||
10 => 0,
|
||||
c => c,
|
||||
};
|
||||
check == d[10]
|
||||
}
|
||||
|
||||
const DE_TAX_WORDS: &[&str] = &[
|
||||
"steuer-id",
|
||||
"steueridentifikationsnummer",
|
||||
"steuerliche identifikationsnummer",
|
||||
"idnr",
|
||||
"identifikationsnummer",
|
||||
"tax id",
|
||||
];
|
||||
|
||||
fn de_tax_id(text: &str, findings: &mut Findings) {
|
||||
for c in DE_TAX.captures_iter(text) {
|
||||
let whole = c.get(0).unwrap();
|
||||
let written = [&c[1], &c[2], &c[3]].iter().all(|s| *s == " ");
|
||||
let n: String = whole.as_str().replace(' ', "");
|
||||
if de_tax_id_valid(&n)
|
||||
&& (written || word_near(text, whole.start(), whole.end(), DE_TAX_WORDS))
|
||||
{
|
||||
findings.insert(n);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// The ID card's document number: a letter from the card's alphabet, eight
|
||||
/// more characters from it, then the check digit.
|
||||
static DE_ID: LazyLock<Regex> =
|
||||
LazyLock::new(|| re(r"\b[CFGHJKLMNPRTVWXYZ][CFGHJKLMNPRTVWXYZ0-9]{8}\d\b"));
|
||||
|
||||
/// ICAO 9303 check digit: weights 7, 3, 1; letters A=10 … Z=35.
|
||||
pub fn icao_check(chars: &str, check: u32) -> bool {
|
||||
let value = |c: char| c.to_digit(10).unwrap_or_else(|| c as u32 - 'A' as u32 + 10);
|
||||
let sum: u32 = chars
|
||||
.chars()
|
||||
.zip([7, 3, 1].iter().cycle())
|
||||
.map(|(c, w)| value(c) * w)
|
||||
.sum();
|
||||
sum % 10 == check
|
||||
}
|
||||
|
||||
fn de_id_card(text: &str, findings: &mut Findings) {
|
||||
for m in DE_ID.find_iter(text) {
|
||||
let s = m.as_str();
|
||||
if icao_check(&s[..9], num(&s[9..])) {
|
||||
findings.insert(s);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// --- France ---------------------------------------------------------------
|
||||
|
||||
/// Sex, year, month, department (with Corsica's 2A and 2B), commune, order,
|
||||
/// then the two-digit key, spaces allowed between groups.
|
||||
static FR_NIR: LazyLock<Regex> = LazyLock::new(|| {
|
||||
re(r"\b([1-478]) ?(\d{2}) ?(\d{2}) ?(\d{2}|2[AB]) ?(\d{3}) ?(\d{3}) ?(\d{2})\b")
|
||||
});
|
||||
|
||||
fn fr_nir(text: &str, findings: &mut Findings) {
|
||||
for c in FR_NIR.captures_iter(text) {
|
||||
let month = num(&c[3]);
|
||||
if !(matches!(month, 1..=12 | 20..=42 | 50..=99)) {
|
||||
continue;
|
||||
}
|
||||
let department = match &c[4] {
|
||||
"2A" => "19",
|
||||
"2B" => "18",
|
||||
d => d,
|
||||
};
|
||||
let body = format!(
|
||||
"{}{}{}{}{}{}",
|
||||
&c[1], &c[2], &c[3], department, &c[5], &c[6]
|
||||
);
|
||||
let Ok(value) = body.parse::<u64>() else {
|
||||
continue;
|
||||
};
|
||||
if 97 - value % 97 == u64::from(num(&c[7])) {
|
||||
findings.insert(format!(
|
||||
"{}{}{}{}{}{}{}",
|
||||
&c[1], &c[2], &c[3], &c[4], &c[5], &c[6], &c[7]
|
||||
));
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// --- Spain ----------------------------------------------------------------
|
||||
|
||||
static ES_ID: LazyLock<Regex> = LazyLock::new(|| re(r"(?i)\b([XYZ]?)[ -]?(\d{7,8})[ -]?([A-Z])\b"));
|
||||
|
||||
const DNI_LETTERS: &[u8] = b"TRWAGMYFPDXBNJZSQVHLCKE";
|
||||
|
||||
fn es_dni_nie(text: &str, findings: &mut Findings) {
|
||||
for c in ES_ID.captures_iter(text) {
|
||||
let prefix = c[1].to_ascii_uppercase();
|
||||
let digits = &c[2];
|
||||
// DNI: eight digits; NIE: X, Y or Z and seven digits
|
||||
let number = match (prefix.as_str(), digits.len()) {
|
||||
("", 8) => digits.to_string(),
|
||||
("X", 7) => format!("0{digits}"),
|
||||
("Y", 7) => format!("1{digits}"),
|
||||
("Z", 7) => format!("2{digits}"),
|
||||
_ => continue,
|
||||
};
|
||||
let letter = c[3].to_ascii_uppercase();
|
||||
if DNI_LETTERS[(num(&number) % 23) as usize] == letter.as_bytes()[0] {
|
||||
findings.insert(format!("{prefix}{digits}{letter}"));
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// --- Italy ----------------------------------------------------------------
|
||||
|
||||
/// Surname and name letters, year, month letter, day, place code, check
|
||||
/// letter; digits may be replaced by letters (omocodia).
|
||||
static IT_CF: LazyLock<Regex> = LazyLock::new(|| {
|
||||
let d = "[0-9LMNPQRSTUV]";
|
||||
re(&format!(
|
||||
r"(?i)\b[A-Z]{{6}}{d}{{2}}[ABCDEHLMPRST]{d}{{2}}[A-Z]{d}{{3}}[A-Z]\b"
|
||||
))
|
||||
});
|
||||
|
||||
/// The Ministry's odd-position values for 0–9 and A–Z.
|
||||
const CF_ODD: [u32; 36] = [
|
||||
1, 0, 5, 7, 9, 13, 15, 17, 19, 21, // 0-9
|
||||
1, 0, 5, 7, 9, 13, 15, 17, 19, 21, 2, 4, 18, 20, 11, 3, 6, 8, 12, 14, 16, 10, 22, 25, 24,
|
||||
23, // A-Z
|
||||
];
|
||||
|
||||
pub fn codice_fiscale_valid(cf: &str) -> bool {
|
||||
let index = |c: u8| {
|
||||
if c.is_ascii_digit() {
|
||||
(c - b'0') as usize
|
||||
} else {
|
||||
(c - b'A') as usize + 10
|
||||
}
|
||||
};
|
||||
let even = |c: u8| {
|
||||
if c.is_ascii_digit() {
|
||||
u32::from(c - b'0')
|
||||
} else {
|
||||
u32::from(c - b'A')
|
||||
}
|
||||
};
|
||||
let bytes = cf.as_bytes();
|
||||
let sum: u32 = bytes[..15]
|
||||
.iter()
|
||||
.enumerate()
|
||||
.map(|(i, &c)| {
|
||||
if i % 2 == 0 {
|
||||
CF_ODD[index(c)]
|
||||
} else {
|
||||
even(c)
|
||||
}
|
||||
})
|
||||
.sum();
|
||||
u32::from(bytes[15] - b'A') == sum % 26
|
||||
}
|
||||
|
||||
fn it_codice_fiscale(text: &str, findings: &mut Findings) {
|
||||
for m in IT_CF.find_iter(text) {
|
||||
let cf = m.as_str().to_ascii_uppercase();
|
||||
if codice_fiscale_valid(&cf) {
|
||||
findings.insert(cf);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// --- Netherlands ----------------------------------------------------------
|
||||
|
||||
/// Nine digits, sometimes written `1112.22.333`.
|
||||
static NL_BSN: LazyLock<Regex> = LazyLock::new(|| re(r"\b(\d{4})(\.?)(\d{2})(\.?)(\d{3})\b"));
|
||||
|
||||
/// The eleven test: weights 9 down to 2, and −1 for the last digit.
|
||||
pub fn bsn_valid(n: &str) -> bool {
|
||||
let d = digit_values(n);
|
||||
let sum: i64 = d[..8]
|
||||
.iter()
|
||||
.zip((2..=9).rev())
|
||||
.map(|(a, w)| i64::from(a * w))
|
||||
.sum::<i64>()
|
||||
- i64::from(d[8]);
|
||||
sum != 0 && sum % 11 == 0
|
||||
}
|
||||
|
||||
const BSN_WORDS: &[&str] = &[
|
||||
"bsn",
|
||||
"burgerservicenummer",
|
||||
"sofinummer",
|
||||
"sofi-nummer",
|
||||
"citizen service number",
|
||||
];
|
||||
|
||||
fn nl_bsn(text: &str, findings: &mut Findings) {
|
||||
for c in NL_BSN.captures_iter(text) {
|
||||
let whole = c.get(0).unwrap();
|
||||
let n = format!("{}{}{}", &c[1], &c[3], &c[5]);
|
||||
let written = &c[2] == "." && &c[4] == ".";
|
||||
if bsn_valid(&n) && (written || word_near(text, whole.start(), whole.end(), BSN_WORDS)) {
|
||||
findings.insert(n);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// --- Belgium --------------------------------------------------------------
|
||||
|
||||
/// `YY.MM.DD-XXX.CC` or eleven digits.
|
||||
static BE_NN: LazyLock<Regex> =
|
||||
LazyLock::new(|| re(r"\b(\d{2})\.?(\d{2})\.?(\d{2})-?(\d{3})\.?(\d{2})\b"));
|
||||
|
||||
fn be_national_number(text: &str, findings: &mut Findings) {
|
||||
for c in BE_NN.captures_iter(text) {
|
||||
let (month, day) = (num(&c[2]), num(&c[3]));
|
||||
// Month 0 and day 0 mean unknown; bis numbers add 20 or 40 to the month
|
||||
if !(month <= 12 || (20..=32).contains(&month) || (40..=52).contains(&month)) || day > 31 {
|
||||
continue;
|
||||
}
|
||||
let body = format!("{}{}{}{}", &c[1], &c[2], &c[3], &c[4]);
|
||||
let check = u64::from(num(&c[5]));
|
||||
let before_2000 = 97 - body.parse::<u64>().unwrap_or(0) % 97;
|
||||
let since_2000 = 97 - format!("2{body}").parse::<u64>().unwrap_or(0) % 97;
|
||||
if check == before_2000 || check == since_2000 {
|
||||
findings.insert(format!("{body}{}", &c[5]));
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// --- Poland ---------------------------------------------------------------
|
||||
|
||||
static ELEVEN: LazyLock<Regex> = LazyLock::new(|| re(r"\b\d{11}\b"));
|
||||
|
||||
/// Weights 1, 3, 7, 9 repeating; the birth date encodes the century in the
|
||||
/// month (+80 for the 1800s, +20 for the 2000s, and so on).
|
||||
pub fn pesel_valid(n: &str) -> bool {
|
||||
let d = digit_values(n);
|
||||
let sum: u32 = d[..10]
|
||||
.iter()
|
||||
.zip([1, 3, 7, 9].iter().cycle())
|
||||
.map(|(a, w)| a * w)
|
||||
.sum();
|
||||
let month = d[2] * 10 + d[3];
|
||||
let (century, month) = match month {
|
||||
81..=92 => (1800, month - 80),
|
||||
1..=12 => (1900, month),
|
||||
21..=32 => (2000, month - 20),
|
||||
41..=52 => (2100, month - 40),
|
||||
_ => return false,
|
||||
};
|
||||
let year = century + d[0] * 10 + d[1];
|
||||
(10 - sum % 10) % 10 == d[10] && (1..=super::days_in(year, month)).contains(&(d[4] * 10 + d[5]))
|
||||
}
|
||||
|
||||
const PESEL_WORDS: &[&str] = &["pesel", "numer pesel", "nr pesel"];
|
||||
|
||||
fn pl_pesel(text: &str, findings: &mut Findings) {
|
||||
for m in ELEVEN.find_iter(text) {
|
||||
if pesel_valid(m.as_str()) && word_near(text, m.start(), m.end(), PESEL_WORDS) {
|
||||
findings.insert(m.as_str());
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// --- Sweden ---------------------------------------------------------------
|
||||
|
||||
/// `YYMMDD-NNNN`, `YYYYMMDD-NNNN` (`+` after 100), or the bare digits.
|
||||
static SE_PNR: LazyLock<Regex> =
|
||||
LazyLock::new(|| re(r"\b(?:\d{2})?(\d{2})(\d{2})(\d{2})([-+]?)(\d{4})\b"));
|
||||
|
||||
const SE_WORDS: &[&str] = &[
|
||||
"personnummer",
|
||||
"personnr",
|
||||
"person nr",
|
||||
"samordningsnummer",
|
||||
"pnr",
|
||||
];
|
||||
|
||||
fn se_personnummer(text: &str, findings: &mut Findings) {
|
||||
for c in SE_PNR.captures_iter(text) {
|
||||
let whole = c.get(0).unwrap();
|
||||
let (yy, month, day) = (num(&c[1]), num(&c[2]), num(&c[3]));
|
||||
// Coordination numbers add 60 to the day
|
||||
let day = if day > 60 { day - 60 } else { day };
|
||||
let ten = format!("{}{}{}{}", &c[1], &c[2], &c[3], &c[5]);
|
||||
let written = !c[4].is_empty();
|
||||
if valid_short_date(yy, month, day)
|
||||
&& checks::luhn(&ten)
|
||||
&& (written || word_near(text, whole.start(), whole.end(), SE_WORDS))
|
||||
{
|
||||
findings.insert(ten);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// --- Denmark --------------------------------------------------------------
|
||||
|
||||
static DK_CPR: LazyLock<Regex> = LazyLock::new(|| re(r"\b(\d{2})(\d{2})(\d{2})-?(\d{4})\b"));
|
||||
|
||||
const CPR_WORDS: &[&str] = &["cpr", "cpr-nr", "cpr nr", "cpr-nummer", "personnummer"];
|
||||
|
||||
fn dk_cpr(text: &str, findings: &mut Findings) {
|
||||
for c in DK_CPR.captures_iter(text) {
|
||||
let whole = c.get(0).unwrap();
|
||||
if valid_short_date(num(&c[3]), num(&c[2]), num(&c[1]))
|
||||
&& word_near(text, whole.start(), whole.end(), CPR_WORDS)
|
||||
{
|
||||
findings.insert(whole.as_str().replace('-', ""));
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// --- Finland --------------------------------------------------------------
|
||||
|
||||
static FI_HETU: LazyLock<Regex> =
|
||||
LazyLock::new(|| re(r"(?i)\b(\d{2})(\d{2})(\d{2})[-+ABCDEFYXWVU](\d{3})([0-9A-Y])\b"));
|
||||
|
||||
const HETU_CHECK: &[u8] = b"0123456789ABCDEFHJKLMNPRSTUVWXY";
|
||||
|
||||
fn fi_hetu(text: &str, findings: &mut Findings) {
|
||||
for c in FI_HETU.captures_iter(text) {
|
||||
let (day, month, yy) = (num(&c[1]), num(&c[2]), num(&c[3]));
|
||||
let n: u64 = format!("{}{}{}{}", &c[1], &c[2], &c[3], &c[4])
|
||||
.parse()
|
||||
.unwrap_or(0);
|
||||
let check = c[5].to_ascii_uppercase().as_bytes()[0];
|
||||
if valid_short_date(yy, month, day) && HETU_CHECK[(n % 31) as usize] == check {
|
||||
findings.insert(c[0].to_ascii_uppercase());
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// --- Ireland --------------------------------------------------------------
|
||||
|
||||
static IE_PPS: LazyLock<Regex> = LazyLock::new(|| re(r"(?i)\b(\d{7})([A-W])([ABHW]?)\b"));
|
||||
|
||||
const PPS_CHECK: &[u8] = b"WABCDEFGHIJKLMNOPQRSTUV";
|
||||
|
||||
fn ie_pps(text: &str, findings: &mut Findings) {
|
||||
for c in IE_PPS.captures_iter(text) {
|
||||
let mut sum: u32 = digit_values(&c[1])
|
||||
.iter()
|
||||
.zip((2..=8).rev())
|
||||
.map(|(a, w)| a * w)
|
||||
.sum();
|
||||
// The second letter counts, times 9; W (the old form) counts as 0
|
||||
let second = c[3].to_ascii_uppercase();
|
||||
if let Some(&letter) = second.as_bytes().first()
|
||||
&& letter != b'W'
|
||||
{
|
||||
sum += u32::from(letter - b'A' + 1) * 9;
|
||||
}
|
||||
let check = c[2].to_ascii_uppercase().as_bytes()[0];
|
||||
if PPS_CHECK[(sum % 23) as usize] == check {
|
||||
findings.insert(c[0].to_ascii_uppercase());
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// --- Portugal -------------------------------------------------------------
|
||||
|
||||
static NINE: LazyLock<Regex> = LazyLock::new(|| re(r"\b\d{9}\b"));
|
||||
|
||||
/// Mod 11 over weights 9 down to 2; a check of 10 or 11 becomes 0.
|
||||
pub fn nif_valid(n: &str) -> bool {
|
||||
let d = digit_values(n);
|
||||
let sum: u32 = d[..8].iter().zip((2..=9).rev()).map(|(a, w)| a * w).sum();
|
||||
let check = match 11 - sum % 11 {
|
||||
10 | 11 => 0,
|
||||
c => c,
|
||||
};
|
||||
matches!(d[0], 1 | 2 | 3 | 5 | 6 | 8 | 9) && check == d[8]
|
||||
}
|
||||
|
||||
const NIF_WORDS: &[&str] = &[
|
||||
"nif",
|
||||
"contribuinte",
|
||||
"número de identificação fiscal",
|
||||
"numero de contribuinte",
|
||||
];
|
||||
|
||||
fn pt_nif(text: &str, findings: &mut Findings) {
|
||||
for m in NINE.find_iter(text) {
|
||||
if nif_valid(m.as_str()) && word_near(text, m.start(), m.end(), NIF_WORDS) {
|
||||
findings.insert(m.as_str());
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// --- Austria --------------------------------------------------------------
|
||||
|
||||
/// A serial and check digit, then the birth date: `1237 010180`.
|
||||
static AT_SVNR: LazyLock<Regex> = LazyLock::new(|| re(r"\b(\d{3})(\d)( ?)(\d{2})(\d{2})(\d{2})\b"));
|
||||
|
||||
const SVNR_WORDS: &[&str] = &[
|
||||
"sozialversicherungsnummer",
|
||||
"svnr",
|
||||
"sv-nr",
|
||||
"sv-nummer",
|
||||
"versicherungsnummer",
|
||||
];
|
||||
|
||||
fn at_svnr(text: &str, findings: &mut Findings) {
|
||||
for c in AT_SVNR.captures_iter(text) {
|
||||
let whole = c.get(0).unwrap();
|
||||
let n = format!("{}{}{}{}{}", &c[1], &c[2], &c[4], &c[5], &c[6]);
|
||||
let d = digit_values(&n);
|
||||
let sum: u32 = d
|
||||
.iter()
|
||||
.zip([3, 7, 9, 0, 5, 8, 4, 2, 1, 6])
|
||||
.map(|(a, w)| a * w)
|
||||
.sum();
|
||||
let written = &c[3] == " ";
|
||||
if d[0] != 0
|
||||
&& sum % 11 == d[3]
|
||||
&& valid_short_date(num(&c[6]), num(&c[5]), num(&c[4]))
|
||||
&& (written || word_near(text, whole.start(), whole.end(), SVNR_WORDS))
|
||||
&& stands_alone(text, whole.start(), whole.end())
|
||||
{
|
||||
findings.insert(n);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use crate::mailflow::detectors::by_id;
|
||||
|
||||
fn count(id: &str, text: &str) -> usize {
|
||||
by_id(id).unwrap().count(text)
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn germany() {
|
||||
assert_eq!(count("de-tax-id", "86 095 742 719"), 1);
|
||||
assert_eq!(count("de-tax-id", "Steuer-ID: 86095742719"), 1);
|
||||
assert_eq!(count("de-tax-id", "Rechnung 86095742719"), 0);
|
||||
assert_eq!(count("de-tax-id", "86 095 742 718"), 0);
|
||||
// ICAO 9303's German specimen card
|
||||
assert_eq!(count("de-id-card", "Ausweis T220001293"), 1);
|
||||
assert_eq!(count("de-id-card", "T220001294"), 0);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn france_spain_italy() {
|
||||
assert_eq!(count("fr-nir", "2 55 08 14 168 025 38"), 1);
|
||||
assert_eq!(count("fr-nir", "255081416802539"), 0);
|
||||
assert_eq!(count("es-dni-nie", "DNI 12345678Z, NIE X-1234567-L"), 2);
|
||||
assert_eq!(count("es-dni-nie", "12345678A"), 0);
|
||||
assert_eq!(count("it-codice-fiscale", "CF: RSSMRA85T10A562S"), 1);
|
||||
assert_eq!(count("it-codice-fiscale", "RSSMRA85T10A562T"), 0);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn benelux() {
|
||||
assert_eq!(count("nl-bsn", "1112.22.333"), 1);
|
||||
assert_eq!(count("nl-bsn", "BSN 111222333"), 1);
|
||||
assert_eq!(count("nl-bsn", "order 111222333"), 0);
|
||||
assert_eq!(count("nl-bsn", "BSN 111222334"), 0);
|
||||
assert_eq!(count("be-national-number", "85.07.30-033.28"), 1);
|
||||
assert_eq!(count("be-national-number", "85073003329"), 0);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn nordics() {
|
||||
assert_eq!(count("se-personnummer", "811218-9876"), 1);
|
||||
assert_eq!(count("se-personnummer", "811218-9875"), 0);
|
||||
assert_eq!(count("se-personnummer", "order 8112189876"), 0);
|
||||
assert_eq!(count("se-personnummer", "personnummer 198112189876"), 1);
|
||||
assert_eq!(count("dk-cpr", "CPR-nr: 010170-1234"), 1);
|
||||
assert_eq!(count("dk-cpr", "010170-1234"), 0);
|
||||
assert_eq!(count("dk-cpr", "CPR 320170-1234"), 0);
|
||||
assert_eq!(count("fi-hetu", "131052-308T"), 1);
|
||||
assert_eq!(count("fi-hetu", "131052-308U"), 0);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn poland_ireland_portugal_austria() {
|
||||
assert_eq!(count("pl-pesel", "PESEL 44051401359, pesel 02070803628"), 2);
|
||||
assert_eq!(count("pl-pesel", "PESEL 44051401358"), 0);
|
||||
assert_eq!(count("pl-pesel", "44051401359"), 0);
|
||||
assert_eq!(count("ie-pps", "PPS 1234567T and 1234567FA"), 2);
|
||||
assert_eq!(count("ie-pps", "1234567U"), 0);
|
||||
assert_eq!(count("pt-nif", "NIF 123456789"), 1);
|
||||
assert_eq!(count("pt-nif", "NIF 123456788"), 0);
|
||||
assert_eq!(count("at-svnr", "1237 010180"), 1);
|
||||
assert_eq!(count("at-svnr", "SVNR 1237010180"), 1);
|
||||
assert_eq!(count("at-svnr", "1238 010180"), 0);
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,116 @@
|
||||
/*
|
||||
* SPDX-FileCopyrightText: 2026 Coffey Labs LLC
|
||||
*
|
||||
* SPDX-License-Identifier: AGPL-3.0-only
|
||||
*/
|
||||
|
||||
//! European identifiers outside the EU (§2.3): Norway's national identity
|
||||
//! number and Switzerland's AHV number.
|
||||
|
||||
use super::{Detector, Findings, Region, Strength, digit_values, valid_short_date};
|
||||
use regex::Regex;
|
||||
use std::sync::LazyLock;
|
||||
|
||||
pub static DETECTORS: &[Detector] = &[
|
||||
Detector::new(
|
||||
"no-fnr",
|
||||
"Norway: national identity number",
|
||||
Region::Europe,
|
||||
Strength::Checked,
|
||||
no_fnr,
|
||||
),
|
||||
Detector::new(
|
||||
"ch-ahv",
|
||||
"Switzerland: AHV number",
|
||||
Region::Europe,
|
||||
Strength::Checked,
|
||||
ch_ahv,
|
||||
),
|
||||
];
|
||||
|
||||
static ELEVEN: LazyLock<Regex> =
|
||||
LazyLock::new(|| Regex::new(r"\b\d{6} ?\d{5}\b").expect("detector pattern"));
|
||||
|
||||
/// Two mod 11 check digits over a birth date (D-numbers add 40 to the day,
|
||||
/// H-numbers 40 to the month): strong enough to count alone.
|
||||
pub fn fnr_valid(n: &str) -> bool {
|
||||
let d = digit_values(n);
|
||||
if d.len() != 11 {
|
||||
return false;
|
||||
}
|
||||
let check =
|
||||
|weights: &[u32]| match 11 - d.iter().zip(weights).map(|(a, w)| a * w).sum::<u32>() % 11 {
|
||||
11 => Some(0),
|
||||
10 => None,
|
||||
c => Some(c),
|
||||
};
|
||||
let day = d[0] * 10 + d[1];
|
||||
let month = d[2] * 10 + d[3];
|
||||
let day = if day > 40 { day - 40 } else { day };
|
||||
let month = if month > 40 { month - 40 } else { month };
|
||||
valid_short_date(d[4] * 10 + d[5], month, day)
|
||||
&& check(&[3, 7, 6, 1, 8, 9, 4, 5, 2]) == Some(d[9])
|
||||
&& check(&[5, 4, 3, 2, 7, 6, 5, 4, 3, 2]) == Some(d[10])
|
||||
}
|
||||
|
||||
fn no_fnr(text: &str, findings: &mut Findings) {
|
||||
for m in ELEVEN.find_iter(text) {
|
||||
let n = m.as_str().replace(' ', "");
|
||||
if fnr_valid(&n) {
|
||||
findings.insert(n);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// `756.1234.5678.97`: the country prefix, then an EAN-13 check digit.
|
||||
static AHV: LazyLock<Regex> = LazyLock::new(|| {
|
||||
Regex::new(r"\b756[. ]?\d{4}[. ]?\d{4}[. ]?\d{2}\b").expect("detector pattern")
|
||||
});
|
||||
|
||||
pub fn ean13_valid(n: &str) -> bool {
|
||||
let d = digit_values(n);
|
||||
if d.len() != 13 {
|
||||
return false;
|
||||
}
|
||||
let sum: u32 = d[..12]
|
||||
.iter()
|
||||
.enumerate()
|
||||
.map(|(i, x)| if i % 2 == 0 { *x } else { x * 3 })
|
||||
.sum();
|
||||
(10 - sum % 10) % 10 == d[12]
|
||||
}
|
||||
|
||||
fn ch_ahv(text: &str, findings: &mut Findings) {
|
||||
for m in AHV.find_iter(text) {
|
||||
let n: String = m.as_str().chars().filter(char::is_ascii_digit).collect();
|
||||
if ean13_valid(&n) {
|
||||
findings.insert(n);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use crate::mailflow::detectors::by_id;
|
||||
|
||||
fn count(id: &str, text: &str) -> usize {
|
||||
by_id(id).unwrap().count(text)
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn norway() {
|
||||
assert_eq!(count("no-fnr", "01019000083"), 1);
|
||||
assert_eq!(count("no-fnr", "010190 00083"), 1);
|
||||
assert_eq!(count("no-fnr", "01019000084"), 0);
|
||||
// Not a date
|
||||
assert_eq!(count("no-fnr", "32019000083"), 0);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn switzerland() {
|
||||
// The federal example
|
||||
assert_eq!(count("ch-ahv", "AHV 756.9217.0769.85"), 1);
|
||||
assert_eq!(count("ch-ahv", "7569217076985"), 1);
|
||||
assert_eq!(count("ch-ahv", "756.9217.0769.86"), 0);
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,278 @@
|
||||
/*
|
||||
* SPDX-FileCopyrightText: 2026 Coffey Labs LLC
|
||||
*
|
||||
* SPDX-License-Identifier: AGPL-3.0-only
|
||||
*/
|
||||
|
||||
//! Detectors (dlp-and-mail-flow-rules spec, §2.3): each finds one kind of
|
||||
//! identifier in text and reports the distinct ones it found.
|
||||
//!
|
||||
//! A detector is one of two strengths:
|
||||
//!
|
||||
//! - **Checked**: the identifier carries a published check digit or
|
||||
//! checksum, so a random number rarely passes; found on its own.
|
||||
//! - **Needs a word**: the format alone is too common, so a candidate counts
|
||||
//! only with a corroborating word within [`WINDOW`] characters either
|
||||
//! side.
|
||||
//!
|
||||
//! Findings are distinct normalized values (digits only, upper case), so the
|
||||
//! same card number pasted twice counts once. They stay in memory: callers
|
||||
//! read only [`Findings::len`].
|
||||
|
||||
pub mod africa;
|
||||
pub mod americas;
|
||||
pub mod any;
|
||||
pub mod asia;
|
||||
pub mod australia;
|
||||
pub mod canada;
|
||||
pub mod checks;
|
||||
pub mod eu;
|
||||
pub mod europe;
|
||||
pub mod templates;
|
||||
pub mod uk;
|
||||
pub mod us;
|
||||
|
||||
use ahash::AHashSet;
|
||||
|
||||
/// How far, in characters, a corroborating word may be from a candidate.
|
||||
pub const WINDOW: usize = 50;
|
||||
|
||||
#[derive(Debug, Clone, Copy, PartialEq, Eq)]
|
||||
pub enum Strength {
|
||||
Checked,
|
||||
NeedsWord,
|
||||
}
|
||||
|
||||
#[derive(Debug, Clone, Copy, PartialEq, Eq)]
|
||||
pub enum Region {
|
||||
Any,
|
||||
Us,
|
||||
Uk,
|
||||
Canada,
|
||||
Australia,
|
||||
Eu,
|
||||
Europe,
|
||||
Asia,
|
||||
Americas,
|
||||
Africa,
|
||||
}
|
||||
|
||||
/// The distinct values one detector found.
|
||||
#[derive(Debug, Default)]
|
||||
pub struct Findings(AHashSet<String>);
|
||||
|
||||
impl Findings {
|
||||
pub fn insert(&mut self, value: impl Into<String>) {
|
||||
self.0.insert(value.into());
|
||||
}
|
||||
|
||||
pub fn len(&self) -> usize {
|
||||
self.0.len()
|
||||
}
|
||||
|
||||
pub fn is_empty(&self) -> bool {
|
||||
self.0.is_empty()
|
||||
}
|
||||
}
|
||||
|
||||
pub struct Detector {
|
||||
/// Stable id, stored in rules: `payment-card`, `iban`, `us-ssn`.
|
||||
pub id: &'static str,
|
||||
pub name: &'static str,
|
||||
pub region: Region,
|
||||
pub strength: Strength,
|
||||
find: fn(&str, &mut Findings),
|
||||
}
|
||||
|
||||
impl Detector {
|
||||
pub const fn new(
|
||||
id: &'static str,
|
||||
name: &'static str,
|
||||
region: Region,
|
||||
strength: Strength,
|
||||
find: fn(&str, &mut Findings),
|
||||
) -> Self {
|
||||
Self {
|
||||
id,
|
||||
name,
|
||||
region,
|
||||
strength,
|
||||
find,
|
||||
}
|
||||
}
|
||||
|
||||
/// Adds what this detector finds in `text` to `findings`. Call once per
|
||||
/// piece of text (subject, each part, each attachment) with the same
|
||||
/// `findings`, then read its length.
|
||||
pub fn find(&self, text: &str, findings: &mut Findings) {
|
||||
(self.find)(text, findings)
|
||||
}
|
||||
|
||||
/// The distinct values found in one text.
|
||||
pub fn count(&self, text: &str) -> usize {
|
||||
let mut findings = Findings::default();
|
||||
self.find(text, &mut findings);
|
||||
findings.len()
|
||||
}
|
||||
}
|
||||
|
||||
/// Every detector, in the order the console lists them.
|
||||
pub fn all() -> impl Iterator<Item = &'static Detector> {
|
||||
[
|
||||
any::DETECTORS,
|
||||
us::DETECTORS,
|
||||
uk::DETECTORS,
|
||||
canada::DETECTORS,
|
||||
australia::DETECTORS,
|
||||
eu::DETECTORS,
|
||||
europe::DETECTORS,
|
||||
asia::DETECTORS,
|
||||
americas::DETECTORS,
|
||||
africa::DETECTORS,
|
||||
]
|
||||
.into_iter()
|
||||
.flatten()
|
||||
}
|
||||
|
||||
pub fn by_id(id: &str) -> Option<&'static Detector> {
|
||||
all().find(|detector| detector.id == id)
|
||||
}
|
||||
|
||||
/// Whether one of `words` appears, as a whole word and ignoring case, within
|
||||
/// [`WINDOW`] characters before `start` or after `end` (byte offsets of the
|
||||
/// candidate in `text`). The window is widened by the longest word, so a
|
||||
/// word that reaches into it still counts whole.
|
||||
pub fn word_near(text: &str, start: usize, end: usize, words: &[&str]) -> bool {
|
||||
let reach = WINDOW + words.iter().map(|w| w.chars().count()).max().unwrap_or(0);
|
||||
let before = text[..start]
|
||||
.char_indices()
|
||||
.rev()
|
||||
.nth(reach - 1)
|
||||
.map_or(0, |(i, _)| i);
|
||||
let after = text[end..]
|
||||
.char_indices()
|
||||
.nth(reach)
|
||||
.map_or(text.len(), |(i, _)| end + i);
|
||||
let window = text[before..after].to_lowercase();
|
||||
words.iter().any(|word| contains_word(&window, word))
|
||||
}
|
||||
|
||||
/// Whether `word` (lower case) appears in `haystack` (lower case) with no
|
||||
/// letter or digit on either side.
|
||||
pub fn contains_word(haystack: &str, word: &str) -> bool {
|
||||
haystack.match_indices(word).any(|(i, _)| {
|
||||
let before_ok = haystack[..i]
|
||||
.chars()
|
||||
.next_back()
|
||||
.is_none_or(|c| !c.is_alphanumeric());
|
||||
let after_ok = haystack[i + word.len()..]
|
||||
.chars()
|
||||
.next()
|
||||
.is_none_or(|c| !c.is_alphanumeric());
|
||||
before_ok && after_ok
|
||||
})
|
||||
}
|
||||
|
||||
/// Whether the match at `start..end` stands alone: no digit or letter
|
||||
/// directly before or after it, so `123-45-6789` isn't found inside a
|
||||
/// longer run of digits.
|
||||
pub fn stands_alone(text: &str, start: usize, end: usize) -> bool {
|
||||
let before = text[..start].chars().next_back();
|
||||
let after = text[end..].chars().next();
|
||||
before.is_none_or(|c| !c.is_alphanumeric()) && after.is_none_or(|c| !c.is_alphanumeric())
|
||||
}
|
||||
|
||||
/// Days in `month` of `year` (0 for a month that doesn't exist).
|
||||
pub fn days_in(year: u32, month: u32) -> u32 {
|
||||
match month {
|
||||
1 | 3 | 5 | 7 | 8 | 10 | 12 => 31,
|
||||
4 | 6 | 9 | 11 => 30,
|
||||
2 if year.is_multiple_of(4) && (!year.is_multiple_of(100) || year.is_multiple_of(400)) => {
|
||||
29
|
||||
}
|
||||
2 => 28,
|
||||
_ => 0,
|
||||
}
|
||||
}
|
||||
|
||||
/// Whether `year`-`month`-`day` is a real date between 1900 and 2100.
|
||||
pub fn valid_date(year: u32, month: u32, day: u32) -> bool {
|
||||
(1900..=2100).contains(&year) && (1..=days_in(year, month)).contains(&day)
|
||||
}
|
||||
|
||||
/// Whether a two-digit year, month and day make a real date in either the
|
||||
/// 1900s or the 2000s.
|
||||
pub fn valid_short_date(yy: u32, month: u32, day: u32) -> bool {
|
||||
valid_date(1900 + yy, month, day) || valid_date(2000 + yy, month, day)
|
||||
}
|
||||
|
||||
/// The value of each digit in `s`.
|
||||
pub fn digit_values(s: &str) -> Vec<u32> {
|
||||
s.bytes()
|
||||
.filter(u8::is_ascii_digit)
|
||||
.map(|b| u32::from(b - b'0'))
|
||||
.collect()
|
||||
}
|
||||
|
||||
/// The ASCII digits of `s`.
|
||||
pub fn digits(s: &str) -> String {
|
||||
s.chars().filter(char::is_ascii_digit).collect()
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::*;
|
||||
|
||||
#[test]
|
||||
fn words_are_whole_and_near() {
|
||||
let text = "Your passport number is X1234567, thanks";
|
||||
let start = text.find("X123").unwrap();
|
||||
assert!(word_near(text, start, start + 8, &["passport"]));
|
||||
assert!(!word_near(text, start, start + 8, &["pass"]));
|
||||
let far = format!("passport{}X1234567", " ".repeat(60));
|
||||
let start = far.find("X123").unwrap();
|
||||
assert!(!word_near(&far, start, start + 8, &["passport"]));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn near_counts_characters_not_bytes() {
|
||||
// 45 two-byte characters between the word and the candidate: within
|
||||
// 50 characters, though over 50 bytes
|
||||
let text = format!("passport {} X1234567", "é".repeat(45));
|
||||
let start = text.find("X123").unwrap();
|
||||
assert!(word_near(&text, start, start + 8, &["passport"]));
|
||||
}
|
||||
|
||||
/// An ordinary business email: order, invoice and tracking numbers,
|
||||
/// dates, amounts, a street address. Nothing here is an identifier, so
|
||||
/// no detector may fire, except the contact ones on the signature.
|
||||
#[test]
|
||||
fn ordinary_mail_finds_nothing() {
|
||||
let text = "Hi Dana,\n\nThanks for order 4471-2290 placed 2026-09-14. Invoice INV-2026-00917 \
|
||||
for $12,480.00 is due 10/31/2026; PO 7731902 covers lines 1-14. Tracking \
|
||||
1Z999AA10123456784, parcel 3 of 5, 12.5 kg, box 40x30x20 cm. Meeting moved to \
|
||||
Tuesday 9:30-10:15 in room 2B, building 1177. Ticket #5520318, case 20260914-0042. \
|
||||
Version 2026.9.28.4, build 118822, commit 5a73a118. Serial SN-88213-X. \
|
||||
Ship to 1600 Amphitheatre Pkwy, Mountain View, CA 94043. Revenue grew 18% to \
|
||||
1,204,332 units; see figures 3.1-3.4 and table 12.\n\nBest,\nSam\n\
|
||||
Sam Rivera | +1 (415) 555-2671 | sam@example.com";
|
||||
let quiet = ["email-addresses", "phone-numbers"];
|
||||
for detector in all().filter(|d| !quiet.contains(&d.id)) {
|
||||
assert_eq!(
|
||||
detector.count(text),
|
||||
0,
|
||||
"{} fired on ordinary mail",
|
||||
detector.id
|
||||
);
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn ids_are_unique() {
|
||||
let mut seen = AHashSet::new();
|
||||
for detector in all() {
|
||||
assert!(seen.insert(detector.id), "duplicate id {}", detector.id);
|
||||
assert!(by_id(detector.id).is_some());
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,97 @@
|
||||
/*
|
||||
* SPDX-FileCopyrightText: 2026 Coffey Labs LLC
|
||||
*
|
||||
* SPDX-License-Identifier: AGPL-3.0-only
|
||||
*/
|
||||
|
||||
//! Templates (§2.3): named sets of detectors, so a policy doesn't pick forty
|
||||
//! one at a time. Each is named for what it finds, never for a law, and is a
|
||||
//! starting point: once added to a rule, its detectors can be changed.
|
||||
|
||||
pub struct Template {
|
||||
pub id: &'static str,
|
||||
pub name: &'static str,
|
||||
pub detectors: &'static [&'static str],
|
||||
}
|
||||
|
||||
pub static TEMPLATES: &[Template] = &[
|
||||
Template {
|
||||
id: "payment-and-bank",
|
||||
name: "Payment cards and bank accounts",
|
||||
detectors: &["payment-card", "iban", "swift-bic", "us-aba-routing"],
|
||||
},
|
||||
Template {
|
||||
id: "us-personal",
|
||||
name: "US personal identifiers",
|
||||
detectors: &[
|
||||
"us-ssn",
|
||||
"us-itin",
|
||||
"us-ein",
|
||||
"us-drivers-license",
|
||||
"passport",
|
||||
"date-of-birth",
|
||||
],
|
||||
},
|
||||
Template {
|
||||
id: "uk-personal",
|
||||
name: "UK personal identifiers",
|
||||
detectors: &["uk-nino", "uk-utr", "uk-nhs", "passport", "date-of-birth"],
|
||||
},
|
||||
Template {
|
||||
id: "eu-national",
|
||||
name: "EU national identifiers",
|
||||
detectors: &[
|
||||
"de-tax-id",
|
||||
"de-id-card",
|
||||
"fr-nir",
|
||||
"es-dni-nie",
|
||||
"it-codice-fiscale",
|
||||
"nl-bsn",
|
||||
"be-national-number",
|
||||
"pl-pesel",
|
||||
"se-personnummer",
|
||||
"dk-cpr",
|
||||
"fi-hetu",
|
||||
"ie-pps",
|
||||
"pt-nif",
|
||||
"at-svnr",
|
||||
],
|
||||
},
|
||||
Template {
|
||||
id: "health",
|
||||
name: "Health identifiers",
|
||||
detectors: &["uk-nhs", "us-mbi", "us-npi", "us-dea", "au-medicare"],
|
||||
},
|
||||
Template {
|
||||
id: "credentials",
|
||||
name: "Credentials and keys",
|
||||
detectors: &["private-key", "credentials"],
|
||||
},
|
||||
Template {
|
||||
id: "contact-lists",
|
||||
name: "Contact lists",
|
||||
detectors: &["email-addresses", "phone-numbers"],
|
||||
},
|
||||
];
|
||||
|
||||
pub fn by_id(id: &str) -> Option<&'static Template> {
|
||||
TEMPLATES.iter().find(|template| template.id == id)
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::*;
|
||||
|
||||
#[test]
|
||||
fn every_template_names_real_detectors() {
|
||||
for template in TEMPLATES {
|
||||
for id in template.detectors {
|
||||
assert!(
|
||||
super::super::by_id(id).is_some(),
|
||||
"{}: no detector {id}",
|
||||
template.id
|
||||
);
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
Loaded 100 of 407 files, more files were not shown because too many files have changed in this diff.
Show more
Reference in new issue
Block a user