Compare commits

..
Author SHA1 Message Date
jcoffey-dev 77404353d5 Merge pull request 'Release 2026.9.28.6' (#107) from hotfix/2026.9.28.6 into release/2026.9.28.6
publish / version (push) Successful in 11s
publish / publish-amd64 (push) Successful in 31m53s
publish / release (push) Successful in 1s
publish / publish-arm64 (push) Successful in 50m13s
publish / binaries (push) Successful in 42s
publish / announce (push) Successful in 22s
2026-09-29 01:29:48 +00:00
jcoffey-dev d2b0750e1d Release 2026.9.28.6
ci / fork-checks (pull_request) Successful in 45s
ci / build (pull_request) Successful in 6m10s
2026-09-28 18:23:26 -07:00
jcoffey-dev 0232e59283 Ports: each node checks the others' ports from outside 2026-09-28 18:16:33 -07:00
jcoffey-dev 9fd1d32799 Explain: don't prepare answers for date fields 2026-09-28 18:16:33 -07:00
jcoffey-dev eafe8bfbfd Webhooks: send one sample event to a saved webhook 2026-09-28 18:16:33 -07:00
404 changed files with 1313 additions and 24594 deletions

No files matched your search

+1 -44
View File
@@ -7,12 +7,7 @@
# instance resolves short `uses:` against itself, never GitHub, so nothing
# unreviewed can be pulled in.
#
# BUILD_ON: when the Actions variable BUILD_ON is 'github' (org or repo),
# fork-checks and build skip here and the `github` job below waits for the
# same work done by .github/workflows/ci.yml on the GitHub mirror, passing or
# failing with it -- so this run still carries the answer pull requests and
# merges look at. Unset, everything builds here as before. If GitHub is
# unavailable, unset BUILD_ON and nothing else has to change.
# Not ported, as on GitLab: publish.yml and release.yml still need doing.
name: ci
on:
@@ -30,7 +25,6 @@ jobs:
# without the AGPL 5(a) notice. Seconds, and needs no toolchain. The notice
# check diffs against the upstream snapshot branch, hence the full fetch.
fork-checks:
if: ${{ vars.BUILD_ON != 'github' }}
runs-on: light
container:
image: python:3.13-slim@sha256:8d9d0b8bcf6506481eae4907c18f5e3e7902e629f5f6d684f9e7c32e85e3ddf0 # 3.13-slim
@@ -58,7 +52,6 @@ jobs:
run: python3 -m unittest discover -s tools/fork/tests
build:
if: ${{ vars.BUILD_ON != 'github' }}
# Either runner (host1 or host2): the build needs no docker socket.
runs-on: light
container:
@@ -108,39 +101,3 @@ jobs:
used=$(du -s --block-size=1G /cache/target 2>/dev/null | cut -f1)
echo "target dir: ${used:-0} GB"
if [ "${used:-0}" -gt 60 ]; then rm -rf /cache/target && echo "over 60 GB: target dir cleared"; fi
# BUILD_ON=github: the GitHub mirror builds this commit and posts the result
# back as the commit status "github/ci (branch)". This waits for that status
# and takes its answer. The mirror pushes on every commit, so a missing
# status means GitHub has not got the push or is not running: after the
# timeout this fails, which is the cue to unset BUILD_ON.
github:
if: ${{ vars.BUILD_ON == 'github' }}
# Its own runner label with plenty of slots: this job only polls, but holds a slot
# for as long as the GitHub build takes, and must not starve the build runners.
runs-on: wait
timeout-minutes: 150
container:
image: python:3.13-slim@sha256:8d9d0b8bcf6506481eae4907c18f5e3e7902e629f5f6d684f9e7c32e85e3ddf0 # 3.13-slim
steps:
- env:
TOKEN: ${{ secrets.GITHUB_TOKEN }}
SHA: ${{ github.event.pull_request.head.sha || github.sha }}
CONTEXT: github/ci (branch)
run: |
python3 - <<'EOF'
import json, os, time, urllib.request
url = (f"{os.environ['CI_SERVER_INTERNAL']}/api/v1/repos/{os.environ['GITHUB_REPOSITORY']}"
f"/commits/{os.environ['SHA']}/statuses?limit=50")
req = urllib.request.Request(url, headers={"Authorization": f"token {os.environ['TOKEN']}"})
ctx, last = os.environ["CONTEXT"], None
print(f"waiting for '{ctx}' on {os.environ['SHA']}", flush=True)
while True:
mine = [s for s in json.load(urllib.request.urlopen(req)) if s["context"] == ctx]
state = max(mine, key=lambda s: s["id"]) if mine else None
if state and state["status"] != last:
last = state["status"]; print(f"{ctx}: {last} {state.get('target_url', '')}", flush=True)
if last == "success": raise SystemExit(0)
if last in ("failure", "error"): raise SystemExit(1)
time.sleep(20)
EOF
+1 -54
View File
@@ -42,14 +42,6 @@
#
# The push logs in with PACKAGE_TOKEN (jcoffey-dev, write:package): the job's
# own token is refused by the container registry.
#
# BUILD_ON: when the Actions variable BUILD_ON is 'github' (org or repo), every
# job here but the announcement skips, and the tag is published by
# .github/workflows/ci.yml on the GitHub mirror instead -- same guards, same
# tags, the same Release and binaries, created here through the API. The
# `github` job waits for that run's commit status, "github/ci (tag)", and the
# announcement follows it as it follows the binaries here. Unset, everything
# runs here as before.
name: publish
on:
@@ -58,7 +50,6 @@ on:
jobs:
version:
if: ${{ vars.BUILD_ON != 'github' }}
runs-on: light
container:
image: python:3.13-slim@sha256:8d9d0b8bcf6506481eae4907c18f5e3e7902e629f5f6d684f9e7c32e85e3ddf0 # 3.13-slim
@@ -97,7 +88,6 @@ jobs:
echo "version $V"
publish-amd64:
if: ${{ vars.BUILD_ON != 'github' }}
needs: [version]
runs-on: docker
container:
@@ -138,7 +128,6 @@ jobs:
run: docker logout "$REGISTRY" || true
publish-arm64:
if: ${{ vars.BUILD_ON != 'github' }}
needs: [version, publish-amd64]
runs-on: docker
container:
@@ -173,46 +162,11 @@ jobs:
- if: always()
run: docker logout "$REGISTRY" || true
# BUILD_ON=github: waits for the GitHub mirror's run for this tag, which
# posts its result back as the commit status "github/ci (tag)", and takes
# its answer. Fails after the timeout if no answer comes.
github:
if: ${{ vars.BUILD_ON == 'github' }}
# Its own runner label with plenty of slots: this job only polls, but holds a slot
# for as long as the GitHub build takes, and must not starve the build runners.
runs-on: wait
timeout-minutes: 240
container:
image: python:3.13-slim@sha256:8d9d0b8bcf6506481eae4907c18f5e3e7902e629f5f6d684f9e7c32e85e3ddf0 # 3.13-slim
steps:
- env:
TOKEN: ${{ secrets.GITHUB_TOKEN }}
SHA: ${{ github.sha }}
CONTEXT: github/ci (tag)
run: |
python3 - <<'EOF'
import json, os, time, urllib.request
url = (f"{os.environ['CI_SERVER_INTERNAL']}/api/v1/repos/{os.environ['GITHUB_REPOSITORY']}"
f"/commits/{os.environ['SHA']}/statuses?limit=50")
req = urllib.request.Request(url, headers={"Authorization": f"token {os.environ['TOKEN']}"})
ctx, last = os.environ["CONTEXT"], None
print(f"waiting for '{ctx}' on {os.environ['SHA']}", flush=True)
while True:
mine = [s for s in json.load(urllib.request.urlopen(req)) if s["context"] == ctx]
state = max(mine, key=lambda s: s["id"]) if mine else None
if state and state["status"] != last:
last = state["status"]; print(f"{ctx}: {last} {state.get('target_url', '')}", flush=True)
if last == "success": raise SystemExit(0)
if last in ("failure", "error"): raise SystemExit(1)
time.sleep(20)
EOF
# The weekly release creates its Release (and so the tag) first; a tag
# pushed by hand has none. Either way the tag ends up with exactly one
# Release, created once the amd64 image exists so its pull instructions
# work; arm64 and the binaries follow.
release:
if: ${{ vars.BUILD_ON != 'github' }}
needs: [version, publish-amd64]
runs-on: light
container:
@@ -262,7 +216,6 @@ jobs:
# `docker create` does not start anything, so pulling an arm64 image on an
# amd64 runner and copying a file out of it needs no emulation.
binaries:
if: ${{ vars.BUILD_ON != 'github' }}
needs: [version, publish-arm64, release]
runs-on: docker
container:
@@ -336,14 +289,8 @@ jobs:
# The release above is made with the job's own token, and Gitea starts no
# workflow for events the Actions bot causes -- announce.yml's
# 'on: release' never fires for it -- so announce it from here.
#
# With BUILD_ON=github the release and binaries come from the GitHub run,
# so the announcement waits for the `github` job instead. The Release that
# run creates for a hand-pushed tag is made with a user token, so
# announce.yml fires for it too; discourse-release keeps one topic per tag.
announce:
needs: [release, binaries, github]
if: ${{ always() && ((needs.release.result == 'success' && needs.binaries.result == 'success') || needs.github.result == 'success') }}
needs: [release, binaries]
runs-on: light
steps:
- uses: coffey-labs/actions/discourse-release@e9293996e2efa770839121fa8f8da93083f216be
+5
View File
@@ -5,6 +5,11 @@
version: 2
updates:
- package-ecosystem: "cargo" # See documentation for possible values
directory: "/" # Location of package manifests
schedule:
interval: "weekly"
# Enable version updates for GitHub Actions
- package-ecosystem: "github-actions"
# Workflow files stored in the default location of `.github/workflows`
@@ -12,7 +12,7 @@ jobs:
runs-on: ubuntu-latest
steps:
- name: Close issues from non-allowed authors
uses: actions/github-script@v9
uses: actions/github-script@v7
with:
script: |
// Users allowed to open issues directly. All other authors will have
@@ -18,7 +18,7 @@ jobs:
sparse-checkout-cone-mode: false
- name: Close PRs from non-allowed authors
uses: actions/github-script@v9
uses: actions/github-script@v7
with:
script: |
const fs = require('fs');
@@ -12,7 +12,7 @@ jobs:
runs-on: ubuntu-latest
steps:
- name: Post support portal redirect
uses: actions/github-script@v9
uses: actions/github-script@v7
with:
script: |
const discussion = context.payload.discussion;
+1 -1
View File
@@ -73,6 +73,6 @@ jobs:
# Upload the results to GitHub's code scanning dashboard (optional).
# Commenting out will disable upload of results to your repo's Code Scanning dashboard
- name: "Upload to code-scanning"
uses: github/codeql-action/[email protected]8.2
uses: github/codeql-action/[email protected]7.4
with:
sarif_file: results.sarif
+1 -1
View File
@@ -36,6 +36,6 @@ jobs:
severity: 'CRITICAL,HIGH'
- name: Upload Trivy scan results to GitHub Security tab
uses: github/codeql-action/[email protected]8.2
uses: github/codeql-action/[email protected]7.4
with:
sarif_file: 'trivy-results.sarif'
+42
View File
@@ -0,0 +1,42 @@
version: 2
updates:
# Cargo. One entry: the workspace has a single lockfile at the root, and
# ~30 manifests that upstream bumps on every release -- pointing entries at
# individual crates would find manifests with no lockfile beside them.
#
# Minor and patch arrive as one pull request a week. Majors are left out of
# the group on purpose: they are migrations rather than bumps, and each one
# deserves its own pull request and its own CI run.
- package-ecosystem: cargo
directory: "/"
schedule:
interval: weekly
day: tuesday
time: "09:00"
timezone: Etc/UTC
open-pull-requests-limit: 5
groups:
minor-and-patch:
update-types:
- minor
- patch
- package-ecosystem: github-actions
directory: "/"
schedule:
interval: weekly
day: tuesday
time: "09:00"
timezone: Etc/UTC
groups:
actions:
patterns:
- "*"
# The Dockerfiles pin their base images, so this is what keeps a published
# image off a stale base between releases.
- package-ecosystem: docker
directory: "/"
schedule:
interval: weekly
day: tuesday
time: "09:00"
timezone: Etc/UTC
+37 -468
View File
@@ -1,482 +1,51 @@
# CI and publishing on GitHub, for the repository Gitea mirrors here.
# What CI can check without a mail server's worth of infrastructure.
#
# Gitea (git.coffeylabs.org) is where this project lives: pull requests,
# issues, releases and the container registry are all there, and it pushes
# every branch and tag to this GitHub copy as it changes. GitHub's hosted
# runners are faster than the self-hosted ones -- and have native arm64 -- so
# the building happens here, and the answer goes back to Gitea as a commit
# status that Gitea's own ci.yml / publish.yml wait on.
# The build, and that every test target compiles. It deliberately does not
# *run* the test suites: the unit tests only build with the integration crate
# in the graph, because that is what switches on the `test_mode` features they
# rely on (docs/spec/SPEC.md 2.2b), and the integration suites need a `STORE`,
# fixed ports, and in most cases a container apiece (docs/spec/
# container-tests.md). Running them here would mean either a green tick that
# skipped everything, or a red one that means "the runner has no Redis".
#
# One switch decides which side builds: the Actions variable BUILD_ON, set on
# both forges. BUILD_ON=github runs every job below and turns Gitea's heavy
# jobs into a wait for this one; anything else leaves Gitea building exactly
# as before and every job here skips. If GitHub is ever unavailable, unset it
# on Gitea and nothing else has to change.
#
# Needs, as organization settings rather than anything in this file:
# variables BUILD_ON=github, REGISTRY (the Gitea container registry),
# GITEA_URL (the Gitea base URL)
# secret GITEA_TOKEN -- jcoffey-dev, write:repository + write:package:
# commit statuses, the release and its assets, the registry push
#
# There is no pull_request trigger: pull requests happen on Gitea, and their
# branch arrives here as an ordinary push. Branch pushes get what Gitea's
# ci.yml checks; v* tags get what its publish.yml does. Schedules (the weekly
# release, the upstream watch) and the release announcement stay on Gitea.
#
# Every `uses:` is pinned to a full commit SHA with the release in the
# trailing comment. A tag is a mutable pointer; do not "simplify" a pin back
# to one. Only GitHub's own actions and the three docker/* ones are used.
name: ci
# So this catches what it can honestly catch -- code that does not compile,
# including test code -- and the suites are run by hand, one at a time, as
# that page describes. If that changes, it changes because someone made the
# suites runnable unattended, not because CI started ignoring failures.
name: CI
on:
push:
branches: ['**']
tags: ['**']
branches: [main]
pull_request:
# Lets CI be run by hand against any ref, including one that predates a CI
# change, without pushing an empty commit to move it.
workflow_dispatch:
# A newer push to a branch cancels the run for the older one, whose answer is
# about code nobody is looking at any more. A tag run is never cancelled: it
# publishes.
# A second push to a branch cancels the run still going for the first: the
# older run's answer is about code nobody is looking at any more.
concurrency:
group: ci-${{ github.ref }}
cancel-in-progress: ${{ github.ref_type == 'branch' }}
permissions:
contents: read
env:
GITEA_URL: ${{ vars.GITEA_URL }}
# The Gitea status this run answers for. Gitea waits on the one matching
# its own event: "(branch)" from ci.yml, "(tag)" from publish.yml.
STATUS_CONTEXT: github/ci (${{ github.ref_type }})
cancel-in-progress: true
jobs:
# Tells Gitea a run has started, so a pull request shows it as pending
# rather than missing while the build is still going.
start:
if: ${{ vars.BUILD_ON == 'github' }}
runs-on: ubuntu-latest
steps:
- env:
GITEA_TOKEN: ${{ secrets.GITEA_TOKEN }}
run: |
jq -n --arg c "$STATUS_CONTEXT" \
--arg u "$GITHUB_SERVER_URL/$GITHUB_REPOSITORY/actions/runs/$GITHUB_RUN_ID" \
'{state:"pending", context:$c, target_url:$u, description:"GitHub Actions"}' |
curl -fsS -o /dev/null -X POST -H "Authorization: token $GITEA_TOKEN" \
-H 'Content-Type: application/json' --data @- \
"$GITEA_URL/api/v1/repos/$GITHUB_REPOSITORY/statuses/$GITHUB_SHA"
# ----------------------------------------------------------- branches ------
# What an upstream merge can bring in or leave behind without a conflict:
# the upstream name in a new string literal, and a changed upstream file
# without the AGPL 5(a) notice. Seconds, and needs no toolchain. The notice
# check diffs against the upstream snapshot in the history, hence the full
# fetch.
fork-checks:
if: ${{ vars.BUILD_ON == 'github' && github.ref_type == 'branch' }}
runs-on: ubuntu-latest
steps:
- uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1
with:
fetch-depth: 0
- run: python3 tools/fork/name-check.py
- if: always()
run: python3 tools/fork/notice-check.py
# Cargo can patch a dependency to a directory in this repository, and
# the image builds from a context .dockerignore prunes to almost
# nothing. CI never sees the difference; a release does.
- if: always()
run: python3 tools/fork/context-check.py
# The personal-data catalog must classify every object and field the
# schema has, and name nothing that is gone.
- if: always()
run: python3 tools/fork/privacy-check.py
# The admin reads each expression field's allowed values and variables
# from the schema; they're generated from the registry and must match it.
- if: always()
run: python3 tools/fork/expr-schema.py --check
- if: always()
run: python3 -m unittest discover -s tools/fork/tests
# The build, and that every test target compiles. The suites are not run:
# they need a store, fixed ports and containers (docs/spec/
# container-tests.md), and are run by hand.
build:
if: ${{ vars.BUILD_ON == 'github' && github.ref_type == 'branch' }}
runs-on: ubuntu-latest
env:
CARGO_INCREMENTAL: "0"
# Debug info is most of a dev target dir, and nothing here runs a
# debugger. Without it the dev and test builds fit the runner's disk and
# the cache below stays small enough to be worth restoring.
CARGO_PROFILE_DEV_DEBUG: "0"
CARGO_PROFILE_TEST_DEBUG: "0"
steps:
# The hosted image carries toolchains this build never touches; a dev,
# test and release build of RocksDB and the workspace needs the room.
- run: |
sudo rm -rf /usr/share/dotnet /usr/local/lib/android /opt/ghc /opt/hostedtoolcache/CodeQL
df -h /
# Every `uses:` here is pinned to a full commit SHA, with the release it
# belongs to in the trailing comment. A tag is a mutable pointer, so
# trusting `@v7` is trusting every future version of that action,
# including one pushed by whoever compromises the account. Dependabot
# updates both halves together -- do not "simplify" a pin back to a tag.
- uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1
# Current stable, as Gitea's rust:1 image is.
- id: rust
run: |
rustup toolchain install stable --profile minimal
rustup default stable
echo "version=$(rustc -V | cut -d' ' -f2)" >> "$GITHUB_OUTPUT"
- run: sudo apt-get update -qq && sudo apt-get install -y -qq --no-install-recommends clang >/dev/null
# Cargo's download cache and the dev/test target dir, keyed on the
# lockfile and the compiler. Saved from main only, so the one cache
# every branch restores is main's, and branches cannot evict it.
- uses: actions/cache/restore@55cc8345863c7cc4c66a329aec7e433d2d1c52a9 # v6.1.0
with:
path: |
~/.cargo/registry/index
~/.cargo/registry/cache
~/.cargo/git/db
target/debug
key: cargo-${{ steps.rust.outputs.version }}-${{ hashFiles('Cargo.lock') }}
restore-keys: cargo-${{ steps.rust.outputs.version }}-
- run: cargo build -p inbuxa --locked
# --no-run: compiles every test target without running them, which
# catches a test that no longer builds without needing a store.
- run: cargo test --workspace --locked --no-run
- if: github.ref == 'refs/heads/main'
uses: actions/cache/save@55cc8345863c7cc4c66a329aec7e433d2d1c52a9 # v6.1.0
with:
path: |
~/.cargo/registry/index
~/.cargo/registry/cache
~/.cargo/git/db
target/debug
key: cargo-${{ steps.rust.outputs.version }}-${{ hashFiles('Cargo.lock') }}
# The release profile, on main only. It is the profile the image is
# built with, and it fails in ways the dev profile does not: v2026.9.24
# was tagged on a commit whose CI was green and whose release build
# could not compile the scim crate at all.
- if: github.ref == 'refs/heads/main'
run: cargo build -p inbuxa --locked --release
# --------------------------------------------------------------- tags ------
# Two guards before anything is pushed, the same as Gitea's publish.yml:
# * the tag must be v<brand_version!>. The version is a string in
# crates/types/src/branding.rs, not Cargo.toml, and the image is tagged
# with it, so a tag beside an unbumped macro would publish an image that
# reports a different version from its tag.
# * the tag must be on main or on a release/* branch, so an image never
# describes code that was never reviewed onto one of them. A release/*
# branch carries a hotfix cut from an earlier release tag.
version:
if: ${{ vars.BUILD_ON == 'github' && github.ref_type == 'tag' && startsWith(github.ref_name, 'v') }}
runs-on: ubuntu-latest
outputs:
version: ${{ steps.v.outputs.version }}
steps:
# Full history, and every branch as origin/*: the ancestry check cannot
# be answered from a shallow clone.
- uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1
with:
fetch-depth: 0
- id: v
env:
TAG: ${{ github.ref_name }}
run: |
set -euo pipefail
# Scoped to the macro body: branding.rs holds other string literals,
# and tagging an image from one of those would be worse than failing.
V="$(awk '/macro_rules! brand_version /,/^}/' crates/types/src/branding.rs \
| grep -om1 '"[0-9][^"]*"' | tr -d '"')"
[ -n "$V" ] || { echo "could not read brand_version! from branding.rs" >&2; exit 1; }
if [ "$TAG" != "v$V" ]; then
echo "Tag $TAG names a commit whose brand_version! says $V." >&2
echo "Refusing to publish an image that would report the wrong version." >&2
exit 1
fi
commit="$(git rev-parse "${TAG}^{commit}")"
on=""
for ref in origin/main $(git for-each-ref --format='%(refname:short)' 'refs/remotes/origin/release/*'); do
if git merge-base --is-ancestor "$commit" "$ref"; then on="$ref"; break; fi
done
[ -n "$on" ] || { echo "$TAG is not on main or a release/* branch" >&2; exit 1; }
echo "$TAG is on $on"
echo "version=$V" >> "$GITHUB_OUTPUT"
# Each architecture on its own native runner, side by side. The Dockerfile
# cross-compiles from the build platform, and on the self-hosted runners one
# machine built both one after the other; here two machines build at once,
# each natively (the builder stage picks the matching target, and the
# aarch64 toolchain it installs exists on arm64 too), and the small final
# stage needs no QEMU. amd64 also moves :<version> as soon as it is done, so
# a production deploy can start from it; :latest waits for the index below,
# so it never names an image without arm64.
publish:
needs: [version]
runs-on: ${{ matrix.runner }}
strategy:
fail-fast: false
matrix:
include:
- arch: amd64
runner: ubuntu-latest
- arch: arm64
runner: ubuntu-24.04-arm
env:
VERSION: ${{ needs.version.outputs.version }}
steps:
- run: |
sudo rm -rf /usr/share/dotnet /usr/local/lib/android /opt/ghc /opt/hostedtoolcache/CodeQL
echo "IMAGE=${{ vars.REGISTRY }}/${GITHUB_REPOSITORY,,}" >> "$GITHUB_ENV"
# The release link (fat LTO, one codegen unit) outgrows the runner's
# 16 GB: v2026.9.30's arm64 link was killed for memory. Swap gives it
# room; buildx's container has no memory limit of its own, so it
# reaches the host's swap.
- run: |
sudo fallocate -l 16G /swap.release
sudo chmod 600 /swap.release
sudo mkswap /swap.release >/dev/null
sudo swapon /swap.release
free -g
df -h /
- uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1
- uses: docker/setup-buildx-action@594f3bf4285d9ea8dc53c9a0c9c4092420091003 # v4.4.0
- uses: docker/login-action@dbcb813823bdd20940b903addbd779551569679f # v4.6.0
with:
registry: ${{ vars.REGISTRY }}
username: jcoffey-dev
password: ${{ secrets.GITEA_TOKEN }}
# Attestations off: they add manifests of their own, and the index
# should hold the two images and nothing else. No build cache: GitHub
# scopes a tag run's cache to that tag, so the next release could never
# read it, and each one would park several GB in the repository's 10 GB
# cache and evict main's cargo cache.
- uses: docker/build-push-action@c3c9e263c25d99ce0380d002d59b67737d91b0dc # v7.4.0
with:
context: .
platforms: linux/${{ matrix.arch }}
provenance: false
sbom: false
push: true
tags: |
${{ env.IMAGE }}:${{ env.VERSION }}-${{ matrix.arch }}
${{ matrix.arch == 'amd64' && format('{0}:{1}', env.IMAGE, env.VERSION) || '' }}
# Joins the two per-architecture tags into :<version> and :latest. Built
# from the per-architecture tags rather than :<version>, which by now is
# the amd64 image and would be read as such.
index:
needs: [version, publish]
runs-on: ubuntu-latest
env:
VERSION: ${{ needs.version.outputs.version }}
steps:
- run: echo "IMAGE=${{ vars.REGISTRY }}/${GITHUB_REPOSITORY,,}" >> "$GITHUB_ENV"
- uses: docker/setup-buildx-action@594f3bf4285d9ea8dc53c9a0c9c4092420091003 # v4.4.0
- uses: docker/login-action@dbcb813823bdd20940b903addbd779551569679f # v4.6.0
with:
registry: ${{ vars.REGISTRY }}
username: jcoffey-dev
password: ${{ secrets.GITEA_TOKEN }}
- run: |
docker buildx imagetools create \
--tag "$IMAGE:$VERSION" \
--tag "$IMAGE:latest" \
"$IMAGE:$VERSION-amd64" "$IMAGE:$VERSION-arm64"
docker buildx imagetools inspect "$IMAGE:$VERSION"
# Gitea keeps a container package on its owner; linking it shows it on
# the repository's Packages tab. Idempotent.
- env:
GITEA_TOKEN: ${{ secrets.GITEA_TOKEN }}
run: |
owner="${GITHUB_REPOSITORY%%/*}"; name="${GITHUB_REPOSITORY#*/}"
curl -fsS -o /dev/null -X POST -H "Authorization: token $GITEA_TOKEN" \
"$GITEA_URL/api/v1/packages/${owner,,}/container/$name/-/link/$name" \
|| echo "package already linked (or link refused); not fatal"
# The weekly release creates its Release (and so the tag) on Gitea first; a
# tag pushed by hand has none. Either way the tag ends up with exactly one
# Release there, created once the image exists so its pull instructions
# work.
release:
needs: [version, index]
runs-on: ubuntu-latest
steps:
- env:
TAG: ${{ github.ref_name }}
VERSION: ${{ needs.version.outputs.version }}
GITEA_TOKEN: ${{ secrets.GITEA_TOKEN }}
REGISTRY: ${{ vars.REGISTRY }}
run: |
set -euo pipefail
api="$GITEA_URL/api/v1/repos/$GITHUB_REPOSITORY"
code="$(curl -sS -o /dev/null -w '%{http_code}' -H "Authorization: token $GITEA_TOKEN" "$api/releases/tags/$TAG")"
if [ "$code" = 200 ]; then echo "$TAG already has a release"; exit 0; fi
[ "$code" = 404 ] || { echo "looking up the release for $TAG answered $code" >&2; exit 1; }
image="$REGISTRY/${GITHUB_REPOSITORY,,}:$VERSION"
body="Container image: \`$image\` (linux/amd64, linux/arm64); also \`:latest\`.
Binaries for a host install are attached: \`inbuxa-linux-amd64.tar.gz\` and \`inbuxa-linux-arm64.tar.gz\`, with \`SHA256SUMS\`. Each is the binary out of this release's image for that architecture, so it is the same build. The image grants it \`cap_net_bind_service\`; a host install has to grant that itself (\`setcap\`, or \`AmbientCapabilities\` in the unit) to bind port 25."
jq -n --arg tag "$TAG" --arg name "INBUXA $VERSION" --arg body "$body" \
'{tag_name:$tag, name:$name, body:$body}' |
curl -fsS -X POST -H "Authorization: token $GITEA_TOKEN" -H 'Content-Type: application/json' \
--data @- "$api/releases" | jq -r '"created release " + .tag_name'
# The binaries for a host install, taken out of the image that was just
# pushed rather than compiled again: the binary in the tarball is the file
# the image runs. `docker create` starts nothing, so copying a file out of
# the arm64 image on an amd64 runner needs no emulation.
binaries:
needs: [version, index, release]
runs-on: ubuntu-latest
env:
VERSION: ${{ needs.version.outputs.version }}
TAG: ${{ github.ref_name }}
GITEA_TOKEN: ${{ secrets.GITEA_TOKEN }}
steps:
- run: echo "IMAGE=${{ vars.REGISTRY }}/${GITHUB_REPOSITORY,,}" >> "$GITHUB_ENV"
- uses: docker/login-action@dbcb813823bdd20940b903addbd779551569679f # v4.6.0
with:
registry: ${{ vars.REGISTRY }}
username: jcoffey-dev
password: ${{ secrets.GITEA_TOKEN }}
- name: take the binaries out of the image
run: |
set -euo pipefail
mkdir -p out && cd out
for arch in amd64 arm64; do
docker pull -q --platform "linux/$arch" "$IMAGE:$VERSION"
id="$(docker create --platform "linux/$arch" "$IMAGE:$VERSION")"
docker cp "$id:/usr/local/bin/inbuxa" inbuxa
docker rm -f "$id" >/dev/null
chmod 0755 inbuxa
tar -czf "inbuxa-linux-$arch.tar.gz" inbuxa
rm inbuxa
done
sha256sum inbuxa-linux-*.tar.gz > SHA256SUMS
cat SHA256SUMS
# A re-run of a tag replaces its assets rather than leaving two files
# with the same name and different contents.
#
# The uploads cross Cloudflare, which dropped 50 MB HTTP/2 uploads
# part-way for v2026.9.30.1 (curl 92, PROTOCOL_ERROR; origin logged
# 400), once on each of two runs. Uploads go over HTTP/1.1 and retry.
- name: attach them to the release
run: |
set -euo pipefail
api="$GITEA_URL/api/v1/repos/$GITHUB_REPOSITORY"
auth="Authorization: token $GITEA_TOKEN"
retry=(--retry 5 --retry-all-errors --retry-delay 15)
rel="$(curl -fsS "${retry[@]}" -H "$auth" "$api/releases/tags/$TAG" | jq -r .id)"
assets="$(curl -fsS "${retry[@]}" -H "$auth" "$api/releases/$rel/assets")"
for f in out/inbuxa-linux-amd64.tar.gz out/inbuxa-linux-arm64.tar.gz out/SHA256SUMS; do
name="$(basename "$f")"
old="$(jq -r --arg n "$name" '.[] | select(.name == $n) | .id' <<<"$assets")"
for id in $old; do curl -fsS "${retry[@]}" -o /dev/null -X DELETE -H "$auth" "$api/releases/$rel/assets/$id"; done
curl -fsS --http1.1 "${retry[@]}" -o /dev/null -X POST -H "$auth" -F "attachment=@$f" "$api/releases/$rel/assets?name=$name"
echo "attached $name"
done
# ------------------------------------------------------ ghcr replica ------
# Copies the release image from the Gitea registry, which stays the
# authoritative one, to ghcr.io under the same version tag and :latest. It is
# a copy, not a second build: the digest on GHCR is the digest on the
# registry, so `docker pull ghcr.io/...` gets exactly the same image. Left
# out of the report to Gitea, like the release copy, so a GHCR problem
# cannot fail a release.
ghcr:
if: ${{ vars.BUILD_ON == 'github' && github.ref_type == 'tag' }}
needs: [version, index]
runs-on: ubuntu-latest
permissions:
contents: read
packages: write
steps:
- env:
GH_TOKEN: ${{ github.token }}
TAG: ${{ needs.version.outputs.version }}
run: |
set -euo pipefail
src="${{ vars.REGISTRY }}/${GITHUB_REPOSITORY,,}"
dst="ghcr.io/${GITHUB_REPOSITORY,,}"
tag="$TAG"
echo "$GH_TOKEN" | docker login ghcr.io -u "$GITHUB_ACTOR" --password-stdin
docker buildx imagetools create -t "$dst:$tag" -t "$dst:latest" "$src:$tag"
want="$(docker buildx imagetools inspect "$src:$tag" --format '{{json .Manifest.Digest}}')"
got="$(docker buildx imagetools inspect "$dst:$tag" --format '{{json .Manifest.Digest}}')"
echo "registry $src:$tag = $want"
echo "ghcr $dst:$tag = $got"
[ "$want" = "$got" ] || echo "::warning::GHCR digest differs from the registry's"
docker logout ghcr.io
# ---------------------------------------------------- github release ------
# Copies this tag's Gitea release -- notes and files -- to a GitHub release,
# so the replica's Releases page, and anyone watching it, keeps up. Gitea's
# release is the real one; this is left out of the report to Gitea, so a
# failure here cannot fail a release. PR and issue numbers in the notes are
# rewritten to Gitea links: on GitHub a bare #16 is some other PR.
github-release:
if: ${{ vars.BUILD_ON == 'github' && github.ref_type == 'tag' }}
needs: [binaries]
runs-on: ubuntu-latest
permissions:
contents: write
env:
GITEA_URL: ${{ vars.GITEA_URL }}
GH_TOKEN: ${{ github.token }}
TAG: ${{ github.ref_name }}
steps:
- run: |
set -euo pipefail
if gh release view "$TAG" --repo "$GITHUB_REPOSITORY" >/dev/null 2>&1; then
echo "GitHub already has a release for $TAG"; exit 0
fi
# The Gitea release exists by now if this run made it; if the weekly
# release job made it, it came before the tag. Allow a few minutes.
code=0
for _ in $(seq 1 15); do
code="$(curl -sS -o rel.json -w '%{http_code}' "$GITEA_URL/api/v1/repos/$GITHUB_REPOSITORY/releases/tags/$TAG")"
[ "$code" = 200 ] && break
sleep 20
done
if [ "$code" != 200 ]; then echo "No Gitea release for $TAG; nothing to copy"; exit 0; fi
if [ "$(jq -r .draft rel.json)" = true ]; then echo "The Gitea release is a draft; not copying"; exit 0; fi
export BASE="$(jq -r '.html_url | sub("/releases/tag/.*$"; "")' rel.json)"
jq -r '.body // ""' rel.json | perl -pe 's{(?<![\w/&\[])#(\d+)\b}{[#$1]($ENV{BASE}/pulls/$1)}g' > notes.md
printf '\n\n_Mirrored from [the Gitea release](%s); report issues on [Gitea](%s/issues)._\n' \
"$(jq -r .html_url rel.json)" "$BASE" >> notes.md
files=()
mkdir -p files
while IFS=$'\t' read -r name url; do
curl -fsSL -o "files/$name" "$url"; files+=("files/$name")
done < <(jq -r '.assets[]? | [.name, .browser_download_url] | @tsv' rel.json)
title="$(jq -r '.name // ""' rel.json)"; [ -n "$title" ] || title="$TAG"
if [ "$(jq -r .prerelease rel.json)" = true ]; then kind=--prerelease; else kind=--latest; fi
gh release create "$TAG" --repo "$GITHUB_REPOSITORY" --verify-tag --title "$title" \
--notes-file notes.md "$kind" "${files[@]}"
echo "created the GitHub release for $TAG with ${#files[@]} file(s)"
# ------------------------------------------------------------- report ------
# One commit status on Gitea for the whole run: what Gitea's ci.yml and
# publish.yml wait on. Skipped jobs (the tag jobs on a branch, and the other
# way round) count as passing; a failed or cancelled one does not.
report:
if: ${{ always() && vars.BUILD_ON == 'github' }}
needs: [start, fork-checks, build, version, publish, index, release, binaries]
runs-on: ubuntu-latest
steps:
- env:
GITEA_TOKEN: ${{ secrets.GITEA_TOKEN }}
STATE: ${{ contains(needs.*.result, 'failure') && 'failure' || (contains(needs.*.result, 'cancelled') && 'cancelled' || 'success') }}
run: |
# A cancelled run was superseded by a newer run for the same commit (the
# mirror can push one commit twice); that run reports. Posting "failure"
# here would fail the Gitea check while the real build is still going.
if [ "$STATE" = cancelled ]; then echo "cancelled: leaving the result to the newer run"; exit 0; fi
jq -n --arg s "$STATE" --arg c "$STATUS_CONTEXT" \
--arg u "$GITHUB_SERVER_URL/$GITHUB_REPOSITORY/actions/runs/$GITHUB_RUN_ID" \
'{state:$s, context:$c, target_url:$u, description:"GitHub Actions"}' |
curl -fsS -o /dev/null -X POST -H "Authorization: token $GITEA_TOKEN" \
-H 'Content-Type: application/json' --data @- \
"$GITEA_URL/api/v1/repos/$GITHUB_REPOSITORY/statuses/$GITHUB_SHA"
echo "$STATUS_CONTEXT: $STATE"
- uses: Swatinem/rust-cache@6323deb102c322ba6fcbdcafc7e3dddab59af2b6 # v2.9.2
- name: System dependencies
# foundationdb and the search backends are off by default, but the
# default feature set still links against the system's C libraries.
run: sudo apt-get update && sudo apt-get install -y --no-install-recommends clang
- name: Build the server
run: cargo build -p inbuxa --locked
- name: Compile every test target
# `--no-run` is the point: it builds the unit tests and the integration
# crate together, which is the combination that resolves the test
# features, and stops short of running anything that wants a store.
run: cargo test --workspace --locked --no-run
+69
View File
@@ -0,0 +1,69 @@
# Prune old image versions from GHCR.
#
# Releases are kept forever -- they carry no assets and their generated notes
# are this project's only changelog, so deleting one destroys history that
# cannot be reconstructed for nothing saved. Images are the opposite: a
# multi-arch build a week, and the by-digest push in publish.yml leaves two
# untagged per-architecture manifests behind each time on top of the tagged
# index. Those accumulate and nobody wants fifty of them.
#
# THE FOOTGUN: the obvious tool for this -- delete-package-versions with
# `delete-only-untagged-versions` -- will happily delete the per-architecture
# manifests that a multi-arch tag points *at*, because they are untagged by
# design. Nothing appears to break: the tag still exists, and pulls simply
# start failing for one architecture. This action understands manifest lists
# and will not orphan a retained index, and `validate` re-checks every
# multi-arch manifest against the registry afterwards.
#
# Separate from publish.yml, and dispatchable on its own, so `dry_run` can show
# exactly what would be deleted without rebuilding and re-pushing an image to
# find out.
name: Prune images
on:
workflow_call:
inputs:
dry_run:
type: boolean
default: false
workflow_dispatch:
inputs:
dry_run:
description: "List what would be deleted, delete nothing"
type: boolean
default: true
jobs:
prune:
runs-on: ubuntu-latest
permissions:
packages: write
steps:
# The only third-party action here that is not published by GitHub or
# Docker, and the one with the most to lose: it is handed
# `packages: write` and its whole job is deletion, so a ref repointed at
# something else -- by a compromise or a mistake upstream -- is a bad
# day. It was pinned to a commit long before the rest of them were.
- uses: dataaxiom/ghcr-cleanup-action@d52806a0dc70b430571a37da1fde39733ffd640f # v1.2.2
with:
owner: inbuxa
package: inbuxa-server
token: ${{ secrets.GITHUB_TOKEN }}
# Ten weekly releases is roughly a quarter of history, which is more
# than enough to roll back to and far less than the year's worth that
# would otherwise pile up. Older *releases* stay either way; this
# only removes the images.
keep-n-tagged: 10
# Belt and braces on top of the action's own manifest awareness:
# `latest` is never a candidate for deletion under any counting.
exclude-tags: latest
delete-untagged: true
# Sweeps the wreckage of a half-failed run: an index whose platform
# images did not all land, and referrers whose parent is gone.
delete-partial-images: true
delete-orphaned-images: true
# Checks every remaining multi-architecture manifest still resolves
# in the registry. This is the step that would catch the footgun
# above rather than leaving a reader to discover it on `docker pull`.
validate: true
dry-run: ${{ inputs.dry_run }}
+198
View File
@@ -0,0 +1,198 @@
# Publish the container image to GHCR.
#
# The README and the docs site have told people to run
# `ghcr.io/inbuxa/inbuxa-server:latest` for a long time, and nothing ever
# pushed it: `docker pull` answered `denied`, because the package did not
# exist. This is the workflow that makes those instructions true. It is also
# the prerequisite for the self-hosted app catalogs -- TrueNAS and Unraid
# both install by pulling an image and neither builds from source.
#
# FIRST RUN: a package GHCR creates for the first time is **private**, even in
# a public repository, and an anonymous `docker pull` will still answer
# `denied`. Nothing in a workflow can change that -- the visibility is set once
# by hand under the package's settings, and until it is, this looks like it
# worked while the docs stay just as wrong as before. Check with a logged-out
# pull, not with one from a machine that has credentials.
#
# Two architectures, each built on its own native runner rather than under
# QEMU. Emulated arm64 has to run `npm ci` and the Vite build through
# instruction translation, which takes tens of minutes and occasionally runs
# out of memory; `ubuntu-24.04-arm` is free for public repositories and does
# the same work at native speed. The cost is the by-digest dance below: each
# runner pushes an untagged image, and a final job joins the two digests into
# one multi-arch tag.
name: Publish image
on:
release:
types: [published]
# Callable, so release.yml can build the release it just cut. This is not a
# stylistic choice: a release created with GITHUB_TOKEN does **not** raise a
# `release` event -- GitHub refuses to let a token trigger another workflow,
# to stop a workflow looping on its own output. A scheduled job that cut a
# release and expected this file to notice would silently never publish. The
# alternatives are a personal access token kept as a secret, or calling the
# workflow directly. This is the one that needs no credential.
workflow_call:
inputs:
ref:
description: "Tag, branch or SHA to build"
required: true
type: string
tag_latest:
description: "Also move :latest to this build"
type: boolean
default: false
# Same reasoning as ci.yml's dispatch trigger: a run GitHub queues and then
# orphans can be neither rerun nor canceled, and this workflow otherwise
# only fires on a release -- which is not something to cut twice because a
# runner died. `ref` also allows publishing an image for a tag that predates
# this workflow, which is how the first one gets built.
workflow_dispatch:
inputs:
ref:
description: "Tag, branch or SHA to build"
required: true
default: main
tag_latest:
description: "Also move :latest to this build"
type: boolean
default: false
env:
# Hardcoded rather than derived from github.repository, which would have to
# be lowercased to be a legal registry path. This is the string the docs name.
IMAGE: ghcr.io/inbuxa/inbuxa-server
jobs:
# The version is read once and handed to both builds, so the two
# architectures cannot disagree about what they are. It is read from the
# macro the binary itself compiles in, which the weekly release commits
# before this runs -- so the image is tagged with the version it reports.
version:
runs-on: ubuntu-latest
outputs:
version: ${{ steps.v.outputs.version }}
steps:
- uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1
with:
ref: ${{ inputs.ref || github.ref }}
- id: v
run: |
set -euo pipefail
# Scoped to the macro body: branding.rs holds other string literals,
# and tagging an image from one of those would be worse than failing.
V="$(awk '/macro_rules! brand_version/,/^}/' crates/types/src/branding.rs \
| grep -om1 '"[0-9][^"]*"' | tr -d '"')"
[ -n "$V" ] || { echo "could not read brand_version! from branding.rs" >&2; exit 1; }
# A date version carries nothing a Docker tag objects to, so there is
# no second, sanitized form of it here.
echo "version=$V" >> "$GITHUB_OUTPUT"
echo "version $V"
build:
needs: version
runs-on: ${{ matrix.runner }}
permissions:
contents: read
packages: write
strategy:
fail-fast: false
matrix:
include:
- platform: linux/amd64
runner: ubuntu-latest
- platform: linux/arm64
runner: ubuntu-24.04-arm
steps:
- uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1
with:
ref: ${{ inputs.ref || github.ref }}
- uses: docker/setup-buildx-action@594f3bf4285d9ea8dc53c9a0c9c4092420091003 # v4.4.0
- uses: docker/login-action@dbcb813823bdd20940b903addbd779551569679f # v4.6.0
with:
registry: ghcr.io
username: ${{ github.actor }}
password: ${{ secrets.GITHUB_TOKEN }}
- name: Build and push by digest
id: push
uses: docker/build-push-action@c3c9e263c25d99ce0380d002d59b67737d91b0dc # v7.4.0
with:
context: .
platforms: ${{ matrix.platform }}
# Attestations are off deliberately: they add manifests of their own
# to the index, and `imagetools create` below expects the two entries
# it pushed rather than four.
provenance: false
sbom: false
cache-from: type=gha,scope=${{ matrix.platform }}
cache-to: type=gha,mode=max,scope=${{ matrix.platform }}
outputs: type=image,name=${{ env.IMAGE }},push-by-digest=true,name-canonical=true,push=true
- name: Save the digest
run: |
mkdir -p /tmp/digests
# The prefix is stripped here and put back in the merge job, so the
# filename is the bare hash. Leaving it on produces
# `image@sha256:sha256:...` when the reference is rebuilt.
digest="${{ steps.push.outputs.digest }}"
touch "/tmp/digests/${digest#sha256:}"
- uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7.0.1
with:
# One artifact per platform; the merge job globs them back together.
name: digest-${{ strategy.job-index }}
path: /tmp/digests/*
retention-days: 1
if-no-files-found: error
# Joins the per-architecture digests into a single tagged manifest, so
# `docker pull ghcr.io/inbuxa/inbuxa-server:<tag>` resolves on both.
publish:
needs: [version, build]
runs-on: ubuntu-latest
permissions:
contents: read
packages: write
steps:
- uses: actions/download-artifact@3e5f45b2cfb9172054b4087a40e8e0b5a5461e7c # v8.0.1
with:
path: /tmp/digests
pattern: digest-*
merge-multiple: true
- uses: docker/setup-buildx-action@594f3bf4285d9ea8dc53c9a0c9c4092420091003 # v4.4.0
- uses: docker/login-action@dbcb813823bdd20940b903addbd779551569679f # v4.6.0
with:
registry: ghcr.io
username: ${{ github.actor }}
password: ${{ secrets.GITHUB_TOKEN }}
- name: Create the manifest
run: |
# Arrays rather than a string: the tags and the digest references
# have to reach docker as separate arguments, and building them by
# word-splitting an unquoted variable is the version of this that
# breaks the day a value contains a space.
tags=(-t "${IMAGE}:${{ needs.version.outputs.version }}")
# :latest follows real releases only. A prerelease that moved it
# would hand every `:latest` deployment an unfinished build, and a
# dispatch run has to ask for it on purpose.
if [ "${{ github.event_name }}" = "release" ] && [ "${{ github.event.release.prerelease }}" = "false" ]; then
tags+=(-t "${IMAGE}:latest")
elif [ "${{ inputs.tag_latest }}" = "true" ]; then
tags+=(-t "${IMAGE}:latest")
fi
refs=()
for f in /tmp/digests/*; do
refs+=("${IMAGE}@sha256:$(basename "$f")")
done
echo "tags: ${tags[*]}"
echo "refs: ${refs[*]}"
docker buildx imagetools create "${tags[@]}" "${refs[@]}"
- name: Show what landed
run: docker buildx imagetools inspect "${IMAGE}:${{ needs.version.outputs.version }}"
# Runs only after a successful publish, because that is the only moment the
# package grows. See cleanup.yml for why this is not the obvious one-liner.
prune:
needs: publish
permissions:
packages: write
uses: ./.github/workflows/cleanup.yml
+246
View File
@@ -0,0 +1,246 @@
# Cut a release once a week, but only if there is something in it.
#
# It does nothing on a quiet week. A release with no commits in it is worse
# than no release: it moves `:latest` to an identical build, spends a version
# number, and mails everybody watching the repository about nothing.
#
# INBUXA's version is a string in crates/types/src/branding.rs, deliberately
# not in Cargo.toml so that upstream's version bumps merge without conflicts.
# So this writes it: the bump is committed to main, and the tag names that
# commit. The tree a tag points at therefore reports the version the tag
# claims, which a tag placed beside an unbumped macro cannot promise.
name: Weekly release
on:
schedule:
# Mondays, 10:07 UTC, and last of the three: INBUXA Admin and the webmail
# release ahead of the server they talk to. Staggered rather than
# simultaneous so three releases do not compete for runners, and so a bad
# Monday names one repository instead of three. GitHub runs scheduled jobs
# best-effort and can delay a run considerably, so the exact minute is not
# a promise; the odd minute keeps it off the crowded top of the hour.
#
# Note also that GitHub disables scheduled workflows in a repository with
# no activity for 60 days, which is worth checking for before assuming
# this file is broken.
- cron: "7 10 * * 1"
workflow_dispatch:
inputs:
dry_run:
description: "Work out what would be released, then stop"
type: boolean
default: false
# One at a time. Two overlapping runs would race to write the same version and
# create the same tag, and the loser fails noisily for a reason that has
# nothing to do with the code.
concurrency:
group: weekly-release
cancel-in-progress: false
jobs:
check:
runs-on: ubuntu-latest
permissions:
contents: read
outputs:
should_release: ${{ steps.decide.outputs.should_release }}
version: ${{ steps.decide.outputs.version }}
tag: ${{ steps.decide.outputs.tag }}
previous: ${{ steps.decide.outputs.previous }}
count: ${{ steps.decide.outputs.count }}
steps:
- uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1
with:
ref: main
fetch-depth: 0
- id: decide
env:
GH_TOKEN: ${{ secrets.GITHUB_TOKEN }}
run: |
set -euo pipefail
# The newest published release, or empty on a repository that has
# never had one -- in which case everything counts as new. Drafts are
# excluded: an unpublished draft is not a release anybody has, so
# counting from it would hide commits that have never shipped.
previous="$(gh release list --limit 1 --exclude-drafts --json tagName --jq '.[0].tagName // ""')"
# A tag named by a release is normally present after a full checkout,
# but a release can outlive its tag. Falling back to the whole
# history is the safe direction to be wrong in: it over-counts, which
# cuts a release that was due anyway, where under-counting would skip
# one that was.
if [ -n "$previous" ] && git rev-parse -q --verify "refs/tags/${previous}" >/dev/null; then
count="$(git rev-list --count "${previous}..HEAD")"
else
count="$(git rev-list --count HEAD)"
fi
# INBUXA's version is the date: YYYY.M.D, unpadded, as branding.rs
# documents. A second release on one day takes a `.N` suffix,
# counting from 2, which is why this asks the tags rather than
# assuming today is free.
today="$(date -u +%Y.%-m.%-d)"
version="$today"
n=2
while git rev-parse -q --verify "refs/tags/v${version}" >/dev/null; do
version="${today}.${n}"
n=$((n + 1))
done
should_release=true
reason=""
if [ "$count" -eq 0 ]; then
should_release=false
reason="no commits since ${previous}"
fi
{
echo "should_release=$should_release"
echo "version=$version"
echo "tag=v${version}"
echo "previous=$previous"
echo "count=$count"
} >> "$GITHUB_OUTPUT"
# Written to the run summary so a skipped week reads as a decision
# rather than as a workflow that quietly did nothing.
{
echo "### Weekly release"
echo
if [ "$should_release" = "true" ]; then
echo "Releasing **v${version}** — ${count} commit(s) since ${previous:-the beginning}."
else
echo "Nothing to release: ${reason}."
fi
} >> "$GITHUB_STEP_SUMMARY"
cut:
needs: check
if: needs.check.outputs.should_release == 'true' && !inputs.dry_run
runs-on: ubuntu-latest
permissions:
contents: write
pull-requests: write
outputs:
sha: ${{ steps.land.outputs.sha }}
steps:
- uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1
with:
ref: main
fetch-depth: 0
- id: bump
env:
VERSION: ${{ needs.check.outputs.version }}
BRANCH: release/v${{ needs.check.outputs.version }}
run: |
set -euo pipefail
# Scoped to the macro body rather than replacing the first quoted
# string in the file, and asserted to have matched exactly once.
# branding.rs holds other string literals, and a bump that silently
# edited one of those -- or none -- would ship a build whose version
# disagrees with its tag.
python3 - <<'PY'
import os, re
path = "crates/types/src/branding.rs"
src = open(path, encoding="utf-8").read()
pattern = re.compile(r'(macro_rules! brand_version \{\s*\(\) => \{\s*")[^"]+(")')
out, n = pattern.subn(lambda m: m.group(1) + os.environ["VERSION"] + m.group(2), src, count=1)
assert n == 1, f"brand_version! not found in {path}"
open(path, "w", encoding="utf-8").write(out)
PY
git config user.name "github-actions[bot]"
git config user.email "41898282+github-actions[bot]@users.noreply.github.com"
git add crates/types/src/branding.rs
git commit -m "Version ${VERSION}"
git push origin "HEAD:refs/heads/${BRANCH}"
# main is protected: it takes a pull request with a green build, and
# GITHUB_TOKEN is not among the bypass actors. So the bump lands the way
# every other change does. The alternative was to hand the release a
# credential that outranks the rule, which is a worse thing to own than
# a slower Monday.
- id: land
env:
VERSION: ${{ needs.check.outputs.version }}
BRANCH: release/v${{ needs.check.outputs.version }}
GH_TOKEN: ${{ secrets.GITHUB_TOKEN }}
run: |
set -euo pipefail
url="$(gh pr create --base main --head "${BRANCH}" \
--title "Version ${VERSION}" \
--body "Weekly release. Bumps \`brand_version!\` to ${VERSION} so the tag names a tree that reports the version the tag claims.")"
# The number, not the branch: the branch is deleted on merge, and a
# deleted branch no longer resolves to its pull request.
pr="${url##*/}"
echo "Opened #${pr}"
# The build is what the rule actually requires, and it is also the
# thing worth waiting for: a release cut from a tree that does not
# compile is the failure this whole arrangement exists to prevent.
# A full build of this tree is long, so the deadline is generous.
deadline=$(( SECONDS + 3600 ))
while :; do
state="$(gh pr view "${pr}" --json statusCheckRollup \
--jq '[.statusCheckRollup[]? | .conclusion // "PENDING"] | join(",")')"
case "${state}" in
*FAILURE*|*CANCELLED*|*TIMED_OUT*)
echo "::error::CI failed on ${BRANCH} (${state}); no release cut. PR #${pr} is left open."
exit 1 ;;
*SUCCESS*) break ;;
esac
if [ "${SECONDS}" -ge "${deadline}" ]; then
echo "::error::timed out waiting for CI on ${BRANCH}. PR #${pr} is left open."
exit 1
fi
sleep 30
done
gh pr merge "${pr}" --rebase --delete-branch
# A rebase merge rewrites the commit, so the sha to tag is the one
# GitHub recorded for the merge, not the tip that was pushed. It can
# take a moment to appear.
sha=""
for _ in $(seq 1 30); do
sha="$(gh pr view "${pr}" --json mergeCommit --jq '.mergeCommit.oid // ""')"
[ -n "${sha}" ] && break
sleep 5
done
if [ -z "${sha}" ]; then
echo "::error::#${pr} merged but GitHub reported no merge commit; nothing safe to tag."
exit 1
fi
echo "sha=${sha}" >> "$GITHUB_OUTPUT"
- env:
GH_TOKEN: ${{ secrets.GITHUB_TOKEN }}
run: |
set -euo pipefail
args=(--target "${{ steps.land.outputs.sha }}"
--title "INBUXA ${{ needs.check.outputs.version }}"
--generate-notes)
# Bound the notes to what is actually new. Without a start tag the
# generator reaches back to whatever it decides is previous, which on
# a repository carrying upstream's tag shapes is not always the last
# release.
if [ -n "${{ needs.check.outputs.previous }}" ]; then
args+=(--notes-start-tag "${{ needs.check.outputs.previous }}")
fi
gh release create "${{ needs.check.outputs.tag }}" "${args[@]}"
# Called rather than left to the `release` trigger on purpose: see the note
# at the top of publish.yml. A release created with GITHUB_TOKEN raises no
# event, so without this the tag would exist and no image would follow it.
publish:
needs: [check, cut]
permissions:
contents: read
packages: write
uses: ./.github/workflows/publish.yml
with:
ref: ${{ needs.cut.outputs.sha }}
tag_latest: true
-27
View File
@@ -2,33 +2,6 @@
All notable changes to this project will be documented in this file. This project adheres to [Semantic Versioning](http://semver.org/).
## [0.16.25] - 2026-10-05
If you are upgrading from v0.16.x, replace the binary (or run `docker pull`). If you are upgrading from v0.15.x and below, please read the [upgrading documentation](https://github.com/stalwartlabs/stalwart/blob/main/UPGRADING/v0_16.md) for more information on how to upgrade from previous versions.
## Added
## Changed
## Fixed
- JMAP: Creating a `MaskedEmail` with `emailDomain` fails with `forbidden` for every domain when the account has addresses on more than one domain.
- Autodiscover: Requests for a response schema other than Outlook's, such as ActiveSync (`mobilesync`), are answered with the Outlook settings instead of error 601.
- IMAP:
- `LOGIN` and `AUTHENTICATE` with a wrong, expired or unknown app password or API key are answered with an untagged `NO`, so clients keep waiting for the command to complete until the connection times out.
- The failed login that exceeds the maximum number of authentication failures is answered with an untagged `NO` before the connection is closed.
- DKIM:
- A rotation moves the active key to retiring even when its successor fails to publish or propagate, so outgoing mail is sent unsigned until a retry publishes the new key. The DNS write failure is also not logged and the task reports success.
- Keys created while DNS management was manual, or before DKIM was added to the published records, are never rotated after DNS management becomes automatic. Domains already affected start rotating once a `DkimManagement` task is created for them.
- After switching DNS management from automatic to manual, a due rotation activates a new key that was never published in DNS, so signatures fail verification, and retiring the old key is retried forever.
- Spam filter:
- Messages with no text line long enough for a Pyzor digest are checked with the digest of empty input and tagged `PYZOR`.
- DNSBL answers with several return codes, such as a Spamhaus ZEN listing in both SBL and PBL, are scored for only the first code returned.
- DNSBL lookups that return "not listed" are cached for 24 hours regardless of the zone's negative TTL.
- Removing a duplicate training sample of a message reclassified on the same day clears the blob link of the sample that is kept.
- MTA: Queue quotas with an empty `match` expression are never enforced, including the global queue quota created on first start.
- RocksDB: The info log (`LOG`, `LOG.old.*`) grows without limit because log rotation and retention are left at RocksDB defaults.
- WebUI: A blob store read error at startup, such as an S3 authentication failure, stops the web interface from being downloaded.
## [0.16.24] - 2026-09-27
If you are upgrading from v0.16.x, replace the binary (or run `docker pull`). If you are upgrading from v0.15.x and below, please read the [upgrading documentation](https://github.com/stalwartlabs/stalwart/blob/main/UPGRADING/v0_16.md) for more information on how to upgrade from previous versions.
+1 -1
View File
@@ -60,7 +60,7 @@ representative at an online or offline event.
Instances of abusive, harassing, or otherwise unacceptable behavior may be
reported to the community leaders responsible for enforcement at
**communityATcoffeylabsDOTorg**.
**johnellisATlinuxDOTcom**.
All complaints will be reviewed and investigated promptly and fairly.
All community leaders are obligated to respect the privacy and security of the
+1 -1
View File
@@ -54,7 +54,7 @@ Coffey Labs" line in place. New files carry:
```
/*
* SPDX-FileCopyrightText: 2026 Coffey Labs LLC
* SPDX-FileCopyrightText: 2026 Coffey Labs
*
* SPDX-License-Identifier: AGPL-3.0-only
*/
Generated
+69 -87
View File
@@ -277,9 +277,9 @@ dependencies = [
[[package]]
name = "async-compression"
version = "0.4.50"
version = "0.4.48"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "ee19bd99b43e3691acbad4e840420a4881cea6c0b66a208125a824f8fd53f5a1"
checksum = "fb61aea1a7def73ee7c350a184f0e70b32c182344e2e75bf70c9b621b83417fd"
dependencies = [
"compression-codecs",
"compression-core",
@@ -1292,7 +1292,7 @@ dependencies = [
[[package]]
name = "common"
version = "0.16.25"
version = "0.16.24"
dependencies = [
"aes-gcm-siv",
"ahash",
@@ -1392,9 +1392,9 @@ dependencies = [
[[package]]
name = "compression-codecs"
version = "0.4.45"
version = "0.4.43"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "98fc98460ba0ad5317075d3632b8dfc45d0be8c4a49347c2a38272019717614a"
checksum = "bef16c47ba2797aa6a909cc37d39911f3a6743811fe7408ac0b0cc0276b656e9"
dependencies = [
"compression-core",
"flate2",
@@ -1477,7 +1477,7 @@ checksum = "3d52eff69cd5e647efe296129160853a42795992097e8af39800e1060caeea9b"
[[package]]
name = "coordinator"
version = "0.16.25"
version = "0.16.24"
dependencies = [
"async-nats",
"futures",
@@ -1889,7 +1889,7 @@ checksum = "4583a4551df46e2792f82ceeac45e850d2e2d5debba0b91f102385cda5b11f06"
[[package]]
name = "dav"
version = "0.16.25"
version = "0.16.24"
dependencies = [
"calcard",
"chrono",
@@ -1912,7 +1912,7 @@ dependencies = [
[[package]]
name = "dav-proto"
version = "0.16.25"
version = "0.16.24"
dependencies = [
"calcard",
"chrono",
@@ -2125,7 +2125,7 @@ dependencies = [
[[package]]
name = "directory"
version = "0.16.25"
version = "0.16.24"
dependencies = [
"ahash",
"argon2 0.6.0",
@@ -2366,7 +2366,7 @@ dependencies = [
[[package]]
name = "email"
version = "0.16.25"
version = "0.16.24"
dependencies = [
"aes 0.9.3",
"aes-gcm 0.11.1",
@@ -2406,16 +2406,6 @@ dependencies = [
"log",
]
[[package]]
name = "encodify"
version = "1.0.1"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "798c447647dd23f673748f2b868ef309a01dd86aaae999182559d36f06ac82f0"
dependencies = [
"memchr",
"simdutf8",
]
[[package]]
name = "encoding_rs"
version = "0.8.42"
@@ -2484,7 +2474,7 @@ dependencies = [
[[package]]
name = "event_macro"
version = "0.16.25"
version = "0.16.24"
dependencies = [
"quote",
"syn 3.0.6",
@@ -3012,7 +3002,7 @@ dependencies = [
[[package]]
name = "groupware"
version = "0.16.25"
version = "0.16.24"
dependencies = [
"ahash",
"calcard",
@@ -3299,7 +3289,7 @@ dependencies = [
[[package]]
name = "http"
version = "0.16.25"
version = "0.16.24"
dependencies = [
"async-stream",
"base64 0.23.1",
@@ -3395,7 +3385,7 @@ dependencies = [
[[package]]
name = "http_proto"
version = "0.16.25"
version = "0.16.24"
dependencies = [
"common",
"compact_str",
@@ -3882,7 +3872,7 @@ checksum = "65b27460c2c92b037f3f94c538ed9a3342f3fdf923606781629ccb35f82d042a"
[[package]]
name = "imap"
version = "0.16.25"
version = "0.16.24"
dependencies = [
"ahash",
"common",
@@ -3907,7 +3897,7 @@ dependencies = [
[[package]]
name = "imap_proto"
version = "0.16.25"
version = "0.16.24"
dependencies = [
"ahash",
"base64 0.23.1",
@@ -3922,7 +3912,7 @@ dependencies = [
[[package]]
name = "inbuxa"
version = "0.16.25"
version = "0.16.24"
dependencies = [
"common",
"coordinator",
@@ -3930,7 +3920,7 @@ dependencies = [
"directory",
"email",
"groupware",
"http 0.16.25",
"http 0.16.24",
"http_proto",
"imap",
"jmap",
@@ -3957,14 +3947,9 @@ name = "inbuxa-features"
version = "0.16.22"
dependencies = [
"ahash",
"aho-corasick",
"base64 0.23.1",
"flate2",
"jmap_proto",
"mail-builder 1.0.0",
"mail-parser",
"quick-xml 0.41.0",
"regex",
"registry",
"serde",
"serde_json",
@@ -3976,7 +3961,6 @@ dependencies = [
"types",
"utils",
"xxhash-rust",
"zip",
]
[[package]]
@@ -4219,7 +4203,7 @@ dependencies = [
[[package]]
name = "jmap"
version = "0.16.25"
version = "0.16.24"
dependencies = [
"async-stream",
"base64 0.23.1",
@@ -4268,14 +4252,14 @@ dependencies = [
[[package]]
name = "jmap-client"
version = "0.4.3"
version = "0.4.2"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "f5b5bc66252cc8e779ef1238f40ab93971d54ad00b5d7afb5c1d447a9647ed62"
checksum = "4deab22e057d24e32122f0fc6e2d667a124fdd6a0d8ef3ed4f8a89923c11084f"
dependencies = [
"ahash",
"async-stream",
"base64 0.22.1",
"chrono",
"encodify",
"futures-util",
"maybe-async",
"parking_lot",
@@ -4302,7 +4286,7 @@ dependencies = [
[[package]]
name = "jmap_proto"
version = "0.16.25"
version = "0.16.24"
dependencies = [
"ahash",
"calcard",
@@ -4512,9 +4496,9 @@ dependencies = [
[[package]]
name = "lazy_static"
version = "1.5.1"
version = "1.5.0"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "20870f649af7073d53e38067b2a84312175d56ea15217e1b15bc83506ec50afb"
checksum = "bbd2bcb4c963f2ddae06a2efc7e9f3591312473c50c6685e1f298068316e66fe"
dependencies = [
"spin 0.9.9",
]
@@ -4808,7 +4792,7 @@ dependencies = [
[[package]]
name = "managesieve"
version = "0.16.25"
version = "0.16.24"
dependencies = [
"common",
"compact_str",
@@ -4943,7 +4927,7 @@ checksum = "c797b9d6bb23aab2fc369c65f871be49214f5c759af65bde26ffaaa2b646b492"
[[package]]
name = "migration"
version = "0.16.25"
version = "0.16.24"
dependencies = [
"common",
"email",
@@ -5193,7 +5177,7 @@ dependencies = [
[[package]]
name = "nlp"
version = "0.16.25"
version = "0.16.24"
dependencies = [
"ahash",
"hashify",
@@ -5513,8 +5497,8 @@ dependencies = [
[[package]]
name = "opentelemetry"
version = "0.33.0"
source = "git+https://github.com/stalwartlabs/opentelemetry-rust#ae66e97b140f70e477ab710686aafce665cc2f8b"
version = "0.32.0"
source = "git+https://github.com/stalwartlabs/opentelemetry-rust#80a14a3b6846f62f85506d68d2600c948fccc9d2"
dependencies = [
"futures-core",
"futures-sink",
@@ -5526,8 +5510,8 @@ dependencies = [
[[package]]
name = "opentelemetry-http"
version = "0.33.0"
source = "git+https://github.com/stalwartlabs/opentelemetry-rust#ae66e97b140f70e477ab710686aafce665cc2f8b"
version = "0.32.0"
source = "git+https://github.com/stalwartlabs/opentelemetry-rust#80a14a3b6846f62f85506d68d2600c948fccc9d2"
dependencies = [
"async-trait",
"bytes",
@@ -5538,8 +5522,8 @@ dependencies = [
[[package]]
name = "opentelemetry-otlp"
version = "0.33.0"
source = "git+https://github.com/stalwartlabs/opentelemetry-rust#ae66e97b140f70e477ab710686aafce665cc2f8b"
version = "0.32.0"
source = "git+https://github.com/stalwartlabs/opentelemetry-rust#80a14a3b6846f62f85506d68d2600c948fccc9d2"
dependencies = [
"http 1.5.0",
"httpdate",
@@ -5557,8 +5541,8 @@ dependencies = [
[[package]]
name = "opentelemetry-proto"
version = "0.33.0"
source = "git+https://github.com/stalwartlabs/opentelemetry-rust#ae66e97b140f70e477ab710686aafce665cc2f8b"
version = "0.32.0"
source = "git+https://github.com/stalwartlabs/opentelemetry-rust#80a14a3b6846f62f85506d68d2600c948fccc9d2"
dependencies = [
"opentelemetry",
"opentelemetry_sdk",
@@ -5569,13 +5553,13 @@ dependencies = [
[[package]]
name = "opentelemetry-semantic-conventions"
version = "0.33.0"
source = "git+https://github.com/stalwartlabs/opentelemetry-rust#ae66e97b140f70e477ab710686aafce665cc2f8b"
version = "0.32.1"
source = "git+https://github.com/stalwartlabs/opentelemetry-rust#80a14a3b6846f62f85506d68d2600c948fccc9d2"
[[package]]
name = "opentelemetry_sdk"
version = "0.33.0"
source = "git+https://github.com/stalwartlabs/opentelemetry-rust#ae66e97b140f70e477ab710686aafce665cc2f8b"
version = "0.32.1"
source = "git+https://github.com/stalwartlabs/opentelemetry-rust#80a14a3b6846f62f85506d68d2600c948fccc9d2"
dependencies = [
"futures-channel",
"futures-executor",
@@ -6025,7 +6009,7 @@ dependencies = [
[[package]]
name = "pop3"
version = "0.16.25"
version = "0.16.24"
dependencies = [
"common",
"directory",
@@ -6395,9 +6379,9 @@ dependencies = [
[[package]]
name = "quinn-proto"
version = "0.11.19"
version = "0.11.18"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "0e750cca55fe4f0439a15d0bb529da9651e79993e8e72c61a899a36d462befbe"
checksum = "a9746dbde176634f4f2f1faf2404e30a31b2bc1e9cafb5329c95d8177a18c9fc"
dependencies = [
"aws-lc-rs",
"bytes",
@@ -6420,9 +6404,9 @@ dependencies = [
[[package]]
name = "quinn-udp"
version = "0.5.16"
version = "0.5.15"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "af66907df18639dcf4db56ca65490cabc4b27a97dbadd96f2926cca73298f016"
checksum = "35a133f956daabe89a61a685c2649f13d82d5aa4bd5d12d1277e1072a21c0694"
dependencies = [
"cfg_aliases",
"libc",
@@ -6845,7 +6829,7 @@ checksum = "d6f6ff9a378485b298a5286656da665ba74413d36db0979633275d2e708145d4"
[[package]]
name = "registry"
version = "0.16.25"
version = "0.16.24"
dependencies = [
"ahash",
"hashify",
@@ -7402,7 +7386,7 @@ dependencies = [
[[package]]
name = "scim"
version = "0.16.25"
version = "0.16.24"
dependencies = [
"ahash",
"base64 0.23.1",
@@ -7428,7 +7412,7 @@ dependencies = [
[[package]]
name = "scim-proto"
version = "0.16.25"
version = "0.16.24"
dependencies = [
"hashify",
"serde",
@@ -7682,9 +7666,9 @@ dependencies = [
[[package]]
name = "serde_with"
version = "3.24.0"
version = "3.23.0"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "df9adc193c780ef8f159aee8b61e2d5801aaa555e6eb0947fe45530ec506296f"
checksum = "935177bb8c0cd8ca1a4e6d1a2ac8988bea69cab4f9d3a31311e012ad27868ea4"
dependencies = [
"base64 0.23.1",
"bs58",
@@ -7703,9 +7687,9 @@ dependencies = [
[[package]]
name = "serde_with_macros"
version = "3.24.0"
version = "3.23.0"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "3e17bbc68e28663bbbb90df47e058aa7eda4fb445b89fe70457bb94fbccf6e49"
checksum = "1d607aa01a3cb0ad757d6fd216136910db3c97b102fe686585689615a02dbcdc"
dependencies = [
"darling 0.24.1",
"proc-macro2",
@@ -7763,7 +7747,7 @@ dependencies = [
[[package]]
name = "services"
version = "0.16.25"
version = "0.16.24"
dependencies = [
"aes-gcm 0.11.1",
"aho-corasick",
@@ -7772,13 +7756,11 @@ dependencies = [
"common",
"dns-update",
"email",
"futures",
"groupware",
"hkdf 0.13.0",
"inbuxa-features",
"jmap-tools",
"jmap_proto",
"mail-auth",
"mail-builder 1.0.0",
"mail-parser",
"memory-stats",
@@ -8078,7 +8060,7 @@ checksum = "f9395f0f0eee849a9b707b2f06bb92a6a422090e2123bb2ef8e87a0e61892a8e"
[[package]]
name = "smtp"
version = "0.16.25"
version = "0.16.24"
dependencies = [
"ahash",
"base64 0.23.1",
@@ -8117,9 +8099,9 @@ dependencies = [
[[package]]
name = "smtp-proto"
version = "0.2.5"
version = "0.2.4"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "142a5a642c6bd7ffd7e1b525ad6a6e9921ccef992dba4663974f8519209a00c1"
checksum = "707104487221ff447b5b796b5049e5c09cf52ff9fc1c8abba4695b89ac4b0f37"
dependencies = [
"memchr",
"rkyv",
@@ -8169,7 +8151,7 @@ dependencies = [
[[package]]
name = "spam-filter"
version = "0.16.25"
version = "0.16.24"
dependencies = [
"common",
"compact_str",
@@ -8289,7 +8271,7 @@ checksum = "a2eb9349b6444b326872e140eb1cf5e7c522154d69e7a0ffb0fb81c06b37543f"
[[package]]
name = "store"
version = "0.16.25"
version = "0.16.24"
dependencies = [
"ahash",
"arc-swap",
@@ -8549,7 +8531,7 @@ dependencies = [
[[package]]
name = "tests"
version = "0.16.25"
version = "0.16.24"
dependencies = [
"ahash",
"aws-lc-rs",
@@ -8571,7 +8553,7 @@ dependencies = [
"form_urlencoded",
"futures",
"groupware",
"http 0.16.25",
"http 0.16.24",
"http_proto",
"hyper",
"hyper-util",
@@ -8828,9 +8810,9 @@ dependencies = [
[[package]]
name = "tokio-rustls"
version = "0.26.6"
version = "0.26.5"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "c9cc2678c2cdd569ef8215e2afd7954ada2ae20b4fdd2c5fe6139a3b02d105db"
checksum = "b0c85f2c3ef0b1cd58b36682f4b17aaa995f0e5db534d85692b4903abce21f67"
dependencies = [
"rustls",
"tokio",
@@ -9164,7 +9146,7 @@ dependencies = [
[[package]]
name = "trc"
version = "0.16.25"
version = "0.16.24"
dependencies = [
"ahash",
"base64 0.23.1",
@@ -9273,7 +9255,7 @@ checksum = "b6f5e870be6c3b371b77fe0ee0bafb859fa4964b4404c27de1d380043c4dda20"
[[package]]
name = "types"
version = "0.16.25"
version = "0.16.24"
dependencies = [
"blake3",
"compact_str",
@@ -9442,7 +9424,7 @@ checksum = "b6c140620e7ffbb22c2dee59cafe6084a59b5ffc27a8859a5f0d494b5d52b6be"
[[package]]
name = "utils"
version = "0.16.25"
version = "0.16.24"
dependencies = [
"ahash",
"arcstr",
@@ -10127,9 +10109,9 @@ dependencies = [
[[package]]
name = "xxhash-rust"
version = "0.8.19"
version = "0.8.18"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "550a2b930b62486a393c52d5c3b84bff264b28aa437ed64694d31e93b1757af7"
checksum = "aee1b19627c7c60102ab80d3a9cbe18de90bfe03bfa6c3715447681f0e8c8af6"
[[package]]
name = "yasna"
@@ -10154,9 +10136,9 @@ dependencies = [
[[package]]
name = "yoke-derive"
version = "0.8.4"
version = "0.8.3"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "ec8ebde2db3681e8c9980cc27822030e68752690ddfa9473e739aeb4dbde6d71"
checksum = "33811428bee40dbceb6d545e95754741d17a6aef9a4849f0fd62e2ba4f412a78"
dependencies = [
"proc-macro2",
"quote",
+1 -1
View File
@@ -4,7 +4,7 @@
# *****************
# Base image for planner & builder
# *****************
FROM --platform=$BUILDPLATFORM rust:1.98.1-slim-trixie AS base
FROM --platform=$BUILDPLATFORM rust:slim-trixie AS base
ENV DEBIAN_FRONTEND="noninteractive" \
BINSTALL_DISABLE_TELEMETRY=true \
-4
View File
@@ -8,10 +8,6 @@
---
> [!NOTE]
> Development happens on [git.coffeylabs.org/inbuxa/inbuxa-server](https://git.coffeylabs.org/inbuxa/inbuxa-server); the copy on GitHub is a read-only mirror.
> Report issues at **[git.coffeylabs.org/inbuxa/inbuxa-server/issues](https://git.coffeylabs.org/inbuxa/inbuxa-server/issues)**, and join discussions at **[community.coffeylabs.org](https://community.coffeylabs.org)**.
**inbuxa** is a mail and collaboration server: JMAP, IMAP, POP3, SMTP,
CalDAV, CardDAV and WebDAV, in one Rust binary, with ihasmail as its web front
end. It is a fork of [Stalwart](https://github.com/stalwartlabs/stalwart).
+2 -2
View File
@@ -17,7 +17,7 @@ visible to everyone, including whoever would use it, before there is a fix.
Report it privately by email to:
**securityATcoffeylabsDOTorg**
**johnellisATlinuxDOTcom**
Include as much as you can of:
@@ -36,7 +36,7 @@ to Stalwart Labs with credit to you, and you'll be told that has happened.
This repository is the mail server. The web front ends have their own:
- [inbuxa-admin](https://git.coffeylabs.org/inbuxa/inbuxa-admin)
- [inbuxa-webmail](https://git.coffeylabs.org/inbuxa/inbuxa-webmail)
- [ihasmail-inbuxa](https://git.coffeylabs.org/inbuxa/ihasmail-inbuxa)
Upstream's own security documents are kept in `.github-upstream/` for
reference. They describe Stalwart Labs' process, not this project's.
+1 -1
View File
@@ -1,6 +1,6 @@
[package]
name = "common"
version = "0.16.25"
version = "0.16.24"
edition = "2024"
build = "build.rs"
+1 -58
View File
@@ -1,5 +1,5 @@
/*
* SPDX-FileCopyrightText: 2026 Coffey Labs LLC
* SPDX-FileCopyrightText: 2026 Coffey Labs
*
* SPDX-License-Identifier: AGPL-3.0-only
*/
@@ -445,63 +445,6 @@ impl Server {
}
}
/// MA-D0a: a message sent from an address that isn't the sender's own:
/// a group's or a shared mailbox's. The message itself only says
/// `From:` that address, so the audit log is where the person who sent
/// it is named. A locked account's delegate's send is AL-9's record, not
/// this one.
pub async fn audit_send_as(
&self,
token: &AccessToken,
submission_account_id: u32,
submission_id: u32,
address: &str,
) {
let Ok(Some(as_account_id)) = self.account_id_from_email(address, true).await else {
return;
};
if as_account_id == token.account_id()
|| token
.delegation(as_account_id)
.is_some_and(|delegation| delegation.kind.is_lock())
{
return;
}
let actor = self.audit_actor(token).await;
let tenant_id = self
.account(as_account_id)
.await
.ok()
.and_then(|account| account.id_tenant);
let details = if submission_account_id == as_account_id {
format!("Sent as {address}")
} else {
format!(
"Sent as {address}, from {}",
self.audit_account_name(submission_account_id).await
)
};
self.audit_note(Record {
at: ms(),
actor,
via: token.origin().cloned(),
remote_ip: None,
action: Action::Create,
target: Target {
kind: "EmailSubmission".into(),
id: Some(Id::from(submission_id).to_string()),
name: Some(address.to_string()),
account_id: Some(as_account_id),
tenant_id,
},
changes: vec![],
details: Some(details),
reason: None,
outcome: Outcome::success(),
})
.await;
}
/// AU-7: removes entries past the retention period.
pub async fn audit_purge(&self) -> trc::Result<usize> {
let settings = log::settings(self.store()).await?;
+4 -84
View File
@@ -36,32 +36,6 @@ use utils::map::bitmap::{Bitmap, BitmapItem};
use xxhash_rust::xxh3;
impl Server {
/// inbuxa: MA-C: whether people in `owner`'s tenant may share their mail
/// (the server's switch, narrowed by the tenant's).
pub async fn mail_sharing_allowed(&self, owner: u32) -> trc::Result<bool> {
let tenant_id = self.account(owner).await.ok().and_then(|account| account.id_tenant);
Ok(
inbuxa_features::security::sharing_policy::effective_for(self.store(), tenant_id)
.await
.caused_by(trc::location!())?
.mail_sharing,
)
}
/// inbuxa: MA-C: whether `owner`'s mail shares give access now. A locked
/// account's or shared mailbox's grants are an administrator's and always
/// do; anyone else's only while their tenant allows mail sharing.
pub async fn mail_shares_honored(&self, owner: u32) -> trc::Result<bool> {
if inbuxa_features::lock::get(self.store(), owner)
.await
.caused_by(trc::location!())?
.is_some()
{
return Ok(true);
}
self.mail_sharing_allowed(owner).await
}
async fn build_access_token(
&self,
account: Account,
@@ -72,22 +46,19 @@ impl Server {
// inbuxa: AL-2, AL-5: whether this account is locked, and which
// locked accounts are handed to it. The token is their cache: every
// change to a lock invalidates the tokens it touches.
let lock_kind = inbuxa_features::lock::get(self.store(), account_id)
let locked = inbuxa_features::lock::get(self.store(), account_id)
.await
.caused_by(trc::location!())?
.map(|lock| lock.kind);
let locked = lock_kind.is_some();
let shared_mailbox = lock_kind == Some(inbuxa_features::lock::Kind::SharedMailbox);
.is_some();
let now_secs = now();
let delegations: Box<[super::Delegation]> =
inbuxa_features::lock::delegated_to(self.store(), account_id)
.await
.caused_by(trc::location!())?
.into_iter()
.filter(|(_, delegate, _)| delegate.is_current(now_secs))
.map(|(locked_id, delegate, kind)| super::Delegation {
.filter(|(_, delegate)| delegate.is_current(now_secs))
.map(|(locked_id, delegate)| super::Delegation {
account_id: locked_id,
kind,
access: delegate.access,
send_as: delegate.send_as,
until: delegate.until,
@@ -126,9 +97,6 @@ impl Server {
.map(|m| m.id() as u32)
.collect::<TinyVec<[u32; 3]>>();
let mut access_to: Vec<AccessTo> = Vec::new();
// inbuxa: MA-C: whether an owner's mail shares are honored,
// looked up once per owner
let mut mail_shares_honored: Vec<(u32, bool)> = Vec::new();
for grant_account_id in [account_id].into_iter().chain(member_of.iter().copied()) {
for acl_item in self
.store()
@@ -149,27 +117,6 @@ impl Server {
.caused_by(trc::location!()));
}
// inbuxa: MA-C: a mail share from an account whose
// tenant (or server) has mail sharing off gives
// nothing while it is off. It stays stored, so it
// comes back when sharing does. A lock's and a
// shared mailbox's grants are an administrator's,
// and always count.
if collection == Collection::Mailbox {
let owner = acl_item.to_account_id;
let honored = match mail_shares_honored.iter().find(|(id, _)| *id == owner) {
Some((_, honored)) => *honored,
None => {
let honored = self.mail_shares_honored(owner).await?;
mail_shares_honored.push((owner, honored));
honored
}
};
if !honored {
continue;
}
}
let mut collections: Bitmap<Collection> = Bitmap::new();
if acl.contains(Acl::Read) {
collections.insert(collection);
@@ -300,7 +247,6 @@ impl Server {
.map(ConcurrencyLimiter::new),
obj_size: 0,
locked,
shared_mailbox,
delegations: delegations.clone(),
revision,
revision_account,
@@ -354,7 +300,6 @@ impl Server {
.map(ConcurrencyLimiter::new),
obj_size: 0,
locked,
shared_mailbox,
delegations: delegations.clone(),
revision,
revision_account,
@@ -608,16 +553,6 @@ impl AccessToken {
|| self.inner.access_to.iter().any(|a| a.account_id == account_id)
}
/// inbuxa: MA-D0: in the account only because it is a group this token
/// belongs to. Such a member has the group's mailbox but may not share it
/// on: who is in a group is an administrator's decision, and a share
/// would let anyone in.
pub fn is_group_member_only(&self, account_id: u32) -> bool {
self.inner.account_id != account_id
&& self.inner.member_of.contains(&account_id)
&& !self.has_permission(Permission::Impersonate)
}
pub fn is_account_id(&self, account_id: u32) -> bool {
self.inner.account_id == account_id
}
@@ -713,7 +648,6 @@ impl AccessToken {
credential_version: old_inner.credential_version,
obj_size: old_inner.obj_size,
locked: old_inner.locked,
shared_mailbox: old_inner.shared_mailbox,
delegations: old_inner.delegations.clone(),
};
@@ -904,18 +838,6 @@ impl AccessToken {
self.inner.locked
}
/// inbuxa: MA-S: the account is a shared mailbox (a lock of that kind).
pub fn is_shared_mailbox(&self) -> bool {
self.inner.shared_mailbox
}
/// inbuxa: MA-S: this account's delegation into `account_id` is to a
/// shared mailbox, not a locked account.
pub fn delegated_shared_mailbox(&self, account_id: u32) -> bool {
self.delegation(account_id)
.is_some_and(|d| d.kind == inbuxa_features::lock::Kind::SharedMailbox)
}
/// inbuxa: AL-5: this account's delegation into a locked account, if it
/// has one that hasn't ended.
/// inbuxa: AL-6, AL-7: a delegate at organize or full, who may add to
@@ -996,7 +918,6 @@ impl AccessToken {
credential_version: Default::default(),
obj_size: Default::default(),
locked: false,
shared_mailbox: false,
delegations: Default::default(),
}),
}
@@ -1057,7 +978,6 @@ impl AccessTokenInner {
credential_version: Default::default(),
obj_size: Default::default(),
locked: false,
shared_mailbox: false,
delegations: Default::default(),
}
}
-4
View File
@@ -152,8 +152,6 @@ pub struct AccessTokenInner {
pub(crate) obj_size: u64,
// inbuxa: AL-2: the account is locked; it may not authenticate
pub(crate) locked: bool,
// inbuxa: MA-S: the lock is a shared mailbox
pub(crate) shared_mailbox: bool,
// inbuxa: AL-5: locked accounts handed to this one
pub(crate) delegations: Box<[Delegation]>,
}
@@ -167,8 +165,6 @@ pub struct Delegation {
pub send_as: bool,
/// Seconds since the epoch.
pub until: Option<u64>,
/// MA-S: a locked account, or a shared mailbox.
pub kind: inbuxa_features::lock::Kind,
}
#[derive(Debug, Default, Hash, Clone)]
-42
View File
@@ -111,9 +111,6 @@ impl Server {
Permission::SysLegalHoldCreate,
Permission::SysLegalHoldUpdate,
Permission::SysLegalHoldExport,
// inbuxa: DL-20: the lists and the check are the server's
Permission::SysDeliverabilityUpdate,
Permission::SysDeliverabilityCheck,
] {
permissions.disabled.set(permission as usize);
}
@@ -168,14 +165,6 @@ impl AccessToken {
mut requested_permissions: Permissions,
) -> Result<(), Vec<Permission>> {
requested_permissions.difference(self.permissions_bits());
// inbuxa: journaling, JR-18: whoever sets up journals may give
// others (or, through a role, themselves) the reading of them,
// which administrators don't hold by default; the role change is
// in the audit log
if self.has_permission(Permission::SysJournalUpdate) {
requested_permissions.clear(Permission::SysJournalSearch as usize);
requested_permissions.clear(Permission::SysJournalExport as usize);
}
if requested_permissions.is_empty() {
Ok(())
} else {
@@ -307,37 +296,6 @@ impl Default for DefaultPermissions {
default.superuser.push(permission);
default.tenant.push(permission);
}
// inbuxa: deliverability spec, DL-20: a tenant administrator
// reads its own domains' findings; the lists and the check
// itself are the server's
Permission::SysDeliverabilityGet => {
default.superuser.push(permission);
default.tenant.push(permission);
}
Permission::SysDeliverabilityUpdate | Permission::SysDeliverabilityCheck => {
default.superuser.push(permission);
}
// inbuxa: DLP and mail flow rules, and held mail, are the
// server's: never a tenant's (dlp-and-mail-flow-rules spec,
// settled answer 3)
Permission::SysMailRuleGet
| Permission::SysMailRuleUpdate
| Permission::SysDlpPolicyGet
| Permission::SysDlpPolicyUpdate
| Permission::SysDlpReviewGet
| Permission::SysDlpReviewUpdate
// inbuxa: every security check is server-wide (security
// to-do list spec)
| Permission::SysSecurityAccept => {
default.superuser.push(permission);
}
// inbuxa: journals are the server's; administrators set them
// up but read what's journaled only if granted it
// (journaling spec, JR-18, settled answer 5)
Permission::SysJournalGet | Permission::SysJournalUpdate => {
default.superuser.push(permission);
}
Permission::SysJournalSearch | Permission::SysJournalExport => {}
// inbuxa: AL-12: tenant administrators lock and delegate
// within their tenant
Permission::SysAccountLockGet
-34
View File
@@ -72,10 +72,6 @@ pub struct Http {
pub cors_origins: Vec<hyper::header::HeaderValue>,
pub use_forwarded: bool,
pub redirect_root: Option<String>,
/// inbuxa: HTTP Basic accepted on every endpoint, not only DAV (contract
/// C-23). True in bootstrap and recovery mode, or with
/// `INBUXA_HTTP_BASIC_AUTH=all`.
pub basic_auth_everywhere: bool,
}
#[derive(Clone)]
@@ -457,35 +453,6 @@ impl Http {
.collect()
};
// inbuxa: outside DAV, HTTP sign-in is a token unless the operator
// says otherwise (contract C-23). The integration suites sign in with
// passwords over JMAP and the API, so test builds accept Basic
// everywhere.
#[cfg(feature = "test_mode")]
let basic_auth_everywhere = true;
#[cfg(not(feature = "test_mode"))]
let basic_auth_everywhere = bp.registry.is_recovery_mode()
|| bp.registry.is_bootstrap_mode()
|| match types::branding::env_var("HTTP_BASIC_AUTH") {
Ok(value) if value.trim().eq_ignore_ascii_case("all") => true,
Ok(value)
if value.trim().is_empty() || value.trim().eq_ignore_ascii_case("dav") =>
{
false
}
Ok(value) => {
bp.build_warning(
ObjectType::Http.singleton(),
format!(
"INBUXA_HTTP_BASIC_AUTH is {value:?}; expected \"dav\" or \"all\". Basic authentication stays on DAV only."
),
);
false
}
Err(_) => false,
};
if use_permissive_cors {
http_headers.push((
hyper::header::ACCESS_CONTROL_ALLOW_ORIGIN,
@@ -545,7 +512,6 @@ impl Http {
cors_origins,
use_forwarded: http.use_x_forwarded,
redirect_root: http.redirect_root,
basic_auth_everywhere,
}
}
}
+1 -1
View File
@@ -1,5 +1,5 @@
/*
* SPDX-FileCopyrightText: 2026 Coffey Labs LLC
* SPDX-FileCopyrightText: 2026 Coffey Labs
*
* SPDX-License-Identifier: AGPL-3.0-only
*/
+1 -1
View File
@@ -1,5 +1,5 @@
/*
* SPDX-FileCopyrightText: 2026 Coffey Labs LLC
* SPDX-FileCopyrightText: 2026 Coffey Labs
*
* SPDX-License-Identifier: AGPL-3.0-only
*/
+1 -1
View File
@@ -1,5 +1,5 @@
/*
* SPDX-FileCopyrightText: 2026 Coffey Labs LLC
* SPDX-FileCopyrightText: 2026 Coffey Labs
*
* SPDX-License-Identifier: AGPL-3.0-only
*/
-2
View File
@@ -88,8 +88,6 @@ pub enum BroadcastEvent {
QueueRefresh,
// inbuxa: AL-3: end an account's open sessions on every node
EndSessions(u32),
// inbuxa: deliverability spec, DL-15: every node checks itself now
DeliverabilityCheck,
}
#[derive(Debug, Clone, Copy)]
+1 -1
View File
@@ -225,7 +225,7 @@ pub struct Caches {
pub dns_ipv6: CacheWithTtl<Box<str>, RecordSet<Ipv6Addr>>,
pub dns_tlsa: CacheWithTtl<Box<str>, Arc<Tlsa>>,
pub dns_mta_sts: CacheWithTtl<Box<str>, Arc<Policy>>,
pub dns_rbl: CacheWithTtl<Box<str>, Option<Arc<[IpResolver]>>>,
pub dns_rbl: CacheWithTtl<Box<str>, Option<Arc<IpResolver>>>,
pub negative_cache_ttl: Duration,
}
+2 -14
View File
@@ -227,22 +227,10 @@ impl WebApplicationManager {
let cached = if force_refresh {
None
} else {
match server
server
.blob_store()
.get_blob(self.blob_key.as_slice(), 0..usize::MAX)
.await
{
Ok(cached) => cached,
Err(err) => {
trc::event!(
Resource(trc::ResourceEvent::Error),
Reason = err,
Url = self.url.clone(),
Details = "Failed to read cached application bundle, downloading it again"
);
None
}
}
.await?
};
let is_cached = cached.is_some();
let bundle = match cached {
+8 -39
View File
@@ -1,5 +1,5 @@
/*
* SPDX-FileCopyrightText: 2026 Coffey Labs LLC
* SPDX-FileCopyrightText: 2026 Coffey Labs
*
* SPDX-License-Identifier: AGPL-3.0-only
*/
@@ -65,14 +65,6 @@ const OFFICER: &[Permission] = &[
Permission::SysLegalHoldUpdate,
Permission::SysLegalHoldExport,
Permission::SysAccountLockGet,
// dlp-and-mail-flow-rules spec, §2.8: see DLP rules, review held mail
Permission::SysDlpPolicyGet,
Permission::SysDlpReviewGet,
Permission::SysDlpReviewUpdate,
// journaling spec, JR-18: see journals, search and export them
Permission::SysJournalGet,
Permission::SysJournalSearch,
Permission::SysJournalExport,
];
/// What a tenant's officer holds besides [`READS`].
@@ -121,11 +113,6 @@ fn created_key(tenant: Option<Id>) -> ValueClass {
})
}
/// The server-level Compliance Officer role the server made, if it has.
pub async fn server_role(data: &Store) -> trc::Result<Option<Id>> {
recorded(data, None).await
}
async fn recorded(data: &Store, tenant: Option<Id>) -> trc::Result<Option<Id>> {
Ok(data
.get_value::<u64>(ValueKey::from(created_key(tenant)))
@@ -185,19 +172,13 @@ pub async fn ensure_compliance_roles(registry: &RegistryStore, data: &Store) ->
/// A new tenant gets its Compliance Officer role.
pub async fn tenant_created(registry: &RegistryStore, data: &Store, tenant: Id) -> trc::Result<()> {
create_once(registry, data, Some(tenant), tenant_role(tenant))
.await
.map(|_| ())
create_once(registry, data, Some(tenant), tenant_role(tenant)).await.map(|_| ())
}
/// Before a tenant is deleted: removes its Compliance Officer role if nobody
/// holds it, so the role doesn't block the delete. Returns whether it did,
/// so a delete refused for another reason can put it back.
pub async fn tenant_deleting(
registry: &RegistryStore,
data: &Store,
tenant: Id,
) -> trc::Result<bool> {
pub async fn tenant_deleting(registry: &RegistryStore, data: &Store, tenant: Id) -> trc::Result<bool> {
let Some(role) = recorded(data, Some(tenant)).await? else {
return Ok(false);
};
@@ -239,9 +220,7 @@ mod tests {
// Beyond what any user holds for their own account
for permission in all.into_iter().filter(|p| !user.contains(p)) {
let name = permission.as_str();
// Placing holds and reviewing held mail are the officer's
// job, not settings (settled answers 2 and 4)
let holds = name.starts_with("sysLegalHold") || name.starts_with("sysDlpReview");
let holds = name.starts_with("sysLegalHold");
assert!(
!(name.ends_with("Update") && !holds)
&& !(name.ends_with("Create") && !holds)
@@ -270,11 +249,7 @@ mod tests {
assert!(officer.contains(&hold));
assert!(!tenant.contains(&hold));
}
for both in [
Permission::SysComplianceGet,
Permission::SysAuditGet,
Permission::SysAccountGet,
] {
for both in [Permission::SysComplianceGet, Permission::SysAuditGet, Permission::SysAccountGet] {
assert!(officer.contains(&both) && tenant.contains(&both));
}
assert!(!officer.contains(&Permission::SysAuditSettingsUpdate));
@@ -282,15 +257,9 @@ mod tests {
#[test]
fn records_are_per_place() {
let ValueClass::Any(server) = created_key(None) else {
panic!()
};
let ValueClass::Any(a) = created_key(Some(Id::from(1u64))) else {
panic!()
};
let ValueClass::Any(b) = created_key(Some(Id::from(2u64))) else {
panic!()
};
let ValueClass::Any(server) = created_key(None) else { panic!() };
let ValueClass::Any(a) = created_key(Some(Id::from(1u64))) else { panic!() };
let ValueClass::Any(b) = created_key(Some(Id::from(2u64))) else { panic!() };
assert_eq!(server.key, b"Pc");
assert_ne!(a.key, b.key);
assert!(a.key.starts_with(b"Pc"));
+3 -3
View File
@@ -1,5 +1,5 @@
/*
* SPDX-FileCopyrightText: 2026 Coffey Labs LLC
* SPDX-FileCopyrightText: 2026 Coffey Labs
*
* SPDX-License-Identifier: AGPL-3.0-only
*/
@@ -14,7 +14,7 @@
//! application names another;
//! - INBUXA Admin hosted elsewhere, as `inbuxa-admin`, when `INBUXA_ADMIN_URL`
//! is set;
//! - inbuxa-webmail, as the confidential client `ihasmail-inbuxa`, when
//! - ihasmail-inbuxa, as the confidential client `ihasmail-inbuxa`, when
//! `INBUXA_WEBMAIL_URL` and `INBUXA_WEBMAIL_CLIENT_SECRET` are set.
//!
//! inbuxa: the environment variables stand in for `x:FrontEnds` (C-4) until
@@ -22,7 +22,7 @@
//! it instead.
//!
//! A missing client is created. An existing one gains any redirect URI it
//! lacks and, for inbuxa-webmail, the configured secret; nothing an operator
//! lacks and, for ihasmail-inbuxa, the configured secret; nothing an operator
//! added is removed.
use directory::core::secret::{hash_secret, verify_secret_hash};
@@ -1,5 +1,5 @@
/*
* SPDX-FileCopyrightText: 2026 Coffey Labs LLC
* SPDX-FileCopyrightText: 2026 Coffey Labs
*
* SPDX-License-Identifier: AGPL-3.0-only
*/
@@ -31,9 +31,7 @@ use types::id::Id;
/// Granted to the default administrator roles: "Explain this"
/// (ai-explain spec, EX-4: superuser by default), the audit log, account
/// locks and legal holds (audit-hold-lock spec, AU-9, AL-12, LH-13), and
/// the data inventory (personal-data catalog spec), accepting security
/// to-do items (security to-do list spec), and the deliverability check
/// (deliverability spec).
/// the data inventory (personal-data catalog spec).
const ADMIN_GRANTS: &[Permission] = &[
Permission::SysAiExplain,
Permission::SysAuditGet,
@@ -48,37 +46,11 @@ const ADMIN_GRANTS: &[Permission] = &[
Permission::SysLegalHoldUpdate,
Permission::SysLegalHoldExport,
Permission::SysComplianceGet,
Permission::SysMailRuleGet,
Permission::SysMailRuleUpdate,
Permission::SysDlpPolicyGet,
Permission::SysDlpPolicyUpdate,
Permission::SysDlpReviewGet,
Permission::SysDlpReviewUpdate,
Permission::SysJournalGet,
Permission::SysJournalUpdate,
Permission::SysSecurityAccept,
Permission::SysDeliverabilityGet,
Permission::SysDeliverabilityUpdate,
Permission::SysDeliverabilityCheck,
];
/// Granted to the server-level Compliance Officer role once it exists:
/// seeing DLP rules and reviewing held mail (dlp-and-mail-flow-rules spec,
/// §2.8, settled answer 4). A new install's role has them from the start.
const OFFICER_GRANTS: &[Permission] = &[
Permission::SysDlpPolicyGet,
Permission::SysDlpReviewGet,
Permission::SysDlpReviewUpdate,
// journaling spec, JR-18: see journals, search and export them
Permission::SysJournalGet,
Permission::SysJournalSearch,
Permission::SysJournalExport,
];
/// Granted to the default tenant administrator roles: reading and exporting
/// the tenant's audit log (AU-9), locking and delegating its accounts
/// (AL-12), the tenant's slice of the data inventory, and its own domains'
/// deliverability findings (DL-20).
/// (AL-12), and the tenant's slice of the data inventory.
const TENANT_GRANTS: &[Permission] = &[
Permission::SysAuditGet,
Permission::SysAuditExport,
@@ -87,23 +59,19 @@ const TENANT_GRANTS: &[Permission] = &[
Permission::SysAccountLockUpdate,
Permission::SysAccountLockDestroy,
Permission::SysComplianceGet,
Permission::SysDeliverabilityGet,
];
#[derive(Clone, Copy, PartialEq, Eq)]
enum Audience {
Admin,
Tenant,
Officer,
}
fn granted_key(permission: Permission, audience: Audience) -> ValueClass {
let mut key = b"Pg".to_vec();
// Admin grants keep the key they were first recorded under
match audience {
Audience::Admin => {}
Audience::Tenant => key.extend_from_slice(b"tenant:"),
Audience::Officer => key.extend_from_slice(b"officer:"),
if audience == Audience::Tenant {
key.extend_from_slice(b"tenant:");
}
key.extend_from_slice(permission.as_str().as_bytes());
ValueClass::Any(AnyClass {
@@ -114,8 +82,7 @@ fn granted_key(permission: Permission, audience: Audience) -> ValueClass {
pub(crate) async fn grant_new_admin_permissions(bp: &mut Bootstrap) -> trc::Result<()> {
grant(bp, Audience::Admin, ADMIN_GRANTS).await?;
grant(bp, Audience::Tenant, TENANT_GRANTS).await?;
grant(bp, Audience::Officer, OFFICER_GRANTS).await
grant(bp, Audience::Tenant, TENANT_GRANTS).await
}
async fn grant(bp: &mut Bootstrap, audience: Audience, grants: &[Permission]) -> trc::Result<()> {
@@ -134,17 +101,10 @@ async fn grant(bp: &mut Bootstrap, audience: Audience, grants: &[Permission]) ->
if pending.is_empty() {
return Ok(());
}
// The officer role is the one the server made, if it has made it yet: a
// new install makes it after this, with the permissions already in it
let admin_roles: Vec<Id> = if audience == Audience::Officer {
super::compliance_roles::server_role(&bp.data_store)
.await?
.into_iter()
.collect()
} else {
// An administrator's default roles include the plain User role, which
// every user also holds; only roles that are the audience's alone get it
bp.registry
let admin_roles: Vec<Id> = bp
.registry
.object::<Authentication>(Id::singleton())
.await?
.map(|auth| {
@@ -158,7 +118,7 @@ async fn grant(bp: &mut Bootstrap, audience: Audience, grants: &[Permission]) ->
]
.concat(),
),
Audience::Tenant | Audience::Officer => (
Audience::Tenant => (
auth.default_tenant_role_ids.as_slice(),
[
auth.default_user_role_ids.as_slice(),
@@ -173,8 +133,7 @@ async fn grant(bp: &mut Bootstrap, audience: Audience, grants: &[Permission]) ->
.copied()
.collect()
})
.unwrap_or_default()
};
.unwrap_or_default();
// Fetched by id: the registry's listing doesn't reach stored roles
for role_id in admin_roles {
let Some(stored) = bp
+1 -1
View File
@@ -1,5 +1,5 @@
/*
* SPDX-FileCopyrightText: 2026 Coffey Labs LLC
* SPDX-FileCopyrightText: 2026 Coffey Labs
*
* SPDX-License-Identifier: AGPL-3.0-only
*/
+1 -1
View File
@@ -1,6 +1,6 @@
/*
* SPDX-FileCopyrightText: 2020 Stalwart Labs LLC <[email protected]>
* SPDX-FileCopyrightText: 2026 Coffey Labs LLC
* SPDX-FileCopyrightText: 2026 Coffey Labs
*
* SPDX-License-Identifier: AGPL-3.0-only OR LicenseRef-SEL
*
@@ -11,7 +11,7 @@ use quick_xml::Reader;
use quick_xml::XmlVersion;
use quick_xml::events::Event;
use registry::schema::{enums::ServiceProtocol, structs::Service};
use std::{borrow::Cow, fmt::Write};
use std::fmt::Write;
use utils::map::vec_map::VecMap;
impl Server {
@@ -20,78 +20,32 @@ impl Server {
body: Option<Vec<u8>>,
) -> trc::Result<Resource<Vec<u8>>> {
// Obtain parameters
let request =
parse_autodiscover_request(body.as_deref().unwrap_or_default()).map_err(|err| {
let emailaddress = parse_autodiscover_request(body.as_deref().unwrap_or_default())
.map_err(|err| {
trc::ResourceEvent::BadParameters
.into_err()
.details("Failed to parse autodiscover request")
.ctx(trc::Key::Reason, err)
})?;
// inbuxa: legacy-protocols LP-7, LP-14a
let legacy_off = match request.email.rsplit_once('@') {
let legacy_off = match emailaddress.rsplit_once('@') {
Some((_, domain)) => self.legacy_off_for(domain).await?,
None => self.legacy_off_for("").await?,
};
let response = match request.response_schema {
ResponseSchema::Outlook => build_autodiscover_response(
&request.email,
Ok(Resource::new(
"application/xml; charset=utf-8",
build_autodiscover_response(
&emailaddress,
&self.core.network.server_name,
&self.core.network.info.services,
|protocol| legacy_off.service(protocol),
)
.into_bytes(),
ResponseSchema::Unsupported => PROVIDER_NOT_AVAILABLE_RESPONSE.as_bytes().to_vec(),
};
Ok(Resource::new("application/xml; charset=utf-8", response))
))
}
}
const OUTLOOK_RESPONSE_SCHEMA: &str =
"http://schemas.microsoft.com/exchange/autodiscover/outlook/responseschema/2006a";
const PROVIDER_NOT_AVAILABLE_RESPONSE: &str = concat!(
"<?xml version=\"1.0\" encoding=\"UTF-8\"?>\n",
"<Autodiscover xmlns=\"http://schemas.microsoft.com/exchange/autodiscover/responseschema/2006\">\n",
"\t<Response>\n",
"\t\t<Error>\n",
"\t\t\t<ErrorCode>601</ErrorCode>\n",
"\t\t\t<Message>Provider is not available</Message>\n",
"\t\t\t<DebugData />\n",
"\t\t</Error>\n",
"\t</Response>\n",
"</Autodiscover>\n",
);
#[derive(Debug, Clone, Copy, PartialEq, Eq)]
enum ResponseSchema {
Outlook,
Unsupported,
}
impl ResponseSchema {
fn parse(value: &str) -> Self {
if value.trim().eq_ignore_ascii_case(OUTLOOK_RESPONSE_SCHEMA) {
ResponseSchema::Outlook
} else {
ResponseSchema::Unsupported
}
}
}
#[derive(Debug, PartialEq, Eq)]
struct AutodiscoverRequest {
email: String,
response_schema: ResponseSchema,
}
#[derive(Clone, Copy)]
enum RequestField {
EmailAddress,
ResponseSchema,
}
fn build_autodiscover_response(
emailaddress: &str,
default_host: &str,
@@ -170,7 +124,7 @@ fn build_autodiscover_response(
config
}
fn parse_autodiscover_request(bytes: &[u8]) -> Result<AutodiscoverRequest, String> {
fn parse_autodiscover_request(bytes: &[u8]) -> Result<String, String> {
if bytes.is_empty() {
return Err("Empty request body".to_string());
}
@@ -178,9 +132,8 @@ fn parse_autodiscover_request(bytes: &[u8]) -> Result<AutodiscoverRequest, Strin
let mut reader = Reader::from_reader(bytes);
reader.config_mut().trim_text(true);
let mut buf = Vec::with_capacity(128);
let mut value_buf = Vec::with_capacity(128);
'outer: for tag_name in ["Autodiscover", "Request"] {
'outer: for tag_name in ["Autodiscover", "Request", "EMailAddress"] {
loop {
match reader.read_event_into(&mut buf) {
Ok(Event::Start(e)) => {
@@ -190,6 +143,30 @@ fn parse_autodiscover_request(bytes: &[u8]) -> Result<AutodiscoverRequest, Strin
.eq_ignore_ascii_case(found_tag_name.as_ref())
{
continue 'outer;
} else if tag_name == "EMailAddress" {
// Skip unsupported tags under Request, such as AcceptableResponseSchema
let mut tag_count = 0;
loop {
match reader.read_event_into(&mut buf) {
Ok(Event::End(_)) => {
if tag_count == 0 {
break;
} else {
tag_count -= 1;
}
}
Ok(Event::Start(_)) => {
tag_count += 1;
}
Ok(Event::Eof) => {
return Err(format!(
"Expected value, found unexpected EOF at position {}.",
reader.buffer_position()
));
}
_ => (),
}
}
} else {
return Err(format!(
"Expected tag {}, found unexpected tag {} at position {}.",
@@ -218,171 +195,37 @@ fn parse_autodiscover_request(bytes: &[u8]) -> Result<AutodiscoverRequest, Strin
}
}
let mut email = None;
let mut response_schema = ResponseSchema::Outlook;
loop {
match reader.read_event_into(&mut buf) {
Ok(Event::Start(e)) => {
let local_name = e.local_name();
let field = hashify::tiny_map_ignore_case!(local_name.as_ref(),
b"EMailAddress" => RequestField::EmailAddress,
b"AcceptableResponseSchema" => RequestField::ResponseSchema,
);
let value = match reader.read_event_into(&mut value_buf) {
Ok(Event::End(_)) => None,
Ok(event) => {
let value = match event {
Event::Text(text) => text
.xml_content(XmlVersion::Implicit1_0)
.ok()
.map(Cow::into_owned),
_ => None,
};
reader
.read_to_end_into(e.name(), &mut value_buf)
.map_err(|err| {
format!("Error at position {}: {:?}", reader.buffer_position(), err)
})?;
value
}
Err(err) => {
return Err(format!(
"Error at position {}: {:?}",
reader.buffer_position(),
err
));
}
};
match (field, value) {
(Some(RequestField::EmailAddress), Some(value)) => {
email = Some(value);
}
(Some(RequestField::ResponseSchema), Some(value)) => {
response_schema = ResponseSchema::parse(&value);
}
_ => (),
}
}
Ok(Event::End(_) | Event::Eof) => break,
Ok(_) => (),
Err(e) => {
return Err(format!(
"Error at position {}: {:?}",
reader.buffer_position(),
e
));
}
}
if let Ok(Event::Text(text)) = reader.read_event_into(&mut buf)
&& let Ok(text) = text.xml_content(XmlVersion::Implicit1_0)
&& text.contains('@')
{
return Ok(text.trim().to_lowercase());
}
match email {
Some(email) if email.contains('@') => Ok(AutodiscoverRequest {
email: email.trim().to_lowercase(),
response_schema,
}),
_ => Err(format!(
Err(format!(
"Expected email address, found unexpected value at position {}.",
reader.buffer_position()
)),
}
))
}
#[cfg(test)]
mod tests {
use super::{AutodiscoverRequest, ResponseSchema, parse_autodiscover_request};
#[test]
fn parse_autodiscover() {
const OUTLOOK: &str =
"http://schemas.microsoft.com/exchange/autodiscover/outlook/responseschema/2006a";
const MOBILESYNC: &str =
"http://schemas.microsoft.com/exchange/autodiscover/mobilesync/responseschema/2006";
for (request, expected) in [
(
format!(
r#"<?xml version="1.0" encoding="utf-8"?>
let r = r#"<?xml version="1.0" encoding="utf-8"?>
<Autodiscover xmlns="http://schemas.microsoft.com/exchange/autodiscover/outlook/requestschema/2006">
<Request>
<EMailAddress>[email protected]</EMailAddress>
<AcceptableResponseSchema>{OUTLOOK}</AcceptableResponseSchema>
</Request>
</Autodiscover>"#
),
ResponseSchema::Outlook,
),
(
format!(
r#"<Autodiscover xmlns="http://schemas.microsoft.com/exchange/autodiscover/outlook/requestschema/2006">
<Request>
<AcceptableResponseSchema>{OUTLOOK}</AcceptableResponseSchema>
<EMailAddress>[email protected]</EMailAddress>
<AcceptableResponseSchema>http://schemas.microsoft.com/exchange/autodiscover/outlook/responseschema/2006a</AcceptableResponseSchema>
</Request>
</Autodiscover>"#
),
ResponseSchema::Outlook,
),
(
r#"<Autodiscover>
<Request>
<EMailAddress>[email protected]</EMailAddress>
</Request>
</Autodiscover>"#
.to_string(),
ResponseSchema::Outlook,
),
(
format!(
r#"<?xml version="1.0" encoding="utf-8"?>
<Autodiscover xmlns="http://schemas.microsoft.com/exchange/autodiscover/mobilesync/requestschema/2006">
<Request>
<EMailAddress>[email protected]</EMailAddress>
<AcceptableResponseSchema>{MOBILESYNC}</AcceptableResponseSchema>
</Request>
</Autodiscover>"#
),
ResponseSchema::Unsupported,
),
(
format!(
r#"<Autodiscover>
<Request>
<LegacyDN>/o=Example/ou=Users/cn=email</LegacyDN>
<Unknown><Nested>value</Nested><Empty/></Unknown>
<AcceptableResponseSchema>{MOBILESYNC}</AcceptableResponseSchema>
<EMailAddress>[email protected]</EMailAddress>
</Request>
</Autodiscover>"#
),
ResponseSchema::Unsupported,
),
] {
assert_eq!(
parse_autodiscover_request(request.as_bytes()).expect("valid request"),
AutodiscoverRequest {
email: "[email protected]".to_string(),
response_schema: expected,
},
"{request}"
);
}
</Autodiscover>"#;
for request in [
"",
"<Autodiscover><Request></Request></Autodiscover>",
"<Autodiscover><Request><EMailAddress>no-domain</EMailAddress></Request></Autodiscover>",
"<Autodiscover><Request><EMailAddress>[email protected]</Request></Autodiscover>",
"<Request><EMailAddress>[email protected]</EMailAddress></Request>",
] {
assert!(
parse_autodiscover_request(request.as_bytes()).is_err(),
"{request}"
assert_eq!(
super::parse_autodiscover_request(r.as_bytes()).unwrap(),
"[email protected]"
);
}
}
#[test]
fn autodiscover_encryption() {
+1 -1
View File
@@ -1,5 +1,5 @@
/*
* SPDX-FileCopyrightText: 2026 Coffey Labs LLC
* SPDX-FileCopyrightText: 2026 Coffey Labs
*
* SPDX-License-Identifier: AGPL-3.0-only
*/
+1 -1
View File
@@ -1,5 +1,5 @@
/*
* SPDX-FileCopyrightText: 2026 Coffey Labs LLC
* SPDX-FileCopyrightText: 2026 Coffey Labs
*
* SPDX-License-Identifier: AGPL-3.0-only
*/
+1 -1
View File
@@ -1,5 +1,5 @@
/*
* SPDX-FileCopyrightText: 2026 Coffey Labs LLC
* SPDX-FileCopyrightText: 2026 Coffey Labs
*
* SPDX-License-Identifier: AGPL-3.0-only
*/
+1 -1
View File
@@ -1,5 +1,5 @@
/*
* SPDX-FileCopyrightText: 2026 Coffey Labs LLC
* SPDX-FileCopyrightText: 2026 Coffey Labs
*
* SPDX-License-Identifier: AGPL-3.0-only
*/
+1 -1
View File
@@ -1,5 +1,5 @@
/*
* SPDX-FileCopyrightText: 2026 Coffey Labs LLC
* SPDX-FileCopyrightText: 2026 Coffey Labs
*
* SPDX-License-Identifier: AGPL-3.0-only
*/
+1 -1
View File
@@ -1,5 +1,5 @@
/*
* SPDX-FileCopyrightText: 2026 Coffey Labs LLC
* SPDX-FileCopyrightText: 2026 Coffey Labs
*
* SPDX-License-Identifier: AGPL-3.0-only
*/
+1 -1
View File
@@ -1,5 +1,5 @@
/*
* SPDX-FileCopyrightText: 2026 Coffey Labs LLC
* SPDX-FileCopyrightText: 2026 Coffey Labs
*
* SPDX-License-Identifier: AGPL-3.0-only
*/
+4 -62
View File
@@ -1,5 +1,5 @@
/*
* SPDX-FileCopyrightText: 2026 Coffey Labs LLC
* SPDX-FileCopyrightText: 2026 Coffey Labs
*
* SPDX-License-Identifier: AGPL-3.0-only
*/
@@ -104,31 +104,14 @@ impl StoredMetric {
pub fn timestamp(&self) -> u64 {
SnowflakeIdGenerator::to_timestamp(self.id)
}
/// The node that wrote the sample. Histogram totals are per node, so a
/// reader diffs them per node.
pub fn node_id(&self) -> u64 {
SnowflakeIdGenerator::to_node_id(self.id)
}
}
/// What the node wrote last, so counters and histograms are written as
/// changes (MON-4). Per process: a restart counts from the start.
static LAST: Mutex<Option<AHashMap<MetricType, (u64, u64)>>> = Mutex::new(None);
/// Gauges that count the whole cluster's data, not this node's. Only the node
/// that computes them (the metrics-calculation role) has a true reading; on
/// the others the queue gauge only moves with local queue events and drifts
/// below zero, and the account and domain counts stay at 0.
const CLUSTER_GAUGES: [MetricType; 3] = [
MetricType::QueueCount,
MetricType::UserCount,
MetricType::DomainCount,
];
/// One tick's samples (MON-4 to MON-6). `calculates` is whether this node
/// computes the cluster-wide gauges; a node that doesn't leaves them out.
pub fn sample(calculates: bool) -> Vec<Metric> {
/// One tick's samples (MON-4 to MON-6).
pub fn sample() -> Vec<Metric> {
let mut last_guard = LAST.lock().unwrap();
let last = last_guard.get_or_insert_with(AHashMap::new);
let mut samples = Vec::new();
@@ -151,9 +134,6 @@ pub fn sample(calculates: bool) -> Vec<Metric> {
// Gauges: the reading, always (MON-5)
for gauge in Collector::collect_gauges() {
if !calculates && CLUSTER_GAUGES.contains(&gauge.id()) {
continue;
}
samples.push(Metric::Gauge(MetricCount {
count: gauge.get(),
metric: gauge.id(),
@@ -195,7 +175,7 @@ impl Server {
if store.is_none() {
return;
}
let samples = sample(self.core.network.roles.metrics_calculate);
let samples = sample();
let count = samples.len();
let started = std::time::Instant::now();
match store.write_metrics(samples, now()).await {
@@ -285,41 +265,3 @@ impl Server {
}
}
}
#[cfg(test)]
mod tests {
use super::*;
fn gauges(samples: &[Metric]) -> Vec<MetricType> {
samples
.iter()
.filter_map(|m| match m {
Metric::Gauge(g) => Some(g.metric),
_ => None,
})
.collect()
}
#[test]
fn only_the_calculating_node_stores_cluster_gauges() {
let all = gauges(&sample(true));
let local = gauges(&sample(false));
for metric in CLUSTER_GAUGES {
assert!(
all.contains(&metric),
"{metric:?} missing on the calculating node"
);
assert!(
!local.contains(&metric),
"{metric:?} stored by a node that doesn't compute it"
);
}
// Per-node gauges are stored either way
for metric in [MetricType::ServerMemory, MetricType::HttpActiveConnections] {
assert!(
all.contains(&metric) && local.contains(&metric),
"{metric:?}"
);
}
}
}
+1 -1
View File
@@ -1,5 +1,5 @@
/*
* SPDX-FileCopyrightText: 2026 Coffey Labs LLC
* SPDX-FileCopyrightText: 2026 Coffey Labs
*
* SPDX-License-Identifier: AGPL-3.0-only
*/
+1 -1
View File
@@ -1,6 +1,6 @@
[package]
name = "coordinator"
version = "0.16.25"
version = "0.16.24"
edition = "2024"
[dependencies]
+1 -1
View File
@@ -1,6 +1,6 @@
[package]
name = "dav-proto"
version = "0.16.25"
version = "0.16.24"
edition = "2024"
[dependencies]
+1 -1
View File
@@ -1,6 +1,6 @@
[package]
name = "dav"
version = "0.16.25"
version = "0.16.24"
edition = "2024"
[dependencies]
+1 -11
View File
@@ -133,10 +133,6 @@ impl DavAclHandler for Server {
{
return Err(DavError::Code(StatusCode::FORBIDDEN));
}
// inbuxa: MA-D0: a group's members don't share what it owns on.
if access_token.is_group_member_only(account_id) {
return Err(DavError::Code(StatusCode::FORBIDDEN));
}
// Validate ACEs
let grants = self
@@ -569,13 +565,7 @@ impl Privileges for AccessToken {
grants: &ArchivedVec<ArchivedAclGrant>,
is_calendar: bool,
) -> Vec<Privilege> {
if self.is_group_member_only(account_id) {
// inbuxa: MA-D0: everything but sharing it on.
Privilege::all(is_calendar)
.into_iter()
.filter(|privilege| !matches!(privilege, Privilege::All | Privilege::WriteAcl))
.collect()
} else if self.is_member(account_id) {
if self.is_member(account_id) {
Privilege::all(is_calendar)
} else {
current_user_privilege_set(grants.effective_acl(self))
+1 -1
View File
@@ -1,6 +1,6 @@
[package]
name = "directory"
version = "0.16.25"
version = "0.16.24"
edition = "2024"
[dependencies]
+1 -1
View File
@@ -1,6 +1,6 @@
[package]
name = "email"
version = "0.16.25"
version = "0.16.24"
edition = "2024"
[dependencies]
+1 -1
View File
@@ -1,5 +1,5 @@
/*
* SPDX-FileCopyrightText: 2026 Coffey Labs LLC
* SPDX-FileCopyrightText: 2026 Coffey Labs
*
* SPDX-License-Identifier: AGPL-3.0-only
*/
+6 -2
View File
@@ -20,7 +20,7 @@ use groupware::{
scheduling::{ItipError, ItipMessages},
};
use mail_parser::{
Header, HeaderName, HeaderValue, Message, MessageParser, MimeHeaders, PartType,
DateTime, Header, HeaderName, HeaderValue, Message, MessageParser, MimeHeaders, PartType,
parsers::fields::thread::thread_name,
};
use registry::{
@@ -924,7 +924,11 @@ impl EmailIngest for Server {
span_id: u64,
) {
if let Some(config) = &self.core.spam.classifier {
let until = now() + config.hold_samples_for;
let mut dt = DateTime::from_timestamp(now() as i64);
dt.hour = 0;
dt.minute = 0;
dt.second = 0;
let until = dt.to_timestamp() as u64 + config.hold_samples_for;
let sample = SpamTrainingSample {
account_id: Some(Id::from(account_id)),
+2 -10
View File
@@ -290,11 +290,7 @@ impl SieveScriptIngest for Server {
// inbuxa: AL-4: a locked account answers no sender, so a
// rejection is kept instead; sieve has already cleared
// the implicit keep, so it is filed here
// A shared mailbox (MA-S) is a role address and answers
// as one: its Sieve script runs as written
Event::Reject { .. }
if access_token.is_locked() && !access_token.is_shared_mailbox() =>
{
Event::Reject { .. } if access_token.is_locked() => {
if let Some(message) = messages.get_mut(0)
&& !message.file_into.contains(&INBOX_ID)
{
@@ -407,11 +403,7 @@ impl SieveScriptIngest for Server {
// inbuxa: AL-4: a locked account sends nothing on its
// own: no redirect, vacation reply or notification. An
// unsent redirect leaves the message to be kept.
// A shared mailbox's acknowledgements and redirects go
// out (MA-S).
Event::SendMessage { .. }
if access_token.is_locked() && !access_token.is_shared_mailbox() =>
{
Event::SendMessage { .. } if access_token.is_locked() => {
trc::event!(
Sieve(SieveEvent::ActionReject),
Details = "Account is locked: nothing is sent",
-7
View File
@@ -21,13 +21,6 @@ base64 = "0.23"
sha2 = "0.11"
flate2 = "1.1"
tokio = { version = "1.53", features = ["sync", "rt"] }
# inbuxa: DLP detectors and attachment text (dlp-and-mail-flow-rules spec)
regex = "1.13.1"
aho-corasick = "1.1"
zip = "8.6"
quick-xml = "0.41"
mail-parser = { version = "0.11", features = ["full_encoding"] }
mail-builder = { version = "1.0" }
[dev-dependencies]
tokio = { version = "1.53", features = ["macros", "rt"] }
+1 -1
View File
@@ -1,5 +1,5 @@
/*
* SPDX-FileCopyrightText: 2026 Coffey Labs LLC
* SPDX-FileCopyrightText: 2026 Coffey Labs
*
* SPDX-License-Identifier: AGPL-3.0-only
*/
+1 -1
View File
@@ -1,5 +1,5 @@
/*
* SPDX-FileCopyrightText: 2026 Coffey Labs LLC
* SPDX-FileCopyrightText: 2026 Coffey Labs
*
* SPDX-License-Identifier: AGPL-3.0-only
*/
+1 -1
View File
@@ -1,5 +1,5 @@
/*
* SPDX-FileCopyrightText: 2026 Coffey Labs LLC
* SPDX-FileCopyrightText: 2026 Coffey Labs
*
* SPDX-License-Identifier: AGPL-3.0-only
*/
+1 -1
View File
@@ -1,5 +1,5 @@
/*
* SPDX-FileCopyrightText: 2026 Coffey Labs LLC
* SPDX-FileCopyrightText: 2026 Coffey Labs
*
* SPDX-License-Identifier: AGPL-3.0-only
*/
+1 -1
View File
@@ -1,5 +1,5 @@
/*
* SPDX-FileCopyrightText: 2026 Coffey Labs LLC
* SPDX-FileCopyrightText: 2026 Coffey Labs
*
* SPDX-License-Identifier: AGPL-3.0-only
*/
+1 -1
View File
@@ -1,5 +1,5 @@
/*
* SPDX-FileCopyrightText: 2026 Coffey Labs LLC
* SPDX-FileCopyrightText: 2026 Coffey Labs
*
* SPDX-License-Identifier: AGPL-3.0-only
*/
+1 -1
View File
@@ -1,5 +1,5 @@
/*
* SPDX-FileCopyrightText: 2026 Coffey Labs LLC
* SPDX-FileCopyrightText: 2026 Coffey Labs
*
* SPDX-License-Identifier: AGPL-3.0-only
*/
+1 -1
View File
@@ -1,5 +1,5 @@
/*
* SPDX-FileCopyrightText: 2026 Coffey Labs LLC
* SPDX-FileCopyrightText: 2026 Coffey Labs
*
* SPDX-License-Identifier: AGPL-3.0-only
*/
+1 -1
View File
@@ -1,5 +1,5 @@
/*
* SPDX-FileCopyrightText: 2026 Coffey Labs LLC
* SPDX-FileCopyrightText: 2026 Coffey Labs
*
* SPDX-License-Identifier: AGPL-3.0-only
*/
+1 -1
View File
@@ -1,5 +1,5 @@
/*
* SPDX-FileCopyrightText: 2026 Coffey Labs LLC
* SPDX-FileCopyrightText: 2026 Coffey Labs
*
* SPDX-License-Identifier: AGPL-3.0-only
*/
+1 -1
View File
@@ -1,5 +1,5 @@
/*
* SPDX-FileCopyrightText: 2026 Coffey Labs LLC
* SPDX-FileCopyrightText: 2026 Coffey Labs
*
* SPDX-License-Identifier: AGPL-3.0-only
*/
+1 -1
View File
@@ -1,5 +1,5 @@
/*
* SPDX-FileCopyrightText: 2026 Coffey Labs LLC
* SPDX-FileCopyrightText: 2026 Coffey Labs
*
* SPDX-License-Identifier: AGPL-3.0-only
*/
+1 -1
View File
@@ -1,5 +1,5 @@
/*
* SPDX-FileCopyrightText: 2026 Coffey Labs LLC
* SPDX-FileCopyrightText: 2026 Coffey Labs
*
* SPDX-License-Identifier: AGPL-3.0-only
*/
+1 -1
View File
@@ -1,5 +1,5 @@
/*
* SPDX-FileCopyrightText: 2026 Coffey Labs LLC
* SPDX-FileCopyrightText: 2026 Coffey Labs
*
* SPDX-License-Identifier: AGPL-3.0-only
*/
+1 -1
View File
@@ -1,5 +1,5 @@
/*
* SPDX-FileCopyrightText: 2026 Coffey Labs LLC
* SPDX-FileCopyrightText: 2026 Coffey Labs
*
* SPDX-License-Identifier: AGPL-3.0-only
*/
+1 -1
View File
@@ -1,5 +1,5 @@
/*
* SPDX-FileCopyrightText: 2026 Coffey Labs LLC
* SPDX-FileCopyrightText: 2026 Coffey Labs
*
* SPDX-License-Identifier: AGPL-3.0-only
*/
+1 -1
View File
@@ -1,5 +1,5 @@
/*
* SPDX-FileCopyrightText: 2026 Coffey Labs LLC
* SPDX-FileCopyrightText: 2026 Coffey Labs
*
* SPDX-License-Identifier: AGPL-3.0-only
*/
+1 -1
View File
@@ -1,5 +1,5 @@
/*
* SPDX-FileCopyrightText: 2026 Coffey Labs LLC
* SPDX-FileCopyrightText: 2026 Coffey Labs
*
* SPDX-License-Identifier: AGPL-3.0-only
*/
+1 -1
View File
@@ -1,5 +1,5 @@
/*
* SPDX-FileCopyrightText: 2026 Coffey Labs LLC
* SPDX-FileCopyrightText: 2026 Coffey Labs
*
* SPDX-License-Identifier: AGPL-3.0-only
*/
+1 -1
View File
@@ -1,5 +1,5 @@
/*
* SPDX-FileCopyrightText: 2026 Coffey Labs LLC
* SPDX-FileCopyrightText: 2026 Coffey Labs
*
* SPDX-License-Identifier: AGPL-3.0-only
*/
+1 -1
View File
@@ -1,5 +1,5 @@
/*
* SPDX-FileCopyrightText: 2026 Coffey Labs LLC
* SPDX-FileCopyrightText: 2026 Coffey Labs
*
* SPDX-License-Identifier: AGPL-3.0-only
*/
-282
View File
@@ -1,282 +0,0 @@
/*
* SPDX-FileCopyrightText: 2026 Coffey Labs LLC
*
* SPDX-License-Identifier: AGPL-3.0-only
*/
//! The blocklists a node asks about itself (deliverability spec, DL-6), and
//! how to read each one's answer.
//!
//! A list answers with an address in 127.0.0.0/8. Each list says which of
//! those mean "listed" and which mean "I won't answer you": Spamhaus, for
//! one, answers `127.255.255.254` to a query that came through a public
//! resolver. A refusal is never read as a listing (DL-4).
use std::net::{IpAddr, Ipv4Addr};
#[derive(Debug, Clone, Copy, PartialEq, Eq)]
pub enum Scope {
/// Looked up by the reversed address: `2.0.0.127.zen.spamhaus.org`.
Ip,
/// Looked up by name: `example.org.dbl.spamhaus.org`.
Domain,
}
#[derive(Debug, Clone, Copy)]
pub struct BlockList {
/// What the page and the settings call it.
pub name: &'static str,
pub zone: &'static str,
pub scope: Scope,
/// Where an administrator looks the address up and asks for removal.
pub lookup: &'static str,
/// Something the page says beside the list.
pub note: Option<&'static str>,
read: fn(Ipv4Addr) -> Answer,
}
/// What a list's answer means.
#[derive(Debug, Clone, PartialEq, Eq)]
pub enum Answer {
Listed(&'static str),
/// The list won't answer this resolver, or not now.
Refused(&'static str),
/// A code the list doesn't define: neither listed nor clean.
Unknown,
}
impl BlockList {
pub fn read(&self, answer: Ipv4Addr) -> Answer {
(self.read)(answer)
}
/// The name to look up for `subject`, or None when the subject doesn't
/// suit the list (a domain on an IP list, or an IPv6 address: none of
/// these lists publish IPv6 zones worth asking).
pub fn query(&self, subject: &Subject<'_>) -> Option<String> {
match (self.scope, subject) {
(Scope::Ip, Subject::Ip(IpAddr::V4(ip))) => {
let [a, b, c, d] = ip.octets();
Some(format!("{d}.{c}.{b}.{a}.{}.", self.zone))
}
(Scope::Domain, Subject::Domain(domain)) => {
Some(format!("{}.{}.", domain.trim_end_matches('.'), self.zone))
}
_ => None,
}
}
}
pub enum Subject<'x> {
Ip(IpAddr),
Domain(&'x str),
}
/// Spamhaus' error codes, the same on every Spamhaus zone.
fn spamhaus_refusal(ip: Ipv4Addr) -> Option<Answer> {
match ip.octets() {
[127, 255, 255, 252] => Some(Answer::Refused("The query was malformed")),
[127, 255, 255, 254] => Some(Answer::Refused(
"Spamhaus doesn't answer public resolvers; use the server's own",
)),
[127, 255, 255, 255] => Some(Answer::Refused("Too many queries from this resolver")),
_ => None,
}
}
fn zen(ip: Ipv4Addr) -> Answer {
if let Some(refused) = spamhaus_refusal(ip) {
return refused;
}
match ip.octets() {
[127, 0, 0, 2] => Answer::Listed("SBL: a known spam source"),
[127, 0, 0, 3] => Answer::Listed("CSS: sent spam recently"),
[127, 0, 0, 4..=7] => Answer::Listed("XBL: a compromised or infected host"),
[127, 0, 0, 9] => Answer::Listed("DROP: a hijacked or criminal network"),
[127, 0, 0, 10 | 11] => {
Answer::Listed("PBL: an address that isn't meant to send mail directly")
}
_ => Answer::Unknown,
}
}
fn dbl(ip: Ipv4Addr) -> Answer {
if let Some(refused) = spamhaus_refusal(ip) {
return refused;
}
match ip.octets() {
[127, 0, 1, 2] => Answer::Listed("A spam domain"),
[127, 0, 1, 4] => Answer::Listed("A phishing domain"),
[127, 0, 1, 5] => Answer::Listed("A malware domain"),
[127, 0, 1, 6] => Answer::Listed("A botnet controller"),
[127, 0, 1, 102..=106] => Answer::Listed("A legitimate domain being abused"),
[127, 0, 1, 255] => Answer::Refused("The query was malformed"),
_ => Answer::Unknown,
}
}
/// Most lists answer 127.0.0.2 for "listed" and define nothing else.
fn just_two(ip: Ipv4Addr) -> Answer {
match ip.octets() {
[127, 0, 0, 2] => Answer::Listed("Listed"),
_ => Answer::Unknown,
}
}
fn surbl(ip: Ipv4Addr) -> Answer {
match ip.octets() {
[127, 0, 0, 1] => Answer::Refused("SURBL doesn't answer this resolver"),
[127, 0, 0, bits] if bits & (8 | 16 | 64 | 128) != 0 => {
Answer::Listed("Seen in phishing, malware, abuse or cracked sites")
}
_ => Answer::Unknown,
}
}
fn uribl(ip: Ipv4Addr) -> Answer {
match ip.octets() {
[127, 0, 0, 1] => Answer::Refused("URIBL doesn't answer public resolvers"),
[127, 0, 0, bits] if bits & (2 | 8) != 0 => Answer::Listed("Seen in spam"),
[127, 0, 0, bits] if bits & 4 != 0 => {
Answer::Listed("Grey: seen in bulk mail some people don't want")
}
_ => Answer::Unknown,
}
}
pub const LISTS: &[BlockList] = &[
BlockList {
name: "Spamhaus ZEN",
zone: "zen.spamhaus.org",
scope: Scope::Ip,
lookup: "https://check.spamhaus.org/",
note: None,
read: zen,
},
BlockList {
name: "SpamCop",
zone: "bl.spamcop.net",
scope: Scope::Ip,
lookup: "https://www.spamcop.net/bl.shtml",
note: None,
read: just_two,
},
BlockList {
name: "Barracuda",
zone: "b.barracudacentral.org",
scope: Scope::Ip,
lookup: "https://www.barracudacentral.org/lookups",
note: Some(
"Barracuda answers only resolvers whose address is registered with it (free, at barracudacentral.org/rbl). Until then its lookups can't be checked.",
),
read: just_two,
},
BlockList {
name: "UCEPROTECT level 1",
zone: "dnsbl-1.uceprotect.net",
scope: Scope::Ip,
lookup: "https://www.uceprotect.net/en/rblcheck.php",
note: None,
read: just_two,
},
BlockList {
name: "Mailspike",
zone: "bl.mailspike.net",
scope: Scope::Ip,
lookup: "https://mailspike.org/iplookup.html",
note: None,
read: just_two,
},
BlockList {
name: "PSBL",
zone: "psbl.surriel.com",
scope: Scope::Ip,
lookup: "https://psbl.org/",
note: None,
read: just_two,
},
BlockList {
name: "Spamhaus DBL",
zone: "dbl.spamhaus.org",
scope: Scope::Domain,
lookup: "https://check.spamhaus.org/",
note: None,
read: dbl,
},
BlockList {
name: "SURBL",
zone: "multi.surbl.org",
scope: Scope::Domain,
lookup: "https://surbl.org/surbl-analysis",
note: None,
read: surbl,
},
BlockList {
name: "URIBL",
zone: "multi.uribl.com",
scope: Scope::Domain,
lookup: "https://admin.uribl.com/",
note: None,
read: uribl,
},
];
pub fn by_name(name: &str) -> Option<&'static BlockList> {
LISTS.iter().find(|list| list.name == name)
}
#[cfg(test)]
mod tests {
use super::*;
fn ip(s: &str) -> Ipv4Addr {
s.parse().unwrap()
}
#[test]
fn a_refusal_is_not_a_listing() {
let zen = by_name("Spamhaus ZEN").unwrap();
assert!(matches!(
zen.read(ip("127.255.255.254")),
Answer::Refused(_)
));
assert!(matches!(zen.read(ip("127.0.0.2")), Answer::Listed(_)));
assert!(matches!(zen.read(ip("127.0.0.10")), Answer::Listed(_)));
assert_eq!(zen.read(ip("127.0.0.200")), Answer::Unknown);
let uribl = by_name("URIBL").unwrap();
assert!(matches!(uribl.read(ip("127.0.0.1")), Answer::Refused(_)));
assert!(matches!(uribl.read(ip("127.0.0.2")), Answer::Listed(_)));
}
#[test]
fn queries_are_built_per_scope() {
let zen = by_name("Spamhaus ZEN").unwrap();
let dbl = by_name("Spamhaus DBL").unwrap();
let v4 = Subject::Ip("192.0.2.10".parse().unwrap());
let v6 = Subject::Ip("2001:db8::1".parse().unwrap());
let domain = Subject::Domain("example.org");
assert_eq!(
zen.query(&v4).as_deref(),
Some("10.2.0.192.zen.spamhaus.org.")
);
assert_eq!(zen.query(&v6), None);
assert_eq!(zen.query(&domain), None);
assert_eq!(
dbl.query(&domain).as_deref(),
Some("example.org.dbl.spamhaus.org.")
);
assert_eq!(dbl.query(&v4), None);
}
#[test]
fn names_are_unique() {
for (i, a) in LISTS.iter().enumerate() {
assert!(
LISTS[i + 1..].iter().all(|b| b.name != a.name),
"{}",
a.name
);
}
}
}
-410
View File
@@ -1,410 +0,0 @@
/*
* SPDX-FileCopyrightText: 2026 Coffey Labs LLC
*
* SPDX-License-Identifier: AGPL-3.0-only
*/
//! The deliverability check (deliverability spec): what other mail servers
//! see when this one sends. Not a rebuild of anything upstream ships.
//!
//! Every node that sends mail checks itself, because only it knows which
//! address it leaves from, and keeps one report. The report holds facts: an
//! address's reverse DNS, what each blocklist answered, what SPF said for
//! each address, whether a DKIM key in DNS matches the one signing. The
//! console grades them, so its wording can change without a server release.
//!
//! Kept in the fork's subspace (`store::SUBSPACE_INBUXA`). Every key starts
//! with `D`, then one byte for the kind:
//!
//! - `r` + node id (u64): that node's last report, as JSON.
//! - `s`: the settings, as JSON.
//!
//! Numbers are big-endian.
pub mod lists;
use serde::{Deserialize as SerdeDeserialize, Serialize as SerdeSerialize};
use store::{
Deserialize, IterateParams, SUBSPACE_INBUXA, Serialize, Store, ValueKey,
write::{AnyClass, BatchBuilder, ValueClass},
};
use trc::AddContext;
const FEATURE: u8 = b'D';
const KIND_REPORT: u8 = b'r';
const KIND_SETTINGS: u8 = b's';
/// DL-15: **Check now** runs a node again only this long after its last run.
pub const MIN_INTERVAL_SECS: u64 = 600;
#[derive(Debug, Clone, Default, PartialEq, SerdeSerialize, SerdeDeserialize)]
#[serde(rename_all = "camelCase", default)]
pub struct Report {
/// The node's cluster id, as metric samples carry it.
pub node_id: u64,
pub hostname: String,
/// Seconds since the epoch.
pub checked_at: u64,
pub addresses: Vec<Address>,
pub domains: Vec<DomainReport>,
pub certificates: Vec<Certificate>,
}
#[derive(Debug, Clone, Default, PartialEq, SerdeSerialize, SerdeDeserialize)]
#[serde(rename_all = "camelCase", default)]
pub struct Address {
pub ip: String,
/// DL-2: how the node came by the address.
pub source: AddressSource,
/// The connection strategy that sends from it.
pub strategy: String,
/// The name the node greets with from this address.
pub ehlo: String,
/// The PTR names, empty when there's none.
pub ptr: Vec<String>,
/// Some PTR name resolves back to the address.
pub forward_confirmed: bool,
/// The forward-confirmed name is the EHLO name.
pub ehlo_matches: bool,
/// Set when the reverse lookup itself failed, rather than found nothing.
pub ptr_error: Option<String>,
pub listings: Vec<Listing>,
}
#[derive(Debug, Clone, Copy, Default, PartialEq, Eq, SerdeSerialize, SerdeDeserialize)]
#[serde(rename_all = "camelCase")]
pub enum AddressSource {
/// Set in the connection strategy's source addresses.
#[default]
Configured,
/// What the EHLO name resolves to.
Ehlo,
}
#[derive(Debug, Clone, Default, PartialEq, SerdeSerialize, SerdeDeserialize)]
#[serde(rename_all = "camelCase", default)]
pub struct Listing {
/// The list's name, as in [`lists::LISTS`].
pub list: String,
pub state: ListingState,
/// The address the list answered, when it answered one.
pub code: Option<String>,
/// What the list says the answer means.
pub meaning: Option<String>,
}
#[derive(Debug, Clone, Copy, Default, PartialEq, Eq, SerdeSerialize, SerdeDeserialize)]
#[serde(rename_all = "camelCase")]
pub enum ListingState {
#[default]
Clean,
Listed,
/// The list wouldn't answer, or the lookup failed: neither listed nor clean.
Refused,
Error,
/// Switched off in the settings, so not asked.
Off,
}
#[derive(Debug, Clone, Default, PartialEq, SerdeSerialize, SerdeDeserialize)]
#[serde(rename_all = "camelCase", default)]
pub struct DomainReport {
pub domain: String,
/// DL-20: a tenant administrator sees only their tenant's domains.
pub tenant_id: Option<u32>,
/// DL-7: what SPF says for each of the node's addresses.
pub spf: Vec<SpfResult>,
/// DL-8: each DKIM key the domain signs with.
pub dkim: Vec<DkimKey>,
/// DL-9: the DMARC record, if there's one.
pub dmarc: Option<Dmarc>,
/// DL-10.
pub mta_sts: MtaSts,
/// DL-11: there's a `_smtp._tls` record.
pub tls_rpt: bool,
/// DL-12.
pub listings: Vec<Listing>,
}
#[derive(Debug, Clone, Default, PartialEq, SerdeSerialize, SerdeDeserialize)]
#[serde(rename_all = "camelCase", default)]
pub struct SpfResult {
pub ip: String,
/// `pass`, `fail`, `softFail`, `neutral`, `none`, `tempError` or `permError`.
pub result: String,
}
#[derive(Debug, Clone, Default, PartialEq, SerdeSerialize, SerdeDeserialize)]
#[serde(rename_all = "camelCase", default)]
pub struct DkimKey {
pub selector: String,
pub state: DkimState,
}
#[derive(Debug, Clone, Copy, Default, PartialEq, Eq, SerdeSerialize, SerdeDeserialize)]
#[serde(rename_all = "camelCase")]
pub enum DkimState {
#[default]
Matches,
/// Nothing published at `<selector>._domainkey.<domain>`.
Missing,
/// Published, but a different key.
Different,
/// The lookup failed.
Error,
}
#[derive(Debug, Clone, Default, PartialEq, SerdeSerialize, SerdeDeserialize)]
#[serde(rename_all = "camelCase", default)]
pub struct Dmarc {
/// `none`, `quarantine` or `reject`.
pub policy: String,
/// DKIM alignment: `relaxed` or `strict`.
pub adkim: String,
/// SPF alignment: `relaxed` or `strict`.
pub aspf: String,
}
#[derive(Debug, Clone, Default, PartialEq, SerdeSerialize, SerdeDeserialize)]
#[serde(rename_all = "camelCase", default)]
pub struct MtaSts {
/// The `_mta-sts` record's id; None when there's no record.
pub record_id: Option<String>,
/// The policy was fetched and parsed. False with a record means the
/// fetch or the parse failed, and `error` says why.
pub fetched: bool,
pub error: Option<String>,
/// `enforce`, `testing` or `none`.
pub mode: Option<String>,
pub max_age: Option<u64>,
/// The domain's MX names no `mx:` line matches.
pub mx_not_covered: Vec<String>,
}
#[derive(Debug, Clone, Default, PartialEq, SerdeSerialize, SerdeDeserialize)]
#[serde(rename_all = "camelCase", default)]
pub struct Certificate {
/// The EHLO name, or an MX name that points at this node.
pub name: String,
/// The node holds a certificate for the name.
pub covered: bool,
}
#[derive(Debug, Clone, Default, PartialEq, SerdeSerialize, SerdeDeserialize)]
#[serde(rename_all = "camelCase", default)]
pub struct Settings {
/// DL-6: lists not to ask, by name.
pub disabled_lists: Vec<String>,
}
impl Settings {
pub fn is_off(&self, list: &str) -> bool {
self.disabled_lists.iter().any(|name| name == list)
}
/// Only the built-in lists' names, once each.
pub fn validate(&self) -> Result<(), String> {
for (i, name) in self.disabled_lists.iter().enumerate() {
if lists::by_name(name).is_none() {
return Err(format!("There's no list called {name:?}."));
}
if self.disabled_lists[..i].contains(name) {
return Err(format!("{name:?} is named twice."));
}
}
Ok(())
}
}
impl Report {
/// DL-20: what a tenant administrator may see: their tenant's domains
/// and nothing about the node's addresses or certificates.
pub fn for_tenant(&self, tenant_id: u32) -> Report {
Report {
node_id: self.node_id,
hostname: self.hostname.clone(),
checked_at: self.checked_at,
addresses: Vec::new(),
domains: self
.domains
.iter()
.filter(|d| d.tenant_id == Some(tenant_id))
.cloned()
.collect(),
certificates: Vec::new(),
}
}
}
// --- Storage --------------------------------------------------------------
struct Json<T>(T);
impl<T: SerdeSerialize> Serialize for Json<T> {
fn serialize(&self) -> trc::Result<Vec<u8>> {
serde_json::to_vec(&self.0).map_err(|err| {
trc::StoreEvent::UnexpectedError
.into_err()
.details("Failed to serialize deliverability data")
.reason(err)
})
}
}
impl<T: for<'de> SerdeDeserialize<'de> + Send + Sync> Deserialize for Json<T> {
fn deserialize(bytes: &[u8]) -> trc::Result<Self> {
serde_json::from_slice(bytes).map(Json).map_err(|err| {
trc::StoreEvent::DataCorruption
.into_err()
.details("Invalid deliverability data")
.reason(err)
})
}
}
fn class(kind: u8, node_id: Option<u64>) -> ValueClass {
let mut key = Vec::with_capacity(10);
key.push(FEATURE);
key.push(kind);
if let Some(node_id) = node_id {
key.extend_from_slice(&node_id.to_be_bytes());
}
ValueClass::Any(AnyClass {
subspace: SUBSPACE_INBUXA,
key,
})
}
pub async fn report(data: &Store, node_id: u64) -> trc::Result<Option<Report>> {
Ok(data
.get_value::<Json<Report>>(ValueKey::from(class(KIND_REPORT, Some(node_id))))
.await
.caused_by(trc::location!())?
.map(|Json(report)| report))
}
/// Every node's report, by node id.
pub async fn reports(data: &Store) -> trc::Result<Vec<Report>> {
let mut out = Vec::new();
data.iterate(
IterateParams::new(
ValueKey::from(class(KIND_REPORT, Some(0))),
ValueKey::from(class(KIND_REPORT, Some(u64::MAX))),
),
|_, value| {
if let Ok(Json(report)) = Json::<Report>::deserialize(value) {
out.push(report);
}
Ok(true)
},
)
.await
.caused_by(trc::location!())?;
out.sort_by_key(|r| r.node_id);
Ok(out)
}
/// Replaces the node's report.
pub async fn put_report(data: &Store, report: &Report) -> trc::Result<()> {
let mut batch = BatchBuilder::new();
batch.set(
class(KIND_REPORT, Some(report.node_id)),
Json(report).serialize()?,
);
data.write(batch.build_all())
.await
.caused_by(trc::location!())?;
Ok(())
}
pub async fn settings(data: &Store) -> trc::Result<Settings> {
Ok(data
.get_value::<Json<Settings>>(ValueKey::from(class(KIND_SETTINGS, None)))
.await
.caused_by(trc::location!())?
.map(|Json(settings)| settings)
.unwrap_or_default())
}
pub async fn put_settings(data: &Store, settings: &Settings) -> trc::Result<()> {
let mut batch = BatchBuilder::new();
batch.set(class(KIND_SETTINGS, None), Json(settings).serialize()?);
data.write(batch.build_all())
.await
.caused_by(trc::location!())?;
Ok(())
}
#[cfg(test)]
mod tests {
use super::*;
#[test]
fn settings_name_only_built_in_lists_once() {
let ok = Settings {
disabled_lists: vec!["Barracuda".into(), "URIBL".into()],
};
assert!(ok.validate().is_ok());
assert!(ok.is_off("Barracuda"));
assert!(!ok.is_off("SpamCop"));
let unknown = Settings {
disabled_lists: vec!["My list".into()],
};
assert!(unknown.validate().is_err());
let twice = Settings {
disabled_lists: vec!["URIBL".into(), "URIBL".into()],
};
assert!(twice.validate().is_err());
}
#[test]
fn a_tenant_sees_only_its_domains() {
let report = Report {
node_id: 2,
hostname: "mx2.example.org".into(),
checked_at: 1,
addresses: vec![Address {
ip: "192.0.2.10".into(),
..Default::default()
}],
domains: vec![
DomainReport {
domain: "a.example".into(),
tenant_id: Some(7),
..Default::default()
},
DomainReport {
domain: "b.example".into(),
tenant_id: Some(8),
..Default::default()
},
DomainReport {
domain: "server.example".into(),
tenant_id: None,
..Default::default()
},
],
certificates: vec![Certificate {
name: "mx2.example.org".into(),
covered: true,
}],
};
let seen = report.for_tenant(7);
assert!(seen.addresses.is_empty());
assert!(seen.certificates.is_empty());
assert_eq!(
seen.domains
.iter()
.map(|d| d.domain.as_str())
.collect::<Vec<_>>(),
["a.example"]
);
}
#[test]
fn a_report_reads_back_with_missing_fields() {
let report: Report = serde_json::from_str(r#"{"nodeId": 3}"#).unwrap();
assert_eq!(report.node_id, 3);
assert!(report.domains.is_empty());
}
}
+1 -1
View File
@@ -1,5 +1,5 @@
/*
* SPDX-FileCopyrightText: 2026 Coffey Labs LLC
* SPDX-FileCopyrightText: 2026 Coffey Labs
*
* SPDX-License-Identifier: AGPL-3.0-only
*/
-120
View File
@@ -1,120 +0,0 @@
/*
* SPDX-FileCopyrightText: 2026 Coffey Labs LLC
*
* SPDX-License-Identifier: AGPL-3.0-only
*/
//! Reports on their way to an outside archive (JR-7). Keys, after `J`:
//!
//! - `o` + the report's queue id: what goes into the built-in journal if
//! the archive never takes the report, as JSON. Cleared once it's
//! delivered or kept.
//! - `w` + journal id (u32): how often that journal's archive didn't take a
//! report, and the last time and reason, for the console's warning.
use super::{FEATURE, Json, entries::Entry};
use serde::{Deserialize as SerdeDeserialize, Serialize as SerdeSerialize};
use store::{
SUBSPACE_INBUXA, Serialize, Store, ValueKey,
write::{AnyClass, BatchBuilder, ValueClass},
};
use trc::AddContext;
const KIND_PENDING: u8 = b'o';
const KIND_FAILURES: u8 = b'w';
/// A report queued to an archive.
#[derive(Debug, Clone, PartialEq, Eq, SerdeSerialize, SerdeDeserialize)]
#[serde(rename_all = "camelCase")]
pub struct Pending {
pub address: String,
/// The entry, should the archive not take it: its own, with the
/// sending journals' retention, whatever else the built-in journal has.
pub entry: Entry,
}
/// How a journal's archive has been taking its reports.
#[derive(Debug, Clone, Default, PartialEq, Eq, SerdeSerialize, SerdeDeserialize)]
#[serde(rename_all = "camelCase")]
pub struct Failures {
pub count: u64,
/// Seconds.
pub last_at: u64,
pub last_reason: String,
}
fn class(kind: u8, id: &[u8]) -> ValueClass {
let mut key = Vec::with_capacity(2 + id.len());
key.push(FEATURE);
key.push(kind);
key.extend_from_slice(id);
ValueClass::Any(AnyClass {
subspace: SUBSPACE_INBUXA,
key,
})
}
pub async fn set_pending(data: &Store, queue_id: u64, pending: &Pending) -> trc::Result<()> {
let mut batch = BatchBuilder::new();
batch.set(
class(KIND_PENDING, &queue_id.to_be_bytes()),
Json(pending).serialize()?,
);
data.write(batch.build_all())
.await
.caused_by(trc::location!())?;
Ok(())
}
pub async fn pending(data: &Store, queue_id: u64) -> trc::Result<Option<Pending>> {
Ok(data
.get_value::<Json<Pending>>(ValueKey::from(class(KIND_PENDING, &queue_id.to_be_bytes())))
.await
.caused_by(trc::location!())?
.map(|Json(pending)| pending))
}
pub async fn clear_pending(data: &Store, queue_id: u64) -> trc::Result<()> {
let mut batch = BatchBuilder::new();
batch.clear(class(KIND_PENDING, &queue_id.to_be_bytes()));
data.write(batch.build_all())
.await
.caused_by(trc::location!())?;
Ok(())
}
pub async fn failures(data: &Store, journal_id: u32) -> trc::Result<Failures> {
Ok(data
.get_value::<Json<Failures>>(ValueKey::from(class(
KIND_FAILURES,
&journal_id.to_be_bytes(),
)))
.await
.caused_by(trc::location!())?
.map(|Json(failures)| failures)
.unwrap_or_default())
}
/// Counts one report an archive didn't take, for each of `journals`.
pub async fn record_failure(
data: &Store,
journals: &[u32],
at: u64,
reason: &str,
) -> trc::Result<()> {
for journal_id in journals {
let mut failures = failures(data, *journal_id).await?;
failures.count += 1;
failures.last_at = at;
failures.last_reason = reason.chars().take(500).collect();
let mut batch = BatchBuilder::new();
batch.set(
class(KIND_FAILURES, &journal_id.to_be_bytes()),
Json(&failures).serialize()?,
);
data.write(batch.build_all())
.await
.caused_by(trc::location!())?;
}
Ok(())
}
-869
View File
@@ -1,869 +0,0 @@
/*
* SPDX-FileCopyrightText: 2026 Coffey Labs LLC
*
* SPDX-License-Identifier: AGPL-3.0-only
*/
//! The built-in journal (JR-5, JR-6, JR-13). Keys, after `J`:
//!
//! - `e` + node + seq: a chain link: its seq, the hash of the link before
//! it, and the SHA-256 of its entry. One chain per node, as the audit log
//! keeps (AU-6), but a link names its entry by hash instead of holding it,
//! so an entry can go at the end of its own retention without breaking
//! the chain: entries don't expire in chain order.
//! - `c` + node + seq: the entry, as JSON; its bytes are what the link's
//! hash names.
//! - `p` + node + seq: when an entry past its retention was purged. A link
//! whose entry is gone without this marker is a broken chain.
//! - `t` + time + node + seq: the time index, for search.
//! - `x` + expiry + node + seq: the expiry index, for purge.
//! - `h` + node: the chain's head: its hash, then its seq as the last eight
//! bytes, which each append asserts.
//! - `f` + node: where the chain starts after purged links at its start
//! were cleared, and the hash the first kept link names.
//!
//! The report itself is a blob, kept by a temporary link that lasts until
//! its entry is purged. Nothing here changes or removes an entry before
//! its time; nothing in JMAP can.
use super::{Direction, FEATURE, Json};
use crate::hold::HELD_UNTIL;
use serde::{Deserialize as SerdeDeserialize, Serialize as SerdeSerialize};
use sha2::{Digest, Sha256};
use std::fmt;
use store::{
BlobStore, Deserialize, IterateParams, SUBSPACE_INBUXA, Serialize, Store, ValueKey,
write::{AnyClass, BatchBuilder, BlobLink, BlobOp, ValueClass, assert::AssertValue},
};
use tokio::sync::Mutex;
use trc::AddContext;
use types::blob_hash::BlobHash;
const KIND_LINK: u8 = b'e';
const KIND_CONTENT: u8 = b'c';
const KIND_PURGED: u8 = b'p';
const KIND_TIME: u8 = b't';
const KIND_EXPIRY: u8 = b'x';
const KIND_HEAD: u8 = b'h';
const KIND_FLOOR: u8 = b'f';
const APPEND_ATTEMPTS: usize = 5;
/// Entries purged per batch.
const PURGE_BATCH: usize = 100;
/// Where one entry sits: its node's chain and its place in it.
#[derive(Debug, Clone, Copy, PartialEq, Eq, Hash, PartialOrd, Ord)]
pub struct EntryId {
pub node: u64,
pub seq: u64,
}
impl EntryId {
/// As one number, for JMAP ids: the node in the top 16 bits.
pub fn to_u64(&self) -> u64 {
(self.node << 48) | (self.seq & ((1 << 48) - 1))
}
pub fn from_u64(id: u64) -> Self {
EntryId {
node: id >> 48,
seq: id & ((1 << 48) - 1),
}
}
}
impl fmt::Display for EntryId {
fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result {
write!(f, "{}-{}", self.node, self.seq)
}
}
/// One journaled message (JR-5).
#[derive(Debug, Clone, PartialEq, Eq, SerdeSerialize, SerdeDeserialize)]
#[serde(rename_all = "camelCase")]
pub struct Entry {
pub queue_id: u64,
/// Seconds.
pub at: u64,
pub direction: Direction,
pub sender: String,
pub authenticated: bool,
pub recipients: Vec<String>,
pub subject: String,
pub message_id: String,
/// The people here on either side, whose holds keep the entry.
pub accounts: Vec<u32>,
pub tenants: Vec<u32>,
/// The journals that took it.
pub journals: Vec<u32>,
pub held: bool,
/// The report's blob, hex.
pub blob: String,
pub size: u64,
/// SHA-256 of the report, hex.
pub sha256: String,
/// Seconds.
pub expires_at: u64,
}
impl Entry {
pub fn blob_hash(&self) -> Option<BlobHash> {
let bytes = unhex(&self.blob)?;
BlobHash::try_from_hash_slice(&bytes).ok()
}
}
#[derive(Debug, Clone, SerdeSerialize, SerdeDeserialize)]
#[serde(rename_all = "camelCase")]
struct Link {
seq: u64,
prev: String,
content: String,
}
#[derive(Debug, Clone, Default, PartialEq, SerdeSerialize, SerdeDeserialize)]
struct Floor {
seq: u64,
prev: String,
}
#[derive(Debug, Clone, Default, PartialEq)]
struct Head {
seq: u64,
hash: String,
}
impl Head {
fn to_bytes(&self) -> Vec<u8> {
let mut bytes = self.hash.as_bytes().to_vec();
bytes.extend_from_slice(&self.seq.to_be_bytes());
bytes
}
}
impl Deserialize for Head {
fn deserialize(bytes: &[u8]) -> trc::Result<Self> {
let split = bytes.len().checked_sub(8).ok_or_else(|| {
trc::StoreEvent::DataCorruption
.into_err()
.details("Invalid journal chain head")
})?;
Ok(Head {
seq: u64::from_be_bytes(bytes[split..].try_into().unwrap()),
hash: String::from_utf8_lossy(&bytes[..split]).into_owned(),
})
}
}
struct Raw(Vec<u8>);
impl Deserialize for Raw {
fn deserialize(bytes: &[u8]) -> trc::Result<Self> {
Ok(Raw(bytes.to_vec()))
}
}
fn class(kind: u8, parts: &[u64]) -> ValueClass {
let mut key = Vec::with_capacity(2 + parts.len() * 8);
key.push(FEATURE);
key.push(kind);
for part in parts {
key.extend_from_slice(&part.to_be_bytes());
}
ValueClass::Any(AnyClass {
subspace: SUBSPACE_INBUXA,
key,
})
}
fn key(kind: u8, parts: &[u64]) -> ValueKey<ValueClass> {
ValueKey::from(class(kind, parts))
}
/// Where an entry's content is kept, for tests that check tampering shows.
pub fn content_key(id: EntryId) -> ValueKey<ValueClass> {
key(KIND_CONTENT, &[id.node, id.seq])
}
/// The numbers after the kind byte, from the key's tail.
fn parse_key(key: &[u8], kind: u8, parts: usize) -> Option<Vec<u64>> {
let len = 2 + parts * 8;
let tail = key.get(key.len().checked_sub(len)?..)?;
(tail[0] == FEATURE && tail[1] == kind).then_some(())?;
Some(
tail[2..]
.chunks_exact(8)
.map(|chunk| u64::from_be_bytes(chunk.try_into().unwrap()))
.collect(),
)
}
pub fn hex(bytes: &[u8]) -> String {
bytes.iter().map(|b| format!("{b:02x}")).collect()
}
fn unhex(value: &str) -> Option<Vec<u8>> {
(value.len() % 2 == 0).then_some(())?;
(0..value.len())
.step_by(2)
.map(|i| u8::from_str_radix(value.get(i..i + 2)?, 16).ok())
.collect()
}
pub fn sha256(bytes: &[u8]) -> String {
hex(&Sha256::digest(bytes))
}
async fn head(data: &Store, node: u64) -> trc::Result<Option<Head>> {
data.get_value::<Head>(key(KIND_HEAD, &[node]))
.await
.caused_by(trc::location!())
}
async fn floor(data: &Store, node: u64) -> trc::Result<Floor> {
Ok(data
.get_value::<Json<Floor>>(key(KIND_FLOOR, &[node]))
.await
.caused_by(trc::location!())?
.map(|Json(floor)| floor)
.unwrap_or(Floor {
seq: 1,
prev: String::new(),
}))
}
async fn nodes(data: &Store) -> trc::Result<Vec<u64>> {
let mut nodes = Vec::new();
data.iterate(
IterateParams::new(key(KIND_HEAD, &[0]), key(KIND_HEAD, &[u64::MAX])).no_values(),
|key, _| {
if let Some(parts) = parse_key(key, KIND_HEAD, 1) {
nodes.push(parts[0]);
}
Ok(true)
},
)
.await
.caused_by(trc::location!())?;
Ok(nodes)
}
/// Lines up this process's appends; the store's assert settles the rest.
static APPENDING: Mutex<()> = Mutex::const_new(());
/// Adds an entry to this node's chain, and links its report's blob (already
/// written) until the entry is purged. An error means nothing was written.
pub async fn append(data: &Store, node: u64, entry: &Entry) -> trc::Result<EntryId> {
let blob = entry.blob_hash().ok_or_else(|| {
trc::StoreEvent::UnexpectedError
.into_err()
.details("Journal entry without a blob")
})?;
let content = Json(entry).serialize()?;
let content_hash = sha256(&content);
let _appending = APPENDING.lock().await;
let mut attempt = 0;
loop {
attempt += 1;
let current = head(data, node).await?;
let (seq, prev) = current
.as_ref()
.map_or((1, String::new()), |head| (head.seq + 1, head.hash.clone()));
let link = Json(&Link {
seq,
prev,
content: content_hash.clone(),
})
.serialize()?;
let new_head = Head {
seq,
hash: sha256(&link),
};
let mut batch = BatchBuilder::new();
batch.assert_value(
class(KIND_HEAD, &[node]),
current.map_or(AssertValue::None, |head| AssertValue::U64(head.seq)),
);
batch
.set(class(KIND_LINK, &[node, seq]), link)
.set(class(KIND_CONTENT, &[node, seq]), content.clone())
.set(class(KIND_TIME, &[entry.at, node, seq]), vec![])
.set(class(KIND_EXPIRY, &[entry.expires_at, node, seq]), vec![])
.set(class(KIND_HEAD, &[node]), new_head.to_bytes())
.set(
BlobOp::Link {
hash: blob.clone(),
to: BlobLink::Temporary { until: HELD_UNTIL },
},
vec![],
)
.set(BlobOp::Commit { hash: blob.clone() }, vec![]);
match data.write(batch.build_all()).await {
Ok(_) => return Ok(EntryId { node, seq }),
Err(err)
if attempt < APPEND_ATTEMPTS
&& matches!(
err.as_ref(),
trc::EventType::Store(trc::StoreEvent::AssertValueFailed)
) => {}
Err(err) => return Err(err.caused_by(trc::location!())),
}
}
}
/// One entry, unless it was purged.
pub async fn get(data: &Store, id: EntryId) -> trc::Result<Option<Entry>> {
Ok(data
.get_value::<Json<Entry>>(key(KIND_CONTENT, &[id.node, id.seq]))
.await
.caused_by(trc::location!())?
.map(|Json(entry)| entry))
}
/// Entries written in `[after, before)` (seconds), newest first, up to
/// `limit`.
pub async fn list(
data: &Store,
after: u64,
before: u64,
limit: usize,
) -> trc::Result<Vec<(EntryId, Entry)>> {
let mut ids = Vec::new();
data.iterate(
IterateParams::new(
key(KIND_TIME, &[after, 0, 0]),
key(KIND_TIME, &[before.saturating_sub(1), u64::MAX, u64::MAX]),
)
.descending()
.no_values(),
|key, _| {
if let Some(parts) = parse_key(key, KIND_TIME, 3) {
ids.push(EntryId {
node: parts[1],
seq: parts[2],
});
}
Ok(ids.len() < limit)
},
)
.await
.caused_by(trc::location!())?;
let mut out = Vec::with_capacity(ids.len());
for id in ids {
if let Some(entry) = get(data, id).await? {
out.push((id, entry));
}
}
Ok(out)
}
/// Most results one search page returns.
pub const MAX_QUERY_LIMIT: usize = 500;
/// A search of the journal (JR-15): conditions that must all hold.
#[derive(Debug, Clone, Default, PartialEq, Eq, SerdeSerialize)]
#[serde(rename_all = "camelCase")]
pub struct Filter {
/// From this time on, in seconds.
#[serde(skip_serializing_if = "Option::is_none")]
pub after: Option<u64>,
/// Before this time, in seconds.
#[serde(skip_serializing_if = "Option::is_none")]
pub before: Option<u64>,
/// Part of the sender's address, ignoring case.
#[serde(skip_serializing_if = "Option::is_none")]
pub sender: Option<String>,
/// Part of any recipient's address, ignoring case.
#[serde(skip_serializing_if = "Option::is_none")]
pub recipient: Option<String>,
/// Part of the sender's or any recipient's address.
#[serde(skip_serializing_if = "Option::is_none")]
pub address: Option<String>,
#[serde(skip_serializing_if = "Option::is_none")]
pub direction: Option<Direction>,
/// Words that must all appear in the subject, ignoring case.
#[serde(skip_serializing_if = "Option::is_none")]
pub text: Option<String>,
#[serde(skip_serializing_if = "Option::is_none")]
pub message_id: Option<String>,
#[serde(skip_serializing_if = "Option::is_none")]
pub journal_id: Option<u32>,
}
impl Filter {
pub fn matches(&self, entry: &Entry) -> bool {
let has = |value: &str, part: &str| value.to_lowercase().contains(&part.to_lowercase());
self.after.is_none_or(|after| entry.at >= after)
&& self.before.is_none_or(|before| entry.at < before)
&& self.sender.as_deref().is_none_or(|s| has(&entry.sender, s))
&& self
.recipient
.as_deref()
.is_none_or(|r| entry.recipients.iter().any(|a| has(a, r)))
&& self
.address
.as_deref()
.is_none_or(|a| has(&entry.sender, a) || entry.recipients.iter().any(|r| has(r, a)))
&& self
.direction
.is_none_or(|d| d == Direction::Any || d == entry.direction)
&& self.text.as_deref().is_none_or(|text| {
let subject = entry.subject.to_lowercase();
text.to_lowercase()
.split_whitespace()
.all(|word| subject.contains(word))
})
&& self.message_id.as_deref().is_none_or(|id| {
entry.message_id.trim_matches(['<', '>']) == id.trim_matches(['<', '>'])
})
&& self.journal_id.is_none_or(|j| entry.journals.contains(&j))
}
}
/// Entries matching `filter`, newest first: a page from `position`, up to
/// `limit`, and, when asked, how many match in all.
pub async fn query(
data: &Store,
filter: &Filter,
position: usize,
limit: usize,
count_all: bool,
) -> trc::Result<(Vec<EntryId>, usize)> {
let after = filter.after.unwrap_or(0);
let before = filter.before.unwrap_or(u64::MAX);
let mut ids = Vec::new();
data.iterate(
IterateParams::new(
key(KIND_TIME, &[after, 0, 0]),
key(KIND_TIME, &[before.saturating_sub(1), u64::MAX, u64::MAX]),
)
.descending()
.no_values(),
|key, _| {
if let Some(parts) = parse_key(key, KIND_TIME, 3) {
ids.push(EntryId {
node: parts[1],
seq: parts[2],
});
}
Ok(true)
},
)
.await
.caused_by(trc::location!())?;
let mut page = Vec::new();
let mut total = 0;
for id in ids {
let Some(entry) = get(data, id).await? else {
continue;
};
if !filter.matches(&entry) {
continue;
}
if total >= position && page.len() < limit {
page.push(id);
}
total += 1;
if !count_all && page.len() >= limit {
break;
}
}
Ok((page, total))
}
/// What a purge did.
#[derive(Debug, Clone, Default, PartialEq, Eq)]
pub struct Purged {
pub removed: usize,
/// Past their time, kept for a legal hold.
pub kept_for_hold: usize,
}
/// Removes entries past their retention (JR-13), except those `held` keeps:
/// the entry, its indexes and its blob's link go; the chain link stays,
/// with a purge marker. Then each chain's start moves past purged links.
pub async fn purge(
data: &Store,
now: u64,
held: impl Fn(&Entry) -> bool + Sync + Send,
) -> trc::Result<Purged> {
let mut due = Vec::new();
data.iterate(
IterateParams::new(
key(KIND_EXPIRY, &[0, 0, 0]),
key(KIND_EXPIRY, &[now, u64::MAX, u64::MAX]),
)
.ascending()
.no_values(),
|key, _| {
if let Some(parts) = parse_key(key, KIND_EXPIRY, 3) {
due.push((
parts[0],
EntryId {
node: parts[1],
seq: parts[2],
},
));
}
Ok(due.len() < 100_000)
},
)
.await
.caused_by(trc::location!())?;
let mut purged = Purged::default();
for chunk in due.chunks(PURGE_BATCH) {
let mut batch = BatchBuilder::new();
for (expires_at, id) in chunk {
let parts = [id.node, id.seq];
let Some(entry) = get(data, *id).await? else {
// Its entry is already gone: only the index is left
batch.clear(class(KIND_EXPIRY, &[*expires_at, id.node, id.seq]));
continue;
};
if held(&entry) {
purged.kept_for_hold += 1;
continue;
}
batch
.clear(class(KIND_CONTENT, &parts))
.clear(class(KIND_TIME, &[entry.at, id.node, id.seq]))
.clear(class(KIND_EXPIRY, &[*expires_at, id.node, id.seq]))
.set(class(KIND_PURGED, &parts), now.to_be_bytes().to_vec());
if let Some(blob) = entry.blob_hash() {
batch.clear(BlobOp::Link {
hash: blob,
to: BlobLink::Temporary { until: HELD_UNTIL },
});
}
purged.removed += 1;
}
if !batch.is_empty() {
data.write(batch.build_all())
.await
.caused_by(trc::location!())?;
}
}
for node in nodes(data).await? {
advance_floor(data, node).await?;
}
Ok(purged)
}
/// Clears the purged links at the start of a node's chain, recording where
/// it now starts and the hash that start names.
async fn advance_floor(data: &Store, node: u64) -> trc::Result<()> {
let start = floor(data, node).await?;
let mut cleared: Vec<u64> = Vec::new();
let mut next = start.clone();
let mut purged_seqs = Vec::new();
data.iterate(
IterateParams::new(
key(KIND_PURGED, &[node, start.seq]),
key(KIND_PURGED, &[node, u64::MAX]),
)
.ascending()
.no_values(),
|key, _| {
if let Some(parts) = parse_key(key, KIND_PURGED, 2) {
purged_seqs.push(parts[1]);
}
Ok(purged_seqs.len() < 100_000)
},
)
.await
.caused_by(trc::location!())?;
for seq in purged_seqs {
if seq != next.seq {
break;
}
let Some(Raw(link)) = data
.get_value::<Raw>(key(KIND_LINK, &[node, seq]))
.await
.caused_by(trc::location!())?
else {
break;
};
next = Floor {
seq: seq + 1,
prev: sha256(&link),
};
cleared.push(seq);
}
if cleared.is_empty() {
return Ok(());
}
// The floor moves first: a run cut short leaves links before it, which
// the next run clears, never a chain that looks broken
let mut batch = BatchBuilder::new();
batch.set(class(KIND_FLOOR, &[node]), Json(&next).serialize()?);
data.write(batch.build_all())
.await
.caused_by(trc::location!())?;
for chunk in cleared.chunks(PURGE_BATCH) {
let mut batch = BatchBuilder::new();
for seq in chunk {
batch
.clear(class(KIND_LINK, &[node, *seq]))
.clear(class(KIND_PURGED, &[node, *seq]));
}
data.write(batch.build_all())
.await
.caused_by(trc::location!())?;
}
Ok(())
}
/// One node's chain, as [`verify`] found it.
#[derive(Debug, Clone, PartialEq, Eq, SerdeSerialize)]
#[serde(rename_all = "camelCase")]
pub struct ChainReport {
pub node: u64,
pub entries: u64,
pub purged: u64,
pub first_seq: u64,
pub last_seq: u64,
#[serde(skip_serializing_if = "Option::is_none")]
pub broken_at: Option<String>,
#[serde(skip_serializing_if = "Option::is_none")]
pub reason: Option<String>,
}
/// Rechecks every node's chain (JR-6): each link names the hash of the one
/// before it, seqs run without gaps, the head matches the last link, each
/// entry hashes to what its link names or was purged, and, with `blobs`,
/// each report is there and hashes to what its entry names.
pub async fn verify(data: &Store, blobs: Option<&BlobStore>) -> trc::Result<Vec<ChainReport>> {
let mut reports = Vec::new();
for node in nodes(data).await? {
let start = floor(data, node).await?;
let head = head(data, node).await?.unwrap_or_default();
let mut report = ChainReport {
node,
entries: 0,
purged: 0,
first_seq: start.seq,
last_seq: start.seq.saturating_sub(1),
broken_at: None,
reason: None,
};
let mut links = Vec::new();
data.iterate(
IterateParams::new(
key(KIND_LINK, &[node, start.seq]),
key(KIND_LINK, &[node, u64::MAX]),
)
.ascending(),
|key, value| {
if let Some(parts) = parse_key(key, KIND_LINK, 2) {
links.push((parts[1], value.to_vec()));
}
Ok(true)
},
)
.await
.caused_by(trc::location!())?;
let mut expected_seq = start.seq;
let mut expected_prev = start.prev.clone();
for (seq, bytes) in links {
let broken = |report: &mut ChainReport, reason: &str| {
report.broken_at = Some(EntryId { node, seq }.to_string());
report.reason = Some(reason.to_string());
};
let Ok(Json(link)) = Json::<Link>::deserialize(&bytes) else {
broken(&mut report, "The link can't be read.");
break;
};
if seq != expected_seq || link.seq != seq {
report.broken_at = Some(EntryId { node, seq }.to_string());
report.reason = Some(format!(
"Entry {expected_seq} is missing; the next one found is {seq}."
));
break;
}
if link.prev != expected_prev {
broken(
&mut report,
"The link doesn't follow from the one before it: one of them was changed.",
);
break;
}
match data
.get_value::<Raw>(key(KIND_CONTENT, &[node, seq]))
.await
.caused_by(trc::location!())?
{
Some(Raw(content)) => {
if sha256(&content) != link.content {
broken(&mut report, "The entry was changed after it was written.");
break;
}
if let Some(blobs) = blobs {
let Ok(Json(entry)) = Json::<Entry>::deserialize(&content) else {
broken(&mut report, "The entry can't be read.");
break;
};
let report_bytes = match entry.blob_hash() {
Some(hash) => blobs
.get_blob(hash.as_slice(), 0..usize::MAX)
.await
.caused_by(trc::location!())?,
None => None,
};
match report_bytes {
Some(bytes) if sha256(&bytes) == entry.sha256 => {}
Some(_) => {
broken(&mut report, "The report doesn't match its entry.");
break;
}
None => {
broken(&mut report, "The report is missing.");
break;
}
}
}
report.entries += 1;
}
None => {
if data
.get_value::<Raw>(key(KIND_PURGED, &[node, seq]))
.await
.caused_by(trc::location!())?
.is_none()
{
broken(&mut report, "The entry was removed before its time.");
break;
}
report.purged += 1;
}
}
expected_prev = sha256(&bytes);
expected_seq = seq + 1;
report.last_seq = seq;
}
if report.broken_at.is_none()
&& (head.seq != report.last_seq
|| (report.last_seq >= report.first_seq && head.hash != expected_prev))
{
report.broken_at = Some(
EntryId {
node,
seq: report.last_seq,
}
.to_string(),
);
report.reason = Some(
"The chain's recorded end doesn't match its last link: entries were removed \
or changed at the end."
.into(),
);
}
reports.push(report);
}
Ok(reports)
}
#[cfg(test)]
mod tests {
use super::*;
#[test]
fn keys_read_back() {
let ValueClass::Any(any) = class(KIND_EXPIRY, &[5, 3, 9]) else {
panic!()
};
assert_eq!(parse_key(&any.key, KIND_EXPIRY, 3), Some(vec![5, 3, 9]));
let mut with_subspace = vec![SUBSPACE_INBUXA];
with_subspace.extend_from_slice(&any.key);
assert_eq!(
parse_key(&with_subspace, KIND_EXPIRY, 3),
Some(vec![5, 3, 9])
);
assert_eq!(parse_key(&any.key, KIND_TIME, 3), None);
}
#[test]
fn filters_match() {
let entry = Entry {
queue_id: 1,
at: 100,
direction: Direction::Outgoing,
sender: "[email protected]".into(),
authenticated: true,
recipients: vec!["[email protected]".into()],
subject: "Q3 figures, final".into(),
message_id: "<[email protected]>".into(),
accounts: vec![3],
tenants: vec![],
journals: vec![2],
held: false,
blob: String::new(),
size: 0,
sha256: String::new(),
expires_at: 0,
};
let yes = |f: Filter| assert!(f.matches(&entry), "{f:?}");
let no = |f: Filter| assert!(!f.matches(&entry), "{f:?}");
yes(Filter::default());
yes(Filter {
sender: Some("alice@".into()),
..Default::default()
});
yes(Filter {
address: Some("BANK".into()),
..Default::default()
});
yes(Filter {
text: Some("final q3".into()),
..Default::default()
});
yes(Filter {
message_id: Some("[email protected]".into()),
..Default::default()
});
yes(Filter {
direction: Some(Direction::Any),
..Default::default()
});
no(Filter {
direction: Some(Direction::Incoming),
..Default::default()
});
no(Filter {
recipient: Some("alice".into()),
..Default::default()
});
no(Filter {
before: Some(100),
..Default::default()
});
yes(Filter {
after: Some(100),
journal_id: Some(2),
..Default::default()
});
no(Filter {
journal_id: Some(5),
..Default::default()
});
}
#[test]
fn hex_round_trips() {
let bytes = [0u8, 1, 0xab, 0xff];
assert_eq!(unhex(&hex(&bytes)), Some(bytes.to_vec()));
assert_eq!(unhex("abc"), None);
assert_eq!(unhex("zz"), None);
}
#[test]
fn ids_read_back() {
let id = EntryId { node: 3, seq: 77 };
assert_eq!(EntryId::from_u64(id.to_u64()), id);
assert_eq!(id.to_string(), "3-77");
}
}
-512
View File
@@ -1,512 +0,0 @@
/*
* SPDX-FileCopyrightText: 2026 Coffey Labs LLC
*
* SPDX-License-Identifier: AGPL-3.0-only
*/
//! Journaling (journaling spec, JR-1 to JR-18): a copy of each message the
//! server queues, with its envelope, kept where nothing in the product
//! changes or removes it before its retention ends.
//!
//! - this module: journals, what makes one valid, and where they're kept;
//! - [`report`]: the journal report around the untouched message (JR-3);
//! - [`entries`]: the built-in journal and its chain (JR-5, JR-6, JR-13).
//!
//! Kept in the fork's subspace (`store::SUBSPACE_INBUXA`). Every key starts
//! with `J`; journals are `j` + id (u32), as JSON. There are few, so they're
//! read whole.
pub mod archive;
pub mod entries;
pub mod report;
use crate::{hold::Member, mailflow::rules::jmap_ids};
use serde::{Deserialize as SerdeDeserialize, Serialize as SerdeSerialize, de::DeserializeOwned};
use std::{
sync::{Arc, RwLock},
time::{Duration, Instant},
};
use store::{
Deserialize, IterateParams, SUBSPACE_INBUXA, Serialize, Store, ValueKey,
write::{AnyClass, BatchBuilder, ValueClass, assert::AssertValue},
};
use trc::AddContext;
pub(crate) const FEATURE: u8 = b'J';
const KIND_JOURNAL: u8 = b'j';
const CREATE_ATTEMPTS: usize = 5;
/// Retention a journal may be given, in days (settled answer 3).
pub const MIN_RETENTION_DAYS: u32 = 30;
pub const MAX_RETENTION_DAYS: u32 = 3650;
/// Most entries in one scope list.
const MAX_LIST: usize = 5_000;
/// Which way a message goes, from this server's side (JR-9).
#[derive(Debug, Clone, Copy, PartialEq, Eq, SerdeSerialize, SerdeDeserialize)]
#[serde(rename_all = "camelCase")]
pub enum Direction {
/// From someone here to at least one recipient elsewhere.
Outgoing,
/// From elsewhere to someone here.
Incoming,
/// From someone here, to people here only.
Internal,
Any,
}
impl Direction {
pub fn as_str(&self) -> &'static str {
match self {
Direction::Outgoing => "outgoing",
Direction::Incoming => "incoming",
Direction::Internal => "internal",
Direction::Any => "any",
}
}
/// A message's direction: `Any` is never one.
pub fn of(sender_local: bool, any_remote: bool, any_local: bool) -> Direction {
match (sender_local, any_remote) {
(true, true) => Direction::Outgoing,
(true, false) => Direction::Internal,
(false, _) if any_local => Direction::Incoming,
// Nobody here on either side: relayed mail counts as outgoing
(false, _) => Direction::Outgoing,
}
}
fn includes(&self, direction: Direction) -> bool {
*self == Direction::Any || *self == direction
}
}
/// Whose mail a journal takes (JR-9): everyone, or people reached through
/// their account, domain, group or tenant. Ids are in the JMAP form.
#[derive(Debug, Clone, Default, PartialEq, Eq, SerdeSerialize, SerdeDeserialize)]
#[serde(rename_all = "camelCase")]
pub struct Scope {
#[serde(default)]
pub everyone: bool,
#[serde(default, with = "jmap_ids")]
pub accounts: Vec<u32>,
#[serde(default, with = "jmap_ids")]
pub groups: Vec<u32>,
#[serde(default, with = "jmap_ids")]
pub domains: Vec<u32>,
#[serde(default, with = "jmap_ids")]
pub tenants: Vec<u32>,
}
impl Scope {
fn lists(&self) -> [&Vec<u32>; 4] {
[&self.accounts, &self.groups, &self.domains, &self.tenants]
}
/// Whether this scope reaches one person here.
pub fn covers(&self, member: &Member) -> bool {
self.everyone
|| self.accounts.contains(&member.account)
|| member.domains.iter().any(|d| self.domains.contains(d))
|| member.groups.iter().any(|g| self.groups.contains(g))
|| member.tenant.is_some_and(|t| self.tenants.contains(&t))
}
}
/// A journal (JR-9): what it takes, and how long its entries are kept.
#[derive(Debug, Clone, PartialEq, Eq, SerdeSerialize, SerdeDeserialize)]
#[serde(rename_all = "camelCase")]
pub struct Journal {
#[serde(default)]
pub id: u32,
pub name: String,
#[serde(default)]
pub description: String,
#[serde(default)]
pub enabled: bool,
pub direction: Direction,
pub scope: Scope,
/// How long an entry this journal writes is kept. An entry keeps the
/// retention it was written with (JR-12).
pub retention_days: u32,
/// Whether entries go into the built-in journal (JR-5).
#[serde(default = "yes")]
pub built_in: bool,
/// An outside archive's journal address, sent each report (JR-7).
#[serde(default, skip_serializing_if = "Option::is_none")]
pub archive_address: Option<String>,
#[serde(default)]
pub created_by: String,
#[serde(default)]
pub created_at: u64,
#[serde(default)]
pub updated_at: u64,
}
/// Why a journal was refused: the property, and what to do.
#[derive(Debug, Clone, PartialEq, Eq)]
pub struct Invalid {
pub property: &'static str,
pub reason: String,
}
fn invalid(property: &'static str, reason: impl Into<String>) -> Result<(), Invalid> {
Err(Invalid {
property,
reason: reason.into(),
})
}
impl Journal {
pub fn validate(&self) -> Result<(), Invalid> {
if self.name.trim().is_empty() {
return invalid("name", "Give the journal a name.");
}
if self.name.len() > 200 || self.description.len() > 2_000 {
return invalid("name", "The name or description is too long.");
}
if !(MIN_RETENTION_DAYS..=MAX_RETENTION_DAYS).contains(&self.retention_days) {
return invalid(
"retentionDays",
format!("Keep entries between {MIN_RETENTION_DAYS} and {MAX_RETENTION_DAYS} days."),
);
}
// Neither is a journal only rules send mail to (JR-10)
let chosen = self.scope.lists().iter().any(|list| !list.is_empty());
if self.scope.everyone && chosen {
return invalid(
"scope",
"Journal everyone, or choose accounts, groups, domains or tenants; not both.",
);
}
if !self.built_in && self.archive_address.is_none() {
return invalid(
"builtIn",
"Keep entries in the built-in journal, send them to an archive, or both.",
);
}
if let Some(address) = &self.archive_address
&& !is_address(address)
{
return invalid(
"archiveAddress",
format!("\"{address}\" isn't an email address."),
);
}
if self.scope.lists().iter().any(|list| list.len() > MAX_LIST) {
return invalid("scope", format!("Choose at most {MAX_LIST} of each."));
}
Ok(())
}
/// Whether this journal takes a message going `direction` with these
/// people here on either side.
/// Whether only rules send this journal mail (JR-10).
pub fn rules_only(&self) -> bool {
!self.scope.everyone && self.scope.lists().iter().all(|list| list.is_empty())
}
pub fn takes(&self, direction: Direction, members: &[Member]) -> bool {
self.enabled
&& self.direction.includes(direction)
&& (self.scope.everyone || members.iter().any(|m| self.scope.covers(m)))
}
}
fn yes() -> bool {
true
}
/// An address an archive can be sent to: one `@`, something either side,
/// nothing that would break an envelope.
fn is_address(address: &str) -> bool {
address.len() <= 320
&& address.split_once('@').is_some_and(|(local, domain)| {
!local.is_empty() && domain.contains('.') && !domain.contains('@')
})
&& !address
.chars()
.any(|c| c.is_whitespace() || c.is_control() || matches!(c, '<' | '>' | ',' | ';'))
}
/// A value stored as JSON.
pub(crate) struct Json<T>(pub T);
impl<T: SerdeSerialize> Serialize for Json<T> {
fn serialize(&self) -> trc::Result<Vec<u8>> {
serde_json::to_vec(&self.0).map_err(|err| {
trc::StoreEvent::UnexpectedError
.into_err()
.details("Failed to serialize a journal record")
.reason(err)
})
}
}
impl<T: DeserializeOwned + Sync + Send> Deserialize for Json<T> {
fn deserialize(bytes: &[u8]) -> trc::Result<Self> {
serde_json::from_slice(bytes).map(Json).map_err(|err| {
trc::StoreEvent::DataCorruption
.into_err()
.details("Invalid journal record")
.reason(err)
})
}
}
fn class(id: u32) -> ValueClass {
let mut key = Vec::with_capacity(6);
key.push(FEATURE);
key.push(KIND_JOURNAL);
key.extend_from_slice(&id.to_be_bytes());
ValueClass::Any(AnyClass {
subspace: SUBSPACE_INBUXA,
key,
})
}
fn key(id: u32) -> ValueKey<ValueClass> {
ValueKey::from(class(id))
}
pub async fn get(data: &Store, id: u32) -> trc::Result<Option<Journal>> {
Ok(data
.get_value::<Json<Journal>>(key(id))
.await
.caused_by(trc::location!())?
.map(|Json(journal)| journal))
}
/// Every journal, oldest first.
pub async fn all(data: &Store) -> trc::Result<Vec<Journal>> {
let mut journals = Vec::new();
data.iterate(IterateParams::new(key(0), key(u32::MAX)), |_, value| {
if let Ok(Json(journal)) = Json::<Journal>::deserialize(value) {
journals.push(journal);
}
Ok(true)
})
.await
.caused_by(trc::location!())?;
journals.sort_by_key(|journal| journal.id);
Ok(journals)
}
/// Writes a new journal under the next free id, which it returns.
pub async fn create(data: &Store, journal: &Journal) -> trc::Result<u32> {
let mut attempt = 0;
loop {
attempt += 1;
let id = all(data).await?.iter().map(|j| j.id).max().unwrap_or(0) + 1;
let stored = Journal {
id,
..journal.clone()
};
let mut batch = BatchBuilder::new();
batch.assert_value(class(id), AssertValue::None);
batch.set(class(id), Json(&stored).serialize()?);
match data.write(batch.build_all()).await {
Ok(_) => {
invalidate();
return Ok(id);
}
Err(err)
if attempt < CREATE_ATTEMPTS
&& matches!(
err.as_ref(),
trc::EventType::Store(trc::StoreEvent::AssertValueFailed)
) => {}
Err(err) => return Err(err.caused_by(trc::location!())),
}
}
}
/// Replaces a stored journal (same id).
pub async fn update(data: &Store, journal: &Journal) -> trc::Result<()> {
let mut batch = BatchBuilder::new();
batch.set(class(journal.id), Json(journal).serialize()?);
data.write(batch.build_all())
.await
.caused_by(trc::location!())?;
invalidate();
Ok(())
}
/// Removes a journal. Its entries stay, each until its own time.
pub async fn delete(data: &Store, id: u32) -> trc::Result<()> {
let mut batch = BatchBuilder::new();
batch.clear(class(id));
data.write(batch.build_all())
.await
.caused_by(trc::location!())?;
invalidate();
Ok(())
}
/// How long a node keeps its copy of the journals before reading them again.
pub const TTL: Duration = Duration::from_secs(30);
type Cached = Option<(Instant, Arc<Vec<Journal>>)>;
static CACHE: RwLock<Cached> = RwLock::new(None);
/// Forgets this node's copy, so the next message reads the journals again.
pub fn invalidate() {
if let Ok(mut cache) = CACHE.write() {
*cache = None;
}
}
/// The enabled journals, from this node's copy (refreshed every [`TTL`]).
pub async fn enabled(data: &Store) -> trc::Result<Arc<Vec<Journal>>> {
if let Ok(cache) = CACHE.read()
&& let Some((at, journals)) = cache.as_ref()
&& at.elapsed() < TTL
{
return Ok(journals.clone());
}
let journals = Arc::new(
all(data)
.await?
.into_iter()
.filter(|journal| journal.enabled)
.collect::<Vec<_>>(),
);
if let Ok(mut cache) = CACHE.write() {
*cache = Some((Instant::now(), journals.clone()));
}
Ok(journals)
}
#[cfg(test)]
mod tests {
use super::*;
fn journal(scope: Scope) -> Journal {
Journal {
id: 1,
name: "Finance".into(),
description: String::new(),
enabled: true,
direction: Direction::Any,
scope,
retention_days: 365,
built_in: true,
archive_address: None,
created_by: String::new(),
created_at: 0,
updated_at: 0,
}
}
fn member(account: u32, groups: Vec<u32>) -> Member {
Member {
account,
domains: vec![1],
groups,
tenant: None,
}
}
#[test]
fn scope_is_everyone_or_chosen() {
assert!(
journal(Scope {
everyone: true,
..Default::default()
})
.validate()
.is_ok()
);
// Nobody chosen: only rules send it mail
let rules_only = journal(Scope::default());
assert!(rules_only.validate().is_ok());
assert!(rules_only.rules_only());
assert!(!rules_only.takes(Direction::Any, &[member(3, vec![7])]));
let both = Scope {
everyone: true,
groups: vec![4],
..Default::default()
};
assert_eq!(journal(both).validate().unwrap_err().property, "scope");
}
#[test]
fn destinations() {
let mut j = journal(Scope {
everyone: true,
..Default::default()
});
j.built_in = false;
assert_eq!(j.validate().unwrap_err().property, "builtIn");
j.archive_address = Some("[email protected]".into());
assert!(j.validate().is_ok());
for bad in [
"archive",
"a@b",
"a [email protected]",
"<[email protected]>",
"a@[email protected]",
] {
j.archive_address = Some(bad.into());
assert_eq!(
j.validate().unwrap_err().property,
"archiveAddress",
"{bad}"
);
}
// Stored before destinations existed: the built-in journal
let old: Journal = serde_json::from_str(
r#"{"name":"Old","direction":"any","scope":{"everyone":true},"retentionDays":30}"#,
)
.unwrap();
assert!(old.built_in && old.archive_address.is_none());
}
#[test]
fn retention_has_bounds() {
let mut j = journal(Scope {
everyone: true,
..Default::default()
});
j.retention_days = 29;
assert_eq!(j.validate().unwrap_err().property, "retentionDays");
j.retention_days = 3651;
assert!(j.validate().is_err());
j.retention_days = 3650;
assert!(j.validate().is_ok());
}
#[test]
fn takes_by_direction_and_member() {
let mut j = journal(Scope {
groups: vec![7],
..Default::default()
});
assert!(j.takes(Direction::Outgoing, &[member(3, vec![7])]));
assert!(!j.takes(Direction::Outgoing, &[member(3, vec![8])]));
assert!(!j.takes(Direction::Outgoing, &[]));
j.direction = Direction::Incoming;
assert!(!j.takes(Direction::Outgoing, &[member(3, vec![7])]));
j.enabled = false;
assert!(!j.takes(Direction::Incoming, &[member(3, vec![7])]));
}
#[test]
fn directions() {
assert_eq!(Direction::of(true, true, true), Direction::Outgoing);
assert_eq!(Direction::of(true, false, true), Direction::Internal);
assert_eq!(Direction::of(false, false, true), Direction::Incoming);
assert_eq!(Direction::of(false, true, true), Direction::Incoming);
}
#[test]
fn scope_ids_are_jmap_ids() {
let scope: Scope = serde_json::from_str(r#"{"groups":["b"],"tenants":[7]}"#).unwrap();
assert_eq!(scope.groups, vec![1]);
assert_eq!(scope.tenants, vec![7]);
assert_eq!(
serde_json::to_value(&scope).unwrap()["tenants"],
serde_json::json!(["h"])
);
}
}
-385
View File
@@ -1,385 +0,0 @@
/*
* SPDX-FileCopyrightText: 2026 Coffey Labs LLC
*
* SPDX-License-Identifier: AGPL-3.0-only
*/
//! The journal report (JR-3, JR-4): a message whose first part lists the
//! envelope, one field a line, and whose second part is the message as it
//! was queued, byte for byte, as `message/rfc822`. Field names are fixed
//! English: a report is a record, and scripts read it.
use super::Direction;
use mail_builder::headers::{Header, date::Date, text::Text};
use mail_parser::MessageParser;
use sha2::{Digest, Sha256};
/// One envelope recipient, with the address it was given as (a list's, for
/// the list's members).
#[derive(Debug, Clone, PartialEq, Eq)]
pub struct Recipient {
pub address: String,
pub orcpt: Option<String>,
/// The mail flow rule that added or redirected to it.
pub added_by: Option<String>,
}
/// What the queue knows about a message.
#[derive(Debug, Clone)]
pub struct Envelope<'x> {
pub sender: &'x str,
pub authenticated: bool,
pub recipients: &'x [Recipient],
pub queue_id: u64,
/// Seconds.
pub received: u64,
pub direction: Direction,
pub held: bool,
}
/// What a report says, besides the envelope's own fields.
#[derive(Debug, Clone, Default, PartialEq, Eq)]
pub struct Fields {
pub subject: String,
pub message_id: String,
pub to: Vec<String>,
pub cc: Vec<String>,
/// Envelope recipients in neither To nor Cc, nor reached through a list.
pub bcc: Vec<String>,
/// A list's address, and its members among the recipients.
pub expanded: Vec<(String, Vec<String>)>,
/// A rule's name, and the recipients it added.
pub added: Vec<(String, Vec<String>)>,
}
/// One line's worth of a value: no line breaks, no control characters.
fn line(value: &str) -> String {
value
.chars()
.map(|c| if c.is_control() { ' ' } else { c })
.collect::<String>()
.trim()
.to_string()
}
/// The address an ORCPT names, without its `rfc822;` type.
fn orcpt_address(orcpt: &str) -> String {
let orcpt = orcpt.trim();
let bare = match orcpt.split_once(';') {
Some((kind, address)) if kind.eq_ignore_ascii_case("rfc822") => address,
_ => orcpt,
};
bare.trim().to_lowercase()
}
/// Sorts the envelope's recipients by how they were addressed.
pub fn fields(envelope: &Envelope<'_>, original: &[u8]) -> Fields {
let parsed = MessageParser::default().parse_headers(original);
let headed = |which: Option<&mail_parser::Address<'_>>| -> Vec<String> {
which
.map(|list| {
list.iter()
.filter_map(|addr| addr.address())
.map(|address| address.to_lowercase())
.collect()
})
.unwrap_or_default()
};
let (subject, message_id, header_to, header_cc) = match &parsed {
Some(message) => (
message.subject().map(line).unwrap_or_default(),
message
.message_id()
.map(|id| format!("<{}>", line(id)))
.unwrap_or_default(),
headed(message.to()),
headed(message.cc()),
),
None => Default::default(),
};
let mut fields = Fields {
subject,
message_id,
..Default::default()
};
for rcpt in envelope.recipients {
let address = rcpt.address.to_lowercase();
let via = rcpt
.orcpt
.as_deref()
.map(orcpt_address)
.filter(|via| !via.is_empty() && *via != address);
if let Some(rule) = &rcpt.added_by {
match fields.added.iter_mut().find(|(name, _)| name == rule) {
Some((_, added)) => added.push(line(&rcpt.address)),
None => fields.added.push((line(rule), vec![line(&rcpt.address)])),
}
} else if header_to.contains(&address) {
fields.to.push(line(&rcpt.address));
} else if header_cc.contains(&address) {
fields.cc.push(line(&rcpt.address));
} else if let Some(via) = via {
match fields.expanded.iter_mut().find(|(list, _)| *list == via) {
Some((_, members)) => members.push(line(&rcpt.address)),
None => fields
.expanded
.push((line(&via), vec![line(&rcpt.address)])),
}
} else {
fields.bcc.push(line(&rcpt.address));
}
}
fields
}
/// The report's first part.
pub fn text(envelope: &Envelope<'_>, fields: &Fields) -> String {
let mut out = String::new();
let mut field = |name: &str, value: &str| {
if !value.is_empty() {
out.push_str(name);
out.push_str(": ");
out.push_str(value);
out.push_str("\r\n");
}
};
let sender = if envelope.sender.is_empty() {
"<>".to_string()
} else {
line(envelope.sender)
};
field("Sender", &sender);
field(
"Authenticated",
if envelope.authenticated { "yes" } else { "no" },
);
field("Subject", &fields.subject);
field("Message-ID", &fields.message_id);
field("Queue ID", &format!("{:x}", envelope.queue_id));
field(
"Received",
&mail_parser::DateTime::from_timestamp(envelope.received as i64).to_rfc3339(),
);
field("Direction", envelope.direction.as_str());
field("To", &fields.to.join(", "));
field("Cc", &fields.cc.join(", "));
field("Bcc", &fields.bcc.join(", "));
for (list, members) in &fields.expanded {
field("Expanded", &format!("{list} -> {}", members.join(", ")));
}
for (rule, added) in &fields.added {
field("Added by rule", &format!("{rule} -> {}", added.join(", ")));
}
if envelope.held {
field("Held for review", "yes");
}
out
}
fn hex(bytes: &[u8]) -> String {
bytes.iter().map(|b| format!("{b:02x}")).collect()
}
/// Whether a message can travel as 8bit: no NULs, no line past 998 bytes.
fn fits_8bit(message: &[u8]) -> bool {
!message.contains(&0) && message.split(|b| *b == b'\n').all(|l| l.len() <= 998)
}
/// The whole report: headers, the fields, then the original untouched.
/// `from` is the address the report is from; `host` names the server in its
/// Message-ID.
pub fn build(
envelope: &Envelope<'_>,
original: &[u8],
from: &str,
host: &str,
) -> (Vec<u8>, Fields) {
let fields = fields(envelope, original);
let body = text(envelope, &fields);
// A boundary that can't occur in the original
let mut boundary = format!("journal-{}", &hex(&Sha256::digest(original))[..32]);
while original
.windows(boundary.len())
.any(|window| window == boundary.as_bytes())
{
boundary.push('x');
}
let mut out: Vec<u8> = Vec::with_capacity(original.len() + body.len() + 1024);
out.extend_from_slice(format!("From: Journal <{}>\r\n", line(from)).as_bytes());
out.extend_from_slice(b"Date: ");
out.extend_from_slice(Date::new(envelope.received as i64).to_rfc822().as_bytes());
out.extend_from_slice(b"\r\n");
out.extend_from_slice(b"Subject: ");
let subject = if fields.subject.is_empty() {
"Journal report".to_string()
} else {
format!("Journal report: {}", fields.subject)
};
Text::new(subject).write_header(&mut out, "Subject: ".len());
out.extend_from_slice(
format!(
"Message-ID: <journal.{:x}.{}@{}>\r\n",
envelope.queue_id,
envelope.received,
line(host)
)
.as_bytes(),
);
out.extend_from_slice(format!("X-Inbuxa-Journal: {:x}\r\n", envelope.queue_id).as_bytes());
out.extend_from_slice(b"MIME-Version: 1.0\r\n");
out.extend_from_slice(
format!("Content-Type: multipart/mixed; boundary=\"{boundary}\"\r\n\r\n").as_bytes(),
);
out.extend_from_slice(format!("--{boundary}\r\n").as_bytes());
out.extend_from_slice(
b"Content-Type: text/plain; charset=utf-8\r\nContent-Transfer-Encoding: 8bit\r\n\r\n",
);
out.extend_from_slice(body.as_bytes());
out.extend_from_slice(format!("\r\n--{boundary}\r\n").as_bytes());
out.extend_from_slice(b"Content-Type: message/rfc822\r\n");
out.extend_from_slice(b"Content-Disposition: attachment; filename=\"original.eml\"\r\n");
out.extend_from_slice(if fits_8bit(original) {
b"Content-Transfer-Encoding: 8bit\r\n\r\n".as_slice()
} else {
b"Content-Transfer-Encoding: binary\r\n\r\n".as_slice()
});
out.extend_from_slice(original);
// The line break before a boundary belongs to the boundary: the
// original keeps its own last one
out.extend_from_slice(format!("\r\n--{boundary}--\r\n").as_bytes());
(out, fields)
}
/// Where the original starts and ends inside a report [`build`] made.
pub fn original(report: &[u8]) -> Option<&[u8]> {
let parsed = MessageParser::default().parse(report)?;
let part = parsed.attachment(0)?;
let start = part.raw_body_offset() as usize;
let end = part.raw_end_offset() as usize;
report.get(start..end)
}
#[cfg(test)]
mod tests {
use super::*;
const ORIGINAL: &[u8] = b"From: [email protected]\r\n\
To: Bank <pay@bank.example>\r\n\
Cc: bob@example.com\r\n\
Subject: Q3 figures\r\n\
Message-ID: <abc@example.com>\r\n\
\r\n\
The figures.\r\n";
fn rcpt(address: &str, orcpt: Option<&str>) -> Recipient {
Recipient {
address: address.into(),
orcpt: orcpt.map(Into::into),
added_by: None,
}
}
fn envelope(recipients: &[Recipient]) -> Envelope<'_> {
Envelope {
sender: "[email protected]",
authenticated: true,
recipients,
queue_id: 0x1a2b,
received: 1_790_000_000,
direction: Direction::Outgoing,
held: false,
}
}
#[test]
fn recipients_sorted_by_how_they_were_addressed() {
let recipients = [
rcpt("[email protected]", None),
rcpt("[email protected]", Some("rfc822;[email protected]")),
rcpt("[email protected]", None),
rcpt("[email protected]", Some("[email protected]")),
rcpt("[email protected]", Some("rfc822;[email protected]")),
];
let fields = fields(&envelope(&recipients), ORIGINAL);
assert_eq!(fields.subject, "Q3 figures");
assert_eq!(fields.message_id, "<[email protected]>");
assert_eq!(fields.to, vec!["[email protected]"]);
assert_eq!(fields.cc, vec!["[email protected]"]);
assert_eq!(fields.bcc, vec!["[email protected]"]);
assert_eq!(
fields.expanded,
vec![(
"[email protected]".to_string(),
vec![
"[email protected]".to_string(),
"[email protected]".to_string()
]
)]
);
}
#[test]
fn report_carries_the_original_untouched() {
let recipients = [
rcpt("[email protected]", None),
rcpt("[email protected]", None),
];
let (report, _) = build(
&envelope(&recipients),
ORIGINAL,
"[email protected]",
"mx.example.com",
);
let text = String::from_utf8_lossy(&report);
assert!(text.contains("Sender: [email protected]\r\n"));
assert!(text.contains("Bcc: [email protected]\r\n"));
assert!(text.contains("Queue ID: 1a2b\r\n"));
assert!(text.contains("Direction: outgoing\r\n"));
assert!(text.contains("Subject: Journal report: Q3 figures\r\n"));
assert!(!text.contains("Held for review"));
assert_eq!(original(&report), Some(ORIGINAL));
let unterminated = &ORIGINAL[..ORIGINAL.len() - 2];
let (report, _) = build(
&envelope(&recipients),
unterminated,
"[email protected]",
"mx.example.com",
);
assert_eq!(original(&report), Some(unterminated));
}
#[test]
fn rule_added_recipients_say_so() {
let mut copied = rcpt("[email protected]", None);
copied.added_by = Some("Copy finance".into());
let recipients = [rcpt("[email protected]", None), copied];
let env = envelope(&recipients);
let fields = fields(&env, ORIGINAL);
assert!(fields.bcc.is_empty(), "{fields:?}");
assert!(
text(&env, &fields).contains("Added by rule: Copy finance -> [email protected]\r\n")
);
}
#[test]
fn values_stay_on_one_line() {
let recipients = [rcpt("[email protected]", None)];
let mut env = envelope(&recipients);
env.sender = "[email protected]\r\nBcc: [email protected]";
env.held = true;
let body = text(&env, &Fields::default());
assert_eq!(body.matches("\r\n").count(), body.lines().count());
assert!(body.contains("Sender: [email protected] Bcc: [email protected]\r\n"));
assert!(body.contains("Held for review: yes\r\n"));
}
#[test]
fn an_empty_sender_is_shown_as_such() {
let recipients = [rcpt("[email protected]", None)];
let mut env = envelope(&recipients);
env.sender = "";
assert!(text(&env, &Fields::default()).starts_with("Sender: <>\r\n"));
}
}
+1 -4
View File
@@ -1,5 +1,5 @@
/*
* SPDX-FileCopyrightText: 2026 Coffey Labs LLC
* SPDX-FileCopyrightText: 2026 Coffey Labs
*
* SPDX-License-Identifier: AGPL-3.0-only
*/
@@ -21,11 +21,8 @@
pub mod ai;
pub mod audit;
pub mod branding;
pub mod deliverability; // inbuxa: the deliverability check (not a rebuild)
pub mod hold;
pub mod journal;
pub mod lock;
pub mod mailflow;
pub mod masked_email;
pub mod privacy;
pub mod security;
+4 -73
View File
@@ -1,5 +1,5 @@
/*
* SPDX-FileCopyrightText: 2026 Coffey Labs LLC
* SPDX-FileCopyrightText: 2026 Coffey Labs
*
* SPDX-License-Identifier: AGPL-3.0-only
*/
@@ -66,53 +66,6 @@ const KIND_DELEGATE: u8 = b'd';
/// Most delegates one lock may have (AL-5).
pub const MAX_DELEGATES: usize = 10;
/// Most people one shared mailbox may have (MA-S): a help desk is bigger
/// than the handful a departed colleague's mail is handed to.
pub const MAX_SHARED_MAILBOX_DELEGATES: usize = 100;
/// What a lock is for (multi-account spec, MA-S).
///
/// Both kinds keep receiving mail, can't be signed in to, and are opened by
/// delegates through real grants. A shared mailbox is a role address such
/// as support@: it needs no reason, holds more people, runs its own Sieve
/// replies (an automatic acknowledgement), records only what is sent as it,
/// and may only send as its own addresses.
#[derive(Debug, Clone, Copy, Default, PartialEq, Eq, Hash, SerdeSerialize, SerdeDeserialize)]
#[serde(rename_all = "camelCase")]
pub enum Kind {
#[default]
Lock,
SharedMailbox,
}
impl Kind {
pub fn as_str(&self) -> &'static str {
match self {
Kind::Lock => "lock",
Kind::SharedMailbox => "sharedMailbox",
}
}
pub fn parse(value: &str) -> Option<Self> {
match value {
"lock" => Some(Kind::Lock),
"sharedMailbox" => Some(Kind::SharedMailbox),
_ => None,
}
}
pub fn is_lock(&self) -> bool {
matches!(self, Kind::Lock)
}
pub fn max_delegates(&self) -> usize {
match self {
Kind::Lock => MAX_DELEGATES,
Kind::SharedMailbox => MAX_SHARED_MAILBOX_DELEGATES,
}
}
}
/// What a delegate may do in the locked account (AL-6).
#[derive(Debug, Clone, Copy, PartialEq, Eq, Hash, SerdeSerialize, SerdeDeserialize)]
#[serde(rename_all = "camelCase")]
@@ -236,9 +189,6 @@ pub struct Replaced {
#[serde(rename_all = "camelCase")]
pub struct Lock {
pub account_id: u32,
/// Absent on locks written before shared mailboxes existed: a lock.
#[serde(default, skip_serializing_if = "Kind::is_lock")]
pub kind: Kind,
pub reason: String,
/// Seconds since the epoch.
pub locked_at: u64,
@@ -451,9 +401,8 @@ pub async fn all(data: &Store) -> trc::Result<Vec<Lock>> {
Ok(locks)
}
/// The accounts delegated to `delegate`, with its delegation in each and
/// the kind of lock it is in.
pub async fn delegated_to(data: &Store, delegate: u32) -> trc::Result<Vec<(u32, Delegate, Kind)>> {
/// The accounts delegated to `delegate`, with its delegation in each.
pub async fn delegated_to(data: &Store, delegate: u32) -> trc::Result<Vec<(u32, Delegate)>> {
let mut locked = Vec::new();
data.iterate(
IterateParams::new(
@@ -476,7 +425,7 @@ pub async fn delegated_to(data: &Store, delegate: u32) -> trc::Result<Vec<(u32,
if let Some(lock) = get(data, account_id).await?
&& let Some(delegation) = lock.delegate(delegate)
{
delegations.push((account_id, delegation.clone(), lock.kind));
delegations.push((account_id, delegation.clone()));
}
}
Ok(delegations)
@@ -523,22 +472,6 @@ pub async fn remove(data: &Store, lock: &Lock) -> trc::Result<()> {
mod tests {
use super::*;
#[test]
fn kind_reads_back_and_defaults_to_lock() {
// MA-S: a lock stored before shared mailboxes existed has no kind
let stored = r#"{"accountId":1,"reason":"r","lockedAt":0,"lockedBy":"admin","delegates":[]}"#;
let lock: Lock = serde_json::from_str(stored).unwrap();
assert_eq!(lock.kind, Kind::Lock);
assert!(!serde_json::to_string(&lock).unwrap().contains("kind"), "a lock is written as before");
let shared = Lock { kind: Kind::SharedMailbox, ..lock };
let written = serde_json::to_string(&shared).unwrap();
assert!(written.contains(r#""kind":"sharedMailbox""#), "{written}");
assert_eq!(serde_json::from_str::<Lock>(&written).unwrap().kind, Kind::SharedMailbox);
assert_eq!(Kind::parse("sharedMailbox"), Some(Kind::SharedMailbox));
assert_eq!(Kind::SharedMailbox.max_delegates(), MAX_SHARED_MAILBOX_DELEGATES);
}
#[test]
fn keys_read_back() {
let ValueClass::Any(any) = class(KIND_DELEGATE, &[7, 9]) else {
@@ -576,7 +509,6 @@ mod tests {
fn lock_with(delegates: Vec<Delegate>, replaced: Vec<Replaced>) -> Lock {
Lock {
account_id: 1,
kind: Kind::Lock,
reason: "r".into(),
locked_at: 0,
locked_by: "admin".into(),
@@ -691,7 +623,6 @@ mod tests {
fn expired_delegations_grant_nothing() {
let lock = Lock {
account_id: 1,
kind: Kind::Lock,
reason: "Left the company".into(),
locked_at: 100,
locked_by: "admin".into(),
-53
View File
@@ -1,53 +0,0 @@
/*
* SPDX-FileCopyrightText: 2026 Coffey Labs LLC
*
* SPDX-License-Identifier: AGPL-3.0-only
*/
//! The compiled rules, kept per node so a message doesn't read the store.
//! A change made on this node applies at once; one made on another node
//! within [`TTL`], when the copy here is next refreshed.
use super::{engine::Compiled, rules};
use std::{
sync::{Arc, RwLock},
time::{Duration, Instant},
};
use store::Store;
/// How long a node keeps its copy before reading the rules again.
pub const TTL: Duration = Duration::from_secs(30);
static CACHE: RwLock<Option<(Instant, Arc<Compiled>)>> = RwLock::new(None);
/// Forgets the copy, so the next message reads the rules again.
pub fn invalidate() {
if let Ok(mut cache) = CACHE.write() {
*cache = None;
}
}
/// The enabled rules, compiled. A rule that no longer compiles is left out
/// and reported, once per refresh.
pub async fn compiled(data: &Store) -> trc::Result<Arc<Compiled>> {
if let Ok(cache) = CACHE.read()
&& let Some((at, compiled)) = cache.as_ref()
&& at.elapsed() < TTL
{
return Ok(compiled.clone());
}
let (compiled, skipped) = Compiled::new(&rules::all(data).await?);
for (id, reason) in skipped {
trc::event!(
Store(trc::StoreEvent::DataCorruption),
Id = u64::from(id),
Reason = reason,
Details = "Mail rule skipped: it no longer compiles"
);
}
let compiled = Arc::new(compiled);
if let Ok(mut cache) = CACHE.write() {
*cache = Some((Instant::now(), compiled.clone()));
}
Ok(compiled)
}
@@ -1,49 +0,0 @@
/*
* SPDX-FileCopyrightText: 2026 Coffey Labs LLC
*
* SPDX-License-Identifier: AGPL-3.0-only
*/
//! African identifiers (§2.3): South Africa's ID number.
use super::{Detector, Findings, Region, Strength, checks, valid_short_date};
use regex::Regex;
use std::sync::LazyLock;
pub static DETECTORS: &[Detector] = &[Detector::new(
"za-id",
"South Africa: ID number",
Region::Africa,
Strength::Checked,
za_id,
)];
/// Birth date `YYMMDD`, four digits, citizenship (0, 1 or 2), 8 or 9, a Luhn
/// check digit. The date and the two fixed digits make it strong enough to
/// count alone.
static ZA_ID: LazyLock<Regex> = LazyLock::new(|| {
Regex::new(r"\b(\d{2})(\d{2})(\d{2})\d{4}[012][89]\d\b").expect("detector pattern")
});
fn za_id(text: &str, findings: &mut Findings) {
for c in ZA_ID.captures_iter(text) {
let n = &c[0];
let num = |s: &str| s.parse::<u32>().unwrap_or(0);
if valid_short_date(num(&c[1]), num(&c[2]), num(&c[3])) && checks::luhn(n) {
findings.insert(n);
}
}
}
#[cfg(test)]
mod tests {
use crate::mailflow::detectors::by_id;
#[test]
fn south_africa() {
let detector = by_id("za-id").unwrap();
assert_eq!(detector.count("ID 8001015009087"), 1);
assert_eq!(detector.count("8001015009088"), 0);
assert_eq!(detector.count("8013015009087"), 0);
}
}
@@ -1,172 +0,0 @@
/*
* SPDX-FileCopyrightText: 2026 Coffey Labs LLC
*
* SPDX-License-Identifier: AGPL-3.0-only
*/
//! Identifiers from the Americas outside the US and Canada (§2.3): Brazil's
//! CPF and CNPJ, and Mexico's CURP.
use super::{Detector, Findings, Region, Strength, digit_values, valid_short_date, word_near};
use regex::Regex;
use std::sync::LazyLock;
pub static DETECTORS: &[Detector] = &[
Detector::new(
"br-cpf",
"Brazil: CPF",
Region::Americas,
Strength::Checked,
br_cpf,
),
Detector::new(
"br-cnpj",
"Brazil: CNPJ",
Region::Americas,
Strength::Checked,
br_cnpj,
),
Detector::new(
"mx-curp",
"Mexico: CURP",
Region::Americas,
Strength::Checked,
mx_curp,
),
];
fn re(pattern: &str) -> Regex {
Regex::new(pattern).expect("detector pattern")
}
/// Brazil's mod 11 check digit over `digits` with `weights`.
fn br_check(digits: &[u32], weights: &[u32]) -> u32 {
match digits.iter().zip(weights).map(|(a, w)| a * w).sum::<u32>() % 11 {
0 | 1 => 0,
r => 11 - r,
}
}
/// `111.444.777-35`, or eleven bare digits.
static CPF: LazyLock<Regex> = LazyLock::new(|| re(r"\b\d{3}(\.?)\d{3}(\.?)\d{3}(-?)\d{2}\b"));
pub fn cpf_valid(n: &str) -> bool {
let d = digit_values(n);
// A run of one digit passes the arithmetic but is never issued
d.len() == 11
&& d.iter().any(|x| *x != d[0])
&& br_check(&d[..9], &[10, 9, 8, 7, 6, 5, 4, 3, 2]) == d[9]
&& br_check(&d[..10], &[11, 10, 9, 8, 7, 6, 5, 4, 3, 2]) == d[10]
}
const CPF_WORDS: &[&str] = &[
"cpf",
"cadastro de pessoas físicas",
"cadastro de pessoa física",
];
fn br_cpf(text: &str, findings: &mut Findings) {
for c in CPF.captures_iter(text) {
let whole = c.get(0).unwrap();
let written = &c[1] == "." && &c[2] == "." && &c[3] == "-";
let n: String = whole
.as_str()
.chars()
.filter(char::is_ascii_digit)
.collect();
if cpf_valid(&n) && (written || word_near(text, whole.start(), whole.end(), CPF_WORDS)) {
findings.insert(n);
}
}
}
/// `11.222.333/0001-81`, or fourteen bare digits.
static CNPJ: LazyLock<Regex> =
LazyLock::new(|| re(r"\b\d{2}(\.?)\d{3}(\.?)\d{3}(/?)\d{4}(-?)\d{2}\b"));
pub fn cnpj_valid(n: &str) -> bool {
let d = digit_values(n);
d.len() == 14
&& d.iter().any(|x| *x != d[0])
&& br_check(&d[..12], &[5, 4, 3, 2, 9, 8, 7, 6, 5, 4, 3, 2]) == d[12]
&& br_check(&d[..13], &[6, 5, 4, 3, 2, 9, 8, 7, 6, 5, 4, 3, 2]) == d[13]
}
const CNPJ_WORDS: &[&str] = &["cnpj", "cadastro nacional da pessoa jurídica"];
fn br_cnpj(text: &str, findings: &mut Findings) {
for c in CNPJ.captures_iter(text) {
let whole = c.get(0).unwrap();
let written = &c[1] == "." && &c[2] == "." && &c[3] == "/" && &c[4] == "-";
let n: String = whole
.as_str()
.chars()
.filter(char::is_ascii_digit)
.collect();
if cnpj_valid(&n) && (written || word_near(text, whole.start(), whole.end(), CNPJ_WORDS)) {
findings.insert(n);
}
}
}
/// Four letters, the birth date, sex (H, M or X), the state, three
/// consonants, a character that tells the century apart, the check digit.
static CURP: LazyLock<Regex> = LazyLock::new(|| {
re(r"(?i)\b[A-Z]{4}(\d{2})(\d{2})(\d{2})[HMX][A-Z]{2}[B-DF-HJ-NP-TV-Z]{3}[A-Z0-9]\d\b")
});
/// RENAPO's check: each character's place in `0-9 A-N Ñ O-Z`, weighted 18
/// down to 2; the digit is 10 minus the sum mod 10 (10 becomes 0).
pub fn curp_valid(curp: &str) -> bool {
const ALPHABET: &str = "0123456789ABCDEFGHIJKLMNÑOPQRSTUVWXYZ";
let mut sum = 0u32;
for (i, c) in curp.chars().take(17).enumerate() {
let Some(value) = ALPHABET.chars().position(|a| a == c) else {
return false;
};
sum += value as u32 * (18 - i as u32);
}
curp.chars().nth(17).and_then(|c| c.to_digit(10)) == Some((10 - sum % 10) % 10)
}
fn mx_curp(text: &str, findings: &mut Findings) {
for c in CURP.captures_iter(text) {
let curp = c[0].to_ascii_uppercase();
if valid_short_date(num(&c[1]), num(&c[2]), num(&c[3])) && curp_valid(&curp) {
findings.insert(curp);
}
}
}
fn num(s: &str) -> u32 {
s.parse().unwrap_or(0)
}
#[cfg(test)]
mod tests {
use crate::mailflow::detectors::by_id;
fn count(id: &str, text: &str) -> usize {
by_id(id).unwrap().count(text)
}
#[test]
fn brazil() {
assert_eq!(count("br-cpf", "CPF 111.444.777-35"), 1);
assert_eq!(count("br-cpf", "111.444.777-36"), 0);
assert_eq!(count("br-cpf", "pedido 11144477735"), 0);
assert_eq!(count("br-cpf", "cpf: 11144477735"), 1);
assert_eq!(count("br-cpf", "CPF 111.111.111-11"), 0);
assert_eq!(count("br-cnpj", "11.222.333/0001-81"), 1);
assert_eq!(count("br-cnpj", "11.222.333/0001-82"), 0);
assert_eq!(count("br-cnpj", "CNPJ 11222333000181"), 1);
}
#[test]
fn mexico() {
// python-stdnum's documented example
assert_eq!(count("mx-curp", "CURP BOXW310820HNERXN09"), 1);
assert_eq!(count("mx-curp", "BOXW310820HNERXN08"), 0);
assert_eq!(count("mx-curp", "BOXW311320HNERXN09"), 0);
}
}
@@ -1,511 +0,0 @@
/*
* SPDX-FileCopyrightText: 2026 Coffey Labs LLC
*
* SPDX-License-Identifier: AGPL-3.0-only
*/
//! Detectors that aren't tied to one country (§2.3, region "Any").
use super::{
Detector, Findings, Region, Strength, checks, digits, stands_alone, valid_date, word_near,
};
use regex::Regex;
use std::sync::LazyLock;
pub static DETECTORS: &[Detector] = &[
Detector::new(
"payment-card",
"Payment card number",
Region::Any,
Strength::Checked,
payment_card,
),
Detector::new("iban", "IBAN", Region::Any, Strength::Checked, iban),
Detector::new(
"swift-bic",
"SWIFT/BIC code",
Region::Any,
Strength::NeedsWord,
swift_bic,
),
Detector::new(
"email-addresses",
"Email addresses",
Region::Any,
Strength::Checked,
email_addresses,
),
Detector::new(
"phone-numbers",
"Phone numbers",
Region::Any,
Strength::NeedsWord,
phone_numbers,
),
Detector::new(
"date-of-birth",
"Date of birth",
Region::Any,
Strength::NeedsWord,
date_of_birth,
),
Detector::new(
"passport",
"Passport number",
Region::Any,
Strength::NeedsWord,
passport,
),
Detector::new(
"private-key",
"Private key",
Region::Any,
Strength::Checked,
private_key,
),
Detector::new(
"credentials",
"Cloud and service credentials",
Region::Any,
Strength::Checked,
credentials,
),
];
fn re(pattern: &str) -> Regex {
Regex::new(pattern).expect("detector pattern")
}
// --- Payment cards --------------------------------------------------------
/// Issuer prefixes (ISO/IEC 7812 IINs) and the lengths each network issues.
fn card_network(number: &str) -> bool {
let len = number.len();
let prefix = |n: usize| number[..n].parse::<u32>().unwrap_or(0);
match number.as_bytes()[0] {
// Visa
b'4' => matches!(len, 13 | 16 | 19),
b'5' => {
// Mastercard 51–55; Maestro 50, 56–58
(51..=55).contains(&prefix(2)) && len == 16
|| matches!(prefix(2), 50 | 56..=58) && (12..=19).contains(&len)
}
// Mastercard 2221–2720
b'2' => (2221..=2720).contains(&prefix(4)) && len == 16,
b'3' => {
// American Express 34, 37; JCB 3528–3589; Diners 300–305, 36, 38, 39
matches!(prefix(2), 34 | 37) && len == 15
|| (3528..=3589).contains(&prefix(4)) && (16..=19).contains(&len)
|| ((300..=305).contains(&prefix(3)) || matches!(prefix(2), 36 | 38 | 39))
&& (14..=19).contains(&len)
}
// Discover 6011, 644–649, 65; UnionPay 62; Maestro 6x
b'6' => (12..=19).contains(&len),
_ => false,
}
}
fn is_card(number: &str) -> bool {
(12..=19).contains(&number.len()) && card_network(number) && checks::luhn(number)
}
static CARD: LazyLock<Regex> = LazyLock::new(|| re(r"\b\d(?:[ -]?\d){11,18}\b"));
fn payment_card(text: &str, findings: &mut Findings) {
for m in CARD.find_iter(text) {
if !stands_alone(text, m.start(), m.end()) {
continue;
}
let whole = digits(m.as_str());
if is_card(&whole) {
findings.insert(whole);
continue;
}
// Two numbers side by side ("4242 4242 4242 4242 2031"): try each
// run of whole groups
let groups: Vec<String> = m.as_str().split([' ', '-']).map(digits).collect();
'runs: for from in 0..groups.len() {
let mut number = String::new();
for group in &groups[from..] {
number.push_str(group);
if is_card(&number) {
findings.insert(number);
break 'runs;
}
}
}
}
}
// --- IBAN -----------------------------------------------------------------
static IBAN: LazyLock<Regex> =
LazyLock::new(|| re(r"\b[A-Za-z]{2}\d{2}(?:[ ]?[A-Za-z0-9]){11,30}"));
fn iban(text: &str, findings: &mut Findings) {
// The pattern can run on into the next words, even the next IBAN: after
// each hit, look again from where that IBAN ended
let mut from = 0;
while let Some(m) = IBAN.find_at(text, from) {
from = m.start() + 1;
let compact = m.as_str().replace(' ', "").to_ascii_uppercase();
let Some(len) = checks::iban_length(&compact[..2]) else {
continue;
};
if compact.len() < len {
continue;
}
// Where the country's length ends in the text, spaces counted
let mut seen = 0;
let Some(end) = m
.as_str()
.char_indices()
.find(|(_, c)| {
if *c != ' ' {
seen += 1;
}
seen == len
})
.map(|(i, c)| m.start() + i + c.len_utf8())
else {
continue;
};
let candidate = &compact[..len];
if stands_alone(text, m.start(), end) && checks::iban(candidate) {
findings.insert(candidate);
from = end;
}
}
}
// --- SWIFT/BIC ------------------------------------------------------------
static BIC: LazyLock<Regex> =
LazyLock::new(|| re(r"\b[A-Z]{4}[A-Z]{2}[A-Z0-9]{2}(?:[A-Z0-9]{3})?\b"));
const BIC_WORDS: &[&str] = &[
"swift",
"bic",
"swift/bic",
"bank",
"banque",
"bankverbindung",
];
fn swift_bic(text: &str, findings: &mut Findings) {
for m in BIC.find_iter(text) {
let code = m.as_str();
if checks::is_country(&code[4..6]) && word_near(text, m.start(), m.end(), BIC_WORDS) {
findings.insert(code);
}
}
}
// --- Contact lists --------------------------------------------------------
static EMAIL: LazyLock<Regex> =
LazyLock::new(|| re(r"(?i)\b[a-z0-9._%+-]+@[a-z0-9-]+(?:\.[a-z0-9-]+)*\.[a-z]{2,}\b"));
fn email_addresses(text: &str, findings: &mut Findings) {
for m in EMAIL.find_iter(text) {
findings.insert(m.as_str().to_lowercase());
}
}
/// International form: found alone. National form: only with a word.
static PHONE_INTL: LazyLock<Regex> = LazyLock::new(|| re(r"\+\d{1,3}(?:[ .-]?\(?\d{1,4}\)?){2,5}"));
static PHONE_NATIONAL: LazyLock<Regex> =
LazyLock::new(|| re(r"\(?\d{2,4}\)?[ .-]\d{3,4}[ .-]\d{3,4}"));
const PHONE_WORDS: &[&str] = &[
"phone",
"tel",
"telephone",
"mobile",
"cell",
"fax",
"telefon",
"téléphone",
"teléfono",
"telefono",
"handy",
"portable",
"móvil",
"cellulare",
"mobiel",
];
fn phone_numbers(text: &str, findings: &mut Findings) {
let mut international = Vec::new();
for m in PHONE_INTL.find_iter(text) {
let number = digits(m.as_str());
if (8..=15).contains(&number.len()) && stands_alone(text, m.start() + 1, m.end()) {
findings.insert(number);
international.push(m.range());
}
}
for m in PHONE_NATIONAL.find_iter(text) {
let number = digits(m.as_str());
// Not the tail of an international number already counted
if international.iter().any(|r| r.contains(&m.start())) {
continue;
}
if (9..=11).contains(&number.len())
&& stands_alone(text, m.start(), m.end())
&& !text[..m.start()].ends_with('+')
&& word_near(text, m.start(), m.end(), PHONE_WORDS)
{
findings.insert(number);
}
}
}
// --- Date of birth --------------------------------------------------------
static DATE_ISO: LazyLock<Regex> = LazyLock::new(|| re(r"\b(\d{4})-(\d{2})-(\d{2})\b"));
static DATE_NUMERIC: LazyLock<Regex> =
LazyLock::new(|| re(r"\b(\d{1,2})[./-](\d{1,2})[./-](\d{4})\b"));
static DATE_WORDS: LazyLock<Regex> = LazyLock::new(|| {
re(
r"(?i)\b(?:(\d{1,2})\s+(jan|feb|mar|apr|may|jun|jul|aug|sep|oct|nov|dec)[a-z]*\.?,?\s+(\d{4})|(jan|feb|mar|apr|may|jun|jul|aug|sep|oct|nov|dec)[a-z]*\.?\s+(\d{1,2}),?\s+(\d{4}))\b",
)
});
const BIRTH_WORDS: &[&str] = &[
"born",
"birth",
"dob",
"d.o.b",
"birthday",
"birthdate",
"geburtsdatum",
"geboren",
"naissance",
"né le",
"née le",
"nacimiento",
"nacido",
"nacida",
"nascita",
"nato il",
"nata il",
"geboortedatum",
"födelsedatum",
"fødselsdato",
"syntymäaika",
"urodzenia",
"nascimento",
];
fn month_number(name: &str) -> u32 {
const MONTHS: [&str; 12] = [
"jan", "feb", "mar", "apr", "may", "jun", "jul", "aug", "sep", "oct", "nov", "dec",
];
let name = name.to_lowercase();
MONTHS
.iter()
.position(|m| *m == name)
.map_or(0, |i| i as u32 + 1)
}
fn date_of_birth(text: &str, findings: &mut Findings) {
let mut add = |start: usize, end: usize, key: String| {
if word_near(text, start, end, BIRTH_WORDS) {
findings.insert(key);
}
};
let num = |s: &str| s.parse::<u32>().unwrap_or(0);
for c in DATE_ISO.captures_iter(text) {
let (y, m, d) = (num(&c[1]), num(&c[2]), num(&c[3]));
let whole = c.get(0).unwrap();
if valid_date(y, m, d) {
add(whole.start(), whole.end(), format!("{y:04}{m:02}{d:02}"));
}
}
for c in DATE_NUMERIC.captures_iter(text) {
let (a, b, y) = (num(&c[1]), num(&c[2]), num(&c[3]));
let whole = c.get(0).unwrap();
// Day first or month first: either reading that is a real date
if valid_date(y, b, a) || valid_date(y, a, b) {
add(whole.start(), whole.end(), whole.as_str().to_string());
}
}
for c in DATE_WORDS.captures_iter(text) {
let whole = c.get(0).unwrap();
let (d, m, y) = match (c.get(1), c.get(4)) {
(Some(d), _) => (num(d.as_str()), month_number(&c[2]), num(&c[3])),
(_, Some(m)) => (num(&c[5]), month_number(m.as_str()), num(&c[6])),
_ => continue,
};
if valid_date(y, m, d) {
add(whole.start(), whole.end(), format!("{y:04}{m:02}{d:02}"));
}
}
}
// --- Passport -------------------------------------------------------------
static PASSPORT: LazyLock<Regex> = LazyLock::new(|| re(r"\b[A-Z0-9]{6,9}\b"));
const PASSPORT_WORDS: &[&str] = &[
"passport",
"passeport",
"reisepass",
"pasaporte",
"passaporto",
"paspoort",
"passnummer",
"pass-nr",
"passport no",
"pasaporte n.º",
"passaporte",
];
fn passport(text: &str, findings: &mut Findings) {
for m in PASSPORT.find_iter(text) {
let value = m.as_str();
if value.bytes().filter(u8::is_ascii_digit).count() >= 5
&& word_near(text, m.start(), m.end(), PASSPORT_WORDS)
{
findings.insert(value);
}
}
}
// --- Keys and credentials -------------------------------------------------
static PRIVATE_KEY: LazyLock<Regex> = LazyLock::new(|| {
re(
r"-----BEGIN (?:(?:RSA|EC|DSA|OPENSSH|ENCRYPTED|PGP) )?PRIVATE KEY(?: BLOCK)?-----\s*([A-Za-z0-9+/=:\s-]{0,64})",
)
});
fn private_key(text: &str, findings: &mut Findings) {
for c in PRIVATE_KEY.captures_iter(text) {
// Each key once, by the start of its body
let body: String = c[1].chars().filter(|c| !c.is_whitespace()).collect();
let whole = c.get(0).unwrap();
findings.insert(if body.is_empty() {
format!("@{}", whole.start())
} else {
body
});
}
}
/// Published token formats: AWS access key IDs, GitHub tokens, Slack
/// tokens, Stripe live secret and restricted keys, Google API keys.
static CREDENTIAL: LazyLock<Regex> = LazyLock::new(|| {
re(concat!(
r"\b(?:",
r"(?:AKIA|ASIA|ABIA|ACCA)[A-Z0-9]{16}",
r"|gh[pousr]_[A-Za-z0-9]{36}",
r"|github_pat_[A-Za-z0-9_]{82}",
r"|xox[abposr]-[A-Za-z0-9-]{10,72}",
r"|(?:sk|rk)_live_[A-Za-z0-9]{24,99}",
r"|AIza[0-9A-Za-z_-]{35}",
r")\b"
))
});
fn credentials(text: &str, findings: &mut Findings) {
for m in CREDENTIAL.find_iter(text) {
findings.insert(m.as_str());
}
}
#[cfg(test)]
mod tests {
use crate::mailflow::detectors::by_id;
fn count(id: &str, text: &str) -> usize {
by_id(id).unwrap().count(text)
}
#[test]
fn payment_cards() {
// Networks' and processors' published test numbers
let text = "Visa 4242 4242 4242 4242, MC 5555-5555-5555-4444, Amex 378282246310005, \
Discover 6011111111111117, JCB 3566002020360505, Diners 30569309025904, \
UnionPay 6200000000000005, Mastercard 2-series 2223003122003222";
assert_eq!(count("payment-card", text), 8);
// Luhn fails, wrong network length, inside a longer number
assert_eq!(count("payment-card", "4242424242424241"), 0);
assert_eq!(count("payment-card", "378282246310005 0"), 1);
assert_eq!(count("payment-card", "order 94242424242424242 shipped"), 0);
// The same number twice counts once
assert_eq!(
count("payment-card", "4242424242424242 and 4242-4242-4242-4242"),
1
);
// A card followed by a year
assert_eq!(count("payment-card", "card 4242 4242 4242 4242 2031"), 1);
}
#[test]
fn ibans() {
let text =
"Pay GB29 NWBK 6016 1331 9268 19 or de89370400440532013000 (NL91ABNA0417164300).";
assert_eq!(count("iban", text), 3);
assert_eq!(count("iban", "GB29 NWBK 6016 1331 9268 18"), 0);
// Runs into the next word: still found at the country's length
assert_eq!(count("iban", "IBAN NL91ABNA0417164300 BIC ABNANL2A"), 1);
}
#[test]
fn swift_codes_need_a_word() {
assert_eq!(count("swift-bic", "SWIFT: DEUTDEFF500"), 1);
assert_eq!(count("swift-bic", "BIC NWBKGB2L"), 1);
assert_eq!(count("swift-bic", "HAPPYDAYS DEUTDEFF"), 0);
// Not a country in positions 5–6
assert_eq!(count("swift-bic", "BIC DEUTZZFF"), 0);
}
#[test]
fn email_and_phone_lists() {
let list = "[email protected], [email protected], [email protected], [email protected]";
assert_eq!(count("email-addresses", list), 3);
assert_eq!(
count("phone-numbers", "+44 20 7946 0958, +1 (415) 555-2671"),
2
);
assert_eq!(count("phone-numbers", "call 020 7946 0958"), 0);
assert_eq!(count("phone-numbers", "Tel: 020 7946 0958"), 1);
assert_eq!(count("phone-numbers", "invoice 020 7946 0958"), 0);
// One number, not also its national tail
assert_eq!(count("phone-numbers", "Tel: +44 20 7946 0958"), 1);
}
#[test]
fn dates_of_birth() {
assert_eq!(count("date-of-birth", "DOB: 1984-02-29"), 1);
assert_eq!(count("date-of-birth", "Geburtsdatum 31.12.1970"), 1);
assert_eq!(count("date-of-birth", "born on March 3, 1962"), 1);
assert_eq!(count("date-of-birth", "date of birth 3 Mar 1962"), 1);
// Not a real date, no word, a meeting
assert_eq!(count("date-of-birth", "DOB: 1985-02-29"), 0);
assert_eq!(count("date-of-birth", "invoice 1984-02-29"), 0);
assert_eq!(count("date-of-birth", "Meeting on 12/05/2026"), 0);
}
#[test]
fn passports_need_a_word() {
assert_eq!(count("passport", "Passport number: 533380006"), 1);
assert_eq!(count("passport", "Reisepass C01X00T47"), 1);
assert_eq!(count("passport", "Order 533380006 shipped"), 0);
// Mostly letters: a word, not a number
assert_eq!(count("passport", "passport PASSWORD"), 0);
}
#[test]
fn keys_and_credentials() {
let key = "-----BEGIN OPENSSH PRIVATE KEY-----\nb3BlbnNzaC1rZXktdjEAAAAABG5vbmUAAAAEbm9uZQ\n-----END OPENSSH PRIVATE KEY-----";
assert_eq!(count("private-key", key), 1);
assert_eq!(count("private-key", "-----BEGIN PUBLIC KEY-----\nMFkw"), 0);
// Documentation examples of each format
let tokens = "AKIAIOSFODNN7EXAMPLE ghp_0123456789abcdefghijklmnopqrstuvwxyz \
AIzaSyA-0123456789abcdefghijklmnopqrstu";
assert_eq!(count("credentials", tokens), 3);
assert_eq!(count("credentials", "AKIA123 ghp_short"), 0);
}
}
@@ -1,281 +0,0 @@
/*
* SPDX-FileCopyrightText: 2026 Coffey Labs LLC
*
* SPDX-License-Identifier: AGPL-3.0-only
*/
//! Asian identifiers (§2.3): India's Aadhaar and PAN, China's resident ID,
//! Japan's My Number, Singapore's NRIC and FIN, and South Korea's resident
//! registration number.
use super::{
Detector, Findings, Region, Strength, digit_values, valid_date, valid_short_date, word_near,
};
use regex::Regex;
use std::sync::LazyLock;
pub static DETECTORS: &[Detector] = &[
Detector::new(
"in-aadhaar",
"India: Aadhaar",
Region::Asia,
Strength::Checked,
in_aadhaar,
),
Detector::new(
"in-pan",
"India: PAN",
Region::Asia,
Strength::NeedsWord,
in_pan,
),
Detector::new(
"cn-resident-id",
"China: resident ID",
Region::Asia,
Strength::Checked,
cn_resident_id,
),
Detector::new(
"jp-my-number",
"Japan: My Number",
Region::Asia,
Strength::Checked,
jp_my_number,
),
Detector::new(
"sg-nric",
"Singapore: NRIC and FIN",
Region::Asia,
Strength::Checked,
sg_nric,
),
Detector::new(
"kr-rrn",
"South Korea: resident registration number",
Region::Asia,
Strength::NeedsWord,
kr_rrn,
),
];
fn re(pattern: &str) -> Regex {
Regex::new(pattern).expect("detector pattern")
}
/// Twelve digits written in fours, or bare.
static TWELVE: LazyLock<Regex> = LazyLock::new(|| re(r"\b(\d{4})( ?)(\d{4})( ?)(\d{4})\b"));
const VERHOEFF_D: [[u8; 10]; 10] = [
[0, 1, 2, 3, 4, 5, 6, 7, 8, 9],
[1, 2, 3, 4, 0, 6, 7, 8, 9, 5],
[2, 3, 4, 0, 1, 7, 8, 9, 5, 6],
[3, 4, 0, 1, 2, 8, 9, 5, 6, 7],
[4, 0, 1, 2, 3, 9, 5, 6, 7, 8],
[5, 9, 8, 7, 6, 0, 4, 3, 2, 1],
[6, 5, 9, 8, 7, 1, 0, 4, 3, 2],
[7, 6, 5, 9, 8, 2, 1, 0, 4, 3],
[8, 7, 6, 5, 9, 3, 2, 1, 0, 4],
[9, 8, 7, 6, 5, 4, 3, 2, 1, 0],
];
const VERHOEFF_P: [[u8; 10]; 8] = [
[0, 1, 2, 3, 4, 5, 6, 7, 8, 9],
[1, 5, 7, 6, 2, 8, 3, 0, 9, 4],
[5, 8, 0, 3, 7, 9, 6, 1, 4, 2],
[8, 9, 1, 6, 0, 4, 3, 5, 2, 7],
[9, 4, 5, 3, 1, 2, 6, 8, 7, 0],
[4, 2, 8, 6, 5, 7, 3, 9, 0, 1],
[2, 7, 9, 3, 8, 0, 6, 4, 1, 5],
[7, 0, 4, 6, 9, 1, 3, 2, 5, 8],
];
/// The Verhoeff check (dihedral group D5).
pub fn verhoeff(n: &str) -> bool {
let mut c = 0u8;
for (i, b) in n.bytes().rev().enumerate() {
c = VERHOEFF_D[c as usize][VERHOEFF_P[i % 8][(b - b'0') as usize] as usize];
}
c == 0
}
const AADHAAR_WORDS: &[&str] = &["aadhaar", "aadhar", "uidai", "uid"];
fn in_aadhaar(text: &str, findings: &mut Findings) {
for c in TWELVE.captures_iter(text) {
let whole = c.get(0).unwrap();
let n = format!("{}{}{}", &c[1], &c[3], &c[5]);
let written = &c[2] == " " && &c[4] == " ";
// Never starts with 0 or 1
if !n.starts_with(['0', '1'])
&& verhoeff(&n)
&& (written || word_near(text, whole.start(), whole.end(), AADHAAR_WORDS))
{
findings.insert(n);
}
}
}
/// Five letters (the fourth names the holder's type), four digits, a letter.
static PAN: LazyLock<Regex> = LazyLock::new(|| re(r"\b[A-Z]{3}[ABCFGHLJPTK][A-Z]\d{4}[A-Z]\b"));
const PAN_WORDS: &[&str] = &["pan", "pan card", "permanent account number", "income tax"];
fn in_pan(text: &str, findings: &mut Findings) {
for m in PAN.find_iter(text) {
if word_near(text, m.start(), m.end(), PAN_WORDS) {
findings.insert(m.as_str());
}
}
}
/// Region, birth date `YYYYMMDD`, sequence, then the ISO 7064 MOD 11-2
/// check (0–9 or X).
static CN_ID: LazyLock<Regex> =
LazyLock::new(|| re(r"(?i)\b[1-8]\d{5}(\d{4})(\d{2})(\d{2})\d{3}[\dX]\b"));
pub fn cn_id_valid(id: &str) -> bool {
const WEIGHTS: [u32; 17] = [7, 9, 10, 5, 8, 4, 2, 1, 6, 3, 7, 9, 10, 5, 8, 4, 2];
const CHECKS: &[u8] = b"10X98765432";
let sum: u32 = digit_values(&id[..17])
.iter()
.zip(WEIGHTS)
.map(|(a, w)| a * w)
.sum();
CHECKS[(sum % 11) as usize] == id.as_bytes()[17].to_ascii_uppercase()
}
fn cn_resident_id(text: &str, findings: &mut Findings) {
for c in CN_ID.captures_iter(text) {
let id = c[0].to_ascii_uppercase();
let (y, m, d) = (num(&c[1]), num(&c[2]), num(&c[3]));
if valid_date(y, m, d) && cn_id_valid(&id) {
findings.insert(id);
}
}
}
/// My Number: weights 2–7 then 2–6 from the right; a remainder of 0 or 1
/// gives 0, else 11 minus it.
pub fn my_number_valid(n: &str) -> bool {
let d = digit_values(n);
let sum: u32 = (1..=11)
.map(|i| d[11 - i] * if i <= 6 { i as u32 + 1 } else { i as u32 - 5 })
.sum();
let check = match sum % 11 {
0 | 1 => 0,
r => 11 - r,
};
check == d[11]
}
const MY_NUMBER_WORDS: &[&str] = &[
"my number",
"mynumber",
"マイナンバー",
"個人番号",
"kojin bango",
];
fn jp_my_number(text: &str, findings: &mut Findings) {
for c in TWELVE.captures_iter(text) {
let whole = c.get(0).unwrap();
let n = format!("{}{}{}", &c[1], &c[3], &c[5]);
let written = &c[2] == " " && &c[4] == " ";
if my_number_valid(&n)
&& (written || word_near(text, whole.start(), whole.end(), MY_NUMBER_WORDS))
{
findings.insert(n);
}
}
}
static NRIC: LazyLock<Regex> = LazyLock::new(|| re(r"(?i)\b([STFGM])(\d{7})([A-Z])\b"));
/// Weights 2, 7, 6, 5, 4, 3, 2; T and G add 4, M adds 3; each series has its
/// own table of check letters.
fn nric_valid(prefix: u8, digits: &str, check: u8) -> bool {
let sum: u32 = digit_values(digits)
.iter()
.zip([2, 7, 6, 5, 4, 3, 2])
.map(|(a, w)| a * w)
.sum::<u32>()
+ match prefix {
b'T' | b'G' => 4,
b'M' => 3,
_ => 0,
};
let table: &[u8] = match prefix {
b'S' | b'T' => b"JZIHGFEDCBA",
b'F' | b'G' => b"XWUTRQPNMLK",
_ => b"KLJNPQRTUWX",
};
table[(sum % 11) as usize] == check
}
fn sg_nric(text: &str, findings: &mut Findings) {
for c in NRIC.captures_iter(text) {
let id = c[0].to_ascii_uppercase();
let bytes = id.as_bytes();
if nric_valid(bytes[0], &c[2], bytes[8]) {
findings.insert(id);
}
}
}
/// `YYMMDD-GNNNNNN`, the seventh digit giving sex and century.
static RRN: LazyLock<Regex> = LazyLock::new(|| re(r"\b(\d{2})(\d{2})(\d{2})-?([1-8])\d{6}\b"));
const RRN_WORDS: &[&str] = &["주민등록번호", "주민번호", "resident registration", "rrn"];
fn kr_rrn(text: &str, findings: &mut Findings) {
for c in RRN.captures_iter(text) {
let whole = c.get(0).unwrap();
if valid_short_date(num(&c[1]), num(&c[2]), num(&c[3]))
&& word_near(text, whole.start(), whole.end(), RRN_WORDS)
{
findings.insert(whole.as_str().replace('-', ""));
}
}
}
fn num(s: &str) -> u32 {
s.parse().unwrap_or(0)
}
#[cfg(test)]
mod tests {
use crate::mailflow::detectors::by_id;
fn count(id: &str, text: &str) -> usize {
by_id(id).unwrap().count(text)
}
#[test]
fn india() {
assert_eq!(count("in-aadhaar", "2345 6789 0124"), 1);
assert_eq!(count("in-aadhaar", "2345 6789 0125"), 0);
assert_eq!(count("in-aadhaar", "order 234567890124"), 0);
assert_eq!(count("in-aadhaar", "Aadhaar 234567890124"), 1);
assert_eq!(count("in-pan", "PAN: ABCPE1234F"), 1);
assert_eq!(count("in-pan", "ABCPE1234F"), 0);
}
#[test]
fn china_japan() {
assert_eq!(count("cn-resident-id", "11010519491231002X"), 1);
assert_eq!(count("cn-resident-id", "110105194912310021"), 0);
assert_eq!(count("cn-resident-id", "11010519491331002X"), 0);
assert_eq!(count("jp-my-number", "1234 5678 9018"), 1);
assert_eq!(count("jp-my-number", "1234 5678 9017"), 0);
assert_eq!(count("jp-my-number", "マイナンバー 123456789018"), 1);
}
#[test]
fn singapore_korea() {
assert_eq!(count("sg-nric", "S1234567D and T1234567J"), 2);
assert_eq!(count("sg-nric", "S1234567E"), 0);
assert_eq!(count("kr-rrn", "주민등록번호 800101-1234567"), 1);
assert_eq!(count("kr-rrn", "800101-1234567"), 0);
assert_eq!(count("kr-rrn", "RRN 801301-1234567"), 0);
}
}
@@ -1,128 +0,0 @@
/*
* SPDX-FileCopyrightText: 2026 Coffey Labs LLC
*
* SPDX-License-Identifier: AGPL-3.0-only
*/
//! Australian identifiers (§2.3): the ATO's Tax File Number and the Medicare
//! card number.
use super::{Detector, Findings, Region, Strength, word_near};
use regex::Regex;
use std::sync::LazyLock;
pub static DETECTORS: &[Detector] = &[
Detector::new(
"au-tfn",
"Australian Tax File Number",
Region::Australia,
Strength::Checked,
tfn,
),
Detector::new(
"au-medicare",
"Australian Medicare number",
Region::Australia,
Strength::Checked,
medicare,
),
];
fn re(pattern: &str) -> Regex {
Regex::new(pattern).expect("detector pattern")
}
/// `NNN NNN NNN` stands alone; bare digits (eight or nine) need a word.
static TFN: LazyLock<Regex> = LazyLock::new(|| re(r"\b(\d{3})( ?)(\d{3})( ?)(\d{2,3})\b"));
/// Weighted sum mod 11, with the ATO's weights for 9- and 8-digit numbers.
pub fn tfn_valid(n: &str) -> bool {
let weights: &[u32] = match n.len() {
9 => &[1, 4, 3, 7, 5, 8, 6, 9, 10],
8 => &[10, 7, 8, 4, 6, 3, 5, 1],
_ => return false,
};
n.bytes()
.zip(weights)
.map(|(b, w)| u32::from(b - b'0') * w)
.sum::<u32>()
% 11
== 0
}
const TFN_WORDS: &[&str] = &["tfn", "tax file number", "tax file no"];
fn tfn(text: &str, findings: &mut Findings) {
for c in TFN.captures_iter(text) {
let whole = c.get(0).unwrap();
let n = format!("{}{}{}", &c[1], &c[3], &c[5]);
let written = n.len() == 9 && c[2] == *" " && c[4] == *" ";
if tfn_valid(&n) && (written || word_near(text, whole.start(), whole.end(), TFN_WORDS)) {
findings.insert(n);
}
}
}
/// `NNNN NNNNN N` (and an optional issue number) stands alone; bare digits
/// need a word.
static MEDICARE: LazyLock<Regex> =
LazyLock::new(|| re(r"\b([2-6]\d{3})( ?)(\d{5})( ?)(\d)(?:[ -]?\d)?\b"));
/// The ninth digit is the weighted sum (1, 3, 7, 9, 1, 3, 7, 9) of the first
/// eight, mod 10.
pub fn medicare_valid(n: &str) -> bool {
let d: Vec<u32> = n.bytes().map(|b| u32::from(b - b'0')).collect();
d.len() >= 9
&& d[..8]
.iter()
.zip([1, 3, 7, 9, 1, 3, 7, 9])
.map(|(a, w)| a * w)
.sum::<u32>()
% 10
== d[8]
}
const MEDICARE_WORDS: &[&str] = &[
"medicare",
"medicare card",
"medicare no",
"medicare number",
];
fn medicare(text: &str, findings: &mut Findings) {
for c in MEDICARE.captures_iter(text) {
let whole = c.get(0).unwrap();
let n = format!("{}{}{}", &c[1], &c[3], &c[5]);
let written = c[2] == *" " && c[4] == *" ";
if medicare_valid(&n)
&& (written || word_near(text, whole.start(), whole.end(), MEDICARE_WORDS))
{
findings.insert(n);
}
}
}
#[cfg(test)]
mod tests {
use crate::mailflow::detectors::by_id;
fn count(id: &str, text: &str) -> usize {
by_id(id).unwrap().count(text)
}
#[test]
fn tax_file_numbers() {
assert_eq!(count("au-tfn", "TFN 123 456 782"), 1);
assert_eq!(count("au-tfn", "123 456 789"), 0);
assert_eq!(count("au-tfn", "order 123456782"), 0);
assert_eq!(count("au-tfn", "tax file number 123456782"), 1);
}
#[test]
fn medicare_numbers() {
assert_eq!(count("au-medicare", "2123 45670 1"), 1);
assert_eq!(count("au-medicare", "2123 45671 1"), 0);
assert_eq!(count("au-medicare", "ref 2123456701"), 0);
assert_eq!(count("au-medicare", "Medicare 2123456701"), 1);
}
}
@@ -1,66 +0,0 @@
/*
* SPDX-FileCopyrightText: 2026 Coffey Labs LLC
*
* SPDX-License-Identifier: AGPL-3.0-only
*/
//! Canadian identifiers (§2.3): the Social Insurance Number.
use super::{Detector, Findings, Region, Strength, checks, word_near};
use regex::Regex;
use std::sync::LazyLock;
pub static DETECTORS: &[Detector] = &[Detector::new(
"ca-sin",
"Canadian Social Insurance Number",
Region::Canada,
Strength::Checked,
sin,
)];
/// `NNN NNN NNN` or `NNN-NNN-NNN` stands alone; nine bare digits need a word.
static SIN: LazyLock<Regex> = LazyLock::new(|| {
Regex::new(r"\b(\d{3})([ -]?)(\d{3})([ -]?)(\d{3})\b").expect("detector pattern")
});
const SIN_WORDS: &[&str] = &[
"sin",
"social insurance",
"nas",
"numéro d'assurance sociale",
"assurance sociale",
];
fn sin(text: &str, findings: &mut Findings) {
for c in SIN.captures_iter(text) {
let whole = c.get(0).unwrap();
let n = format!("{}{}{}", &c[1], &c[3], &c[5]);
let written = !c[2].is_empty() && c[2] == c[4];
// 0 and 8 are never issued as a first digit
if !n.starts_with(['0', '8'])
&& checks::luhn(&n)
&& (written || word_near(text, whole.start(), whole.end(), SIN_WORDS))
{
findings.insert(n);
}
}
}
#[cfg(test)]
mod tests {
use crate::mailflow::detectors::by_id;
fn count(text: &str) -> usize {
by_id("ca-sin").unwrap().count(text)
}
#[test]
fn social_insurance_numbers() {
assert_eq!(count("130 692 544 and 193-456-787"), 2);
assert_eq!(count("130 692 545"), 0);
// The government's printed example starts with 0, never issued
assert_eq!(count("046 454 286"), 0);
assert_eq!(count("order 130692544"), 0);
assert_eq!(count("SIN: 130692544"), 1);
}
}
@@ -1,227 +0,0 @@
/*
* SPDX-FileCopyrightText: 2026 Coffey Labs LLC
*
* SPDX-License-Identifier: AGPL-3.0-only
*/
//! Check-digit algorithms, each from its public definition.
/// The Luhn check (ISO/IEC 7812-1, Annex B) over a string of ASCII digits.
pub fn luhn(digits: &str) -> bool {
if digits.len() < 2 || !digits.bytes().all(|b| b.is_ascii_digit()) {
return false;
}
let sum: u32 = digits
.bytes()
.rev()
.enumerate()
.map(|(i, b)| {
let d = u32::from(b - b'0');
if i % 2 == 1 {
let d = d * 2;
if d > 9 { d - 9 } else { d }
} else {
d
}
})
.sum();
sum.is_multiple_of(10)
}
/// ISO 13616 IBAN lengths, by country, from the IBAN registry.
const IBAN_LENGTHS: &[(&str, usize)] = &[
("AD", 24),
("AE", 23),
("AL", 28),
("AT", 20),
("AZ", 28),
("BA", 20),
("BE", 16),
("BG", 22),
("BH", 22),
("BI", 27),
("BR", 29),
("BY", 28),
("CH", 21),
("CR", 22),
("CY", 28),
("CZ", 24),
("DE", 22),
("DJ", 27),
("DK", 18),
("DO", 28),
("EE", 20),
("EG", 29),
("ES", 24),
("FI", 18),
("FK", 18),
("FO", 18),
("FR", 27),
("GB", 22),
("GE", 22),
("GI", 23),
("GL", 18),
("GR", 27),
("GT", 28),
("HN", 28),
("HR", 21),
("HU", 28),
("IE", 22),
("IL", 23),
("IQ", 23),
("IS", 26),
("IT", 27),
("JO", 30),
("KW", 30),
("KZ", 20),
("LB", 28),
("LC", 32),
("LI", 21),
("LT", 20),
("LU", 20),
("LV", 21),
("LY", 25),
("MC", 27),
("MD", 24),
("ME", 22),
("MK", 19),
("MN", 20),
("MR", 27),
("MT", 31),
("MU", 30),
("NI", 28),
("NL", 18),
("NO", 15),
("OM", 23),
("PK", 24),
("PL", 28),
("PS", 29),
("PT", 25),
("QA", 29),
("RO", 24),
("RS", 22),
("RU", 33),
("SA", 24),
("SC", 31),
("SD", 18),
("SE", 24),
("SI", 19),
("SK", 24),
("SM", 27),
("SO", 23),
("ST", 25),
("SV", 28),
("TL", 23),
("TN", 24),
("TR", 26),
("UA", 29),
("VA", 22),
("VG", 24),
("XK", 20),
("YE", 30),
];
/// The IBAN length for a country code, if the country uses IBANs.
pub fn iban_length(country: &str) -> Option<usize> {
IBAN_LENGTHS
.iter()
.find(|(code, _)| *code == country)
.map(|(_, len)| *len)
}
/// ISO 13616 / ISO 7064 MOD 97-10 over an IBAN with no spaces, upper case:
/// move the first four characters to the end, turn letters into 10–35, and
/// the number mod 97 must be 1. Also checks the country's length.
pub fn iban(iban: &str) -> bool {
if iban.len() < 5
|| !iban
.bytes()
.all(|b| b.is_ascii_uppercase() || b.is_ascii_digit())
{
return false;
}
if iban_length(&iban[..2]) != Some(iban.len())
|| !iban[2..4].bytes().all(|b| b.is_ascii_digit())
{
return false;
}
let mut remainder: u32 = 0;
for b in iban[4..].bytes().chain(iban[..4].bytes()) {
let value = if b.is_ascii_digit() {
u32::from(b - b'0')
} else {
u32::from(b - b'A') + 10
};
remainder = if value >= 10 {
(remainder * 100 + value) % 97
} else {
(remainder * 10 + value) % 97
};
}
remainder == 1
}
/// ISO 3166-1 alpha-2 country codes, for SWIFT/BIC positions 5–6.
const COUNTRIES: &str = "AD AE AF AG AI AL AM AO AQ AR AS AT AU AW AX AZ BA BB BD BE BF BG BH BI BJ \
BL BM BN BO BQ BR BS BT BV BW BY BZ CA CC CD CF CG CH CI CK CL CM CN CO CR CU CV CW CX CY CZ DE DJ \
DK DM DO DZ EC EE EG EH ER ES ET FI FJ FK FM FO FR GA GB GD GE GF GG GH GI GL GM GN GP GQ GR GS GT \
GU GW GY HK HM HN HR HT HU ID IE IL IM IN IO IQ IR IS IT JE JM JO JP KE KG KH KI KM KN KP KR KW KY \
KZ LA LB LC LI LK LR LS LT LU LV LY MA MC MD ME MF MG MH MK ML MM MN MO MP MQ MR MS MT MU MV MW MX \
MY MZ NA NC NE NF NG NI NL NO NP NR NU NZ OM PA PE PF PG PH PK PL PM PN PR PS PT PW PY QA RE RO RS \
RU RW SA SB SC SD SE SG SH SI SJ SK SL SM SN SO SR SS ST SV SX SY SZ TC TD TF TG TH TJ TK TL TM TN \
TO TR TT TV TW TZ UA UG UM US UY UZ VA VC VE VG VI VN VU WF WS XK YE YT ZA ZM ZW";
pub fn is_country(code: &str) -> bool {
code.len() == 2 && COUNTRIES.split(' ').any(|c| c == code)
}
#[cfg(test)]
mod tests {
use super::*;
#[test]
fn luhn_known_numbers() {
// Published test card numbers
for good in [
"4242424242424242",
"5555555555554444",
"378282246310005",
"79927398713",
] {
assert!(luhn(good), "{good}");
}
for bad in ["4242424242424241", "79927398710", "1", "12a4"] {
assert!(!luhn(bad), "{bad}");
}
}
#[test]
fn iban_registry_examples() {
// The IBAN registry's own examples
for good in [
"GB29NWBK60161331926819",
"DE89370400440532013000",
"FR1420041010050500013M02606",
"NL91ABNA0417164300",
"BE68539007547034",
"NO9386011117947",
"CH9300762011623852957",
] {
assert!(iban(good), "{good}");
}
for bad in [
"GB29NWBK60161331926818", // check fails
"GB29NWBK6016133192681", // too short for GB
"ZZ29NWBK60161331926819", // no such country
"DE8937040044053201300A", // letters where DE has none still fail mod 97
] {
assert!(!iban(bad), "{bad}");
}
}
#[test]
fn countries() {
assert!(is_country("DE") && is_country("US") && is_country("XK"));
assert!(!is_country("ZZ") && !is_country("D"));
}
}
@@ -1,646 +0,0 @@
/*
* SPDX-FileCopyrightText: 2026 Coffey Labs LLC
*
* SPDX-License-Identifier: AGPL-3.0-only
*/
//! European Union national identifiers (§2.3), each from its issuer's
//! published rules. An identifier that is only digits and whose check a
//! random number passes often (mod 10, mod 11) counts alone only in its
//! written form, and as bare digits only beside a word.
use super::{
Detector, Findings, Region, Strength, checks, digit_values, stands_alone, valid_short_date,
word_near,
};
use regex::Regex;
use std::sync::LazyLock;
pub static DETECTORS: &[Detector] = &[
Detector::new(
"de-tax-id",
"Germany: tax ID (Steuer-ID)",
Region::Eu,
Strength::Checked,
de_tax_id,
),
Detector::new(
"de-id-card",
"Germany: ID card number",
Region::Eu,
Strength::Checked,
de_id_card,
),
Detector::new(
"fr-nir",
"France: social security number (NIR)",
Region::Eu,
Strength::Checked,
fr_nir,
),
Detector::new(
"es-dni-nie",
"Spain: DNI and NIE",
Region::Eu,
Strength::Checked,
es_dni_nie,
),
Detector::new(
"it-codice-fiscale",
"Italy: codice fiscale",
Region::Eu,
Strength::Checked,
it_codice_fiscale,
),
Detector::new(
"nl-bsn",
"Netherlands: BSN",
Region::Eu,
Strength::Checked,
nl_bsn,
),
Detector::new(
"be-national-number",
"Belgium: national number",
Region::Eu,
Strength::Checked,
be_national_number,
),
Detector::new(
"pl-pesel",
"Poland: PESEL",
Region::Eu,
Strength::Checked,
pl_pesel,
),
Detector::new(
"se-personnummer",
"Sweden: personnummer",
Region::Eu,
Strength::Checked,
se_personnummer,
),
Detector::new(
"dk-cpr",
"Denmark: CPR number",
Region::Eu,
Strength::NeedsWord,
dk_cpr,
),
Detector::new(
"fi-hetu",
"Finland: personal identity code",
Region::Eu,
Strength::Checked,
fi_hetu,
),
Detector::new(
"ie-pps",
"Ireland: PPS number",
Region::Eu,
Strength::Checked,
ie_pps,
),
Detector::new(
"pt-nif",
"Portugal: NIF",
Region::Eu,
Strength::Checked,
pt_nif,
),
Detector::new(
"at-svnr",
"Austria: social insurance number",
Region::Eu,
Strength::Checked,
at_svnr,
),
];
fn re(pattern: &str) -> Regex {
Regex::new(pattern).expect("detector pattern")
}
fn num(s: &str) -> u32 {
s.parse().unwrap_or(0)
}
// --- Germany --------------------------------------------------------------
/// Eleven digits, written `86 095 742 719` on the BZSt's letters.
static DE_TAX: LazyLock<Regex> = LazyLock::new(|| re(r"\b\d{2}( ?)\d{3}( ?)\d{3}( ?)\d{3}\b"));
/// ISO 7064 MOD 11,10; no leading zero; in the first ten digits one digit
/// appears two or three times and every other at most once.
pub fn de_tax_id_valid(n: &str) -> bool {
let d = digit_values(n);
if d.len() != 11 || d[0] == 0 {
return false;
}
let mut counts = [0u8; 10];
for &x in &d[..10] {
counts[x as usize] += 1;
}
let repeated = counts.iter().filter(|&&c| c >= 2).count();
if repeated != 1 || counts.iter().any(|&c| c > 3) {
return false;
}
let mut product = 10;
for &x in &d[..10] {
let mut sum = (x + product) % 10;
if sum == 0 {
sum = 10;
}
product = (2 * sum) % 11;
}
let check = match 11 - product {
10 => 0,
c => c,
};
check == d[10]
}
const DE_TAX_WORDS: &[&str] = &[
"steuer-id",
"steueridentifikationsnummer",
"steuerliche identifikationsnummer",
"idnr",
"identifikationsnummer",
"tax id",
];
fn de_tax_id(text: &str, findings: &mut Findings) {
for c in DE_TAX.captures_iter(text) {
let whole = c.get(0).unwrap();
let written = [&c[1], &c[2], &c[3]].iter().all(|s| *s == " ");
let n: String = whole.as_str().replace(' ', "");
if de_tax_id_valid(&n)
&& (written || word_near(text, whole.start(), whole.end(), DE_TAX_WORDS))
{
findings.insert(n);
}
}
}
/// The ID card's document number: a letter from the card's alphabet, eight
/// more characters from it, then the check digit.
static DE_ID: LazyLock<Regex> =
LazyLock::new(|| re(r"\b[CFGHJKLMNPRTVWXYZ][CFGHJKLMNPRTVWXYZ0-9]{8}\d\b"));
/// ICAO 9303 check digit: weights 7, 3, 1; letters A=10 … Z=35.
pub fn icao_check(chars: &str, check: u32) -> bool {
let value = |c: char| c.to_digit(10).unwrap_or_else(|| c as u32 - 'A' as u32 + 10);
let sum: u32 = chars
.chars()
.zip([7, 3, 1].iter().cycle())
.map(|(c, w)| value(c) * w)
.sum();
sum % 10 == check
}
fn de_id_card(text: &str, findings: &mut Findings) {
for m in DE_ID.find_iter(text) {
let s = m.as_str();
if icao_check(&s[..9], num(&s[9..])) {
findings.insert(s);
}
}
}
// --- France ---------------------------------------------------------------
/// Sex, year, month, department (with Corsica's 2A and 2B), commune, order,
/// then the two-digit key, spaces allowed between groups.
static FR_NIR: LazyLock<Regex> = LazyLock::new(|| {
re(r"\b([1-478]) ?(\d{2}) ?(\d{2}) ?(\d{2}|2[AB]) ?(\d{3}) ?(\d{3}) ?(\d{2})\b")
});
fn fr_nir(text: &str, findings: &mut Findings) {
for c in FR_NIR.captures_iter(text) {
let month = num(&c[3]);
if !(matches!(month, 1..=12 | 20..=42 | 50..=99)) {
continue;
}
let department = match &c[4] {
"2A" => "19",
"2B" => "18",
d => d,
};
let body = format!(
"{}{}{}{}{}{}",
&c[1], &c[2], &c[3], department, &c[5], &c[6]
);
let Ok(value) = body.parse::<u64>() else {
continue;
};
if 97 - value % 97 == u64::from(num(&c[7])) {
findings.insert(format!(
"{}{}{}{}{}{}{}",
&c[1], &c[2], &c[3], &c[4], &c[5], &c[6], &c[7]
));
}
}
}
// --- Spain ----------------------------------------------------------------
static ES_ID: LazyLock<Regex> = LazyLock::new(|| re(r"(?i)\b([XYZ]?)[ -]?(\d{7,8})[ -]?([A-Z])\b"));
const DNI_LETTERS: &[u8] = b"TRWAGMYFPDXBNJZSQVHLCKE";
fn es_dni_nie(text: &str, findings: &mut Findings) {
for c in ES_ID.captures_iter(text) {
let prefix = c[1].to_ascii_uppercase();
let digits = &c[2];
// DNI: eight digits; NIE: X, Y or Z and seven digits
let number = match (prefix.as_str(), digits.len()) {
("", 8) => digits.to_string(),
("X", 7) => format!("0{digits}"),
("Y", 7) => format!("1{digits}"),
("Z", 7) => format!("2{digits}"),
_ => continue,
};
let letter = c[3].to_ascii_uppercase();
if DNI_LETTERS[(num(&number) % 23) as usize] == letter.as_bytes()[0] {
findings.insert(format!("{prefix}{digits}{letter}"));
}
}
}
// --- Italy ----------------------------------------------------------------
/// Surname and name letters, year, month letter, day, place code, check
/// letter; digits may be replaced by letters (omocodia).
static IT_CF: LazyLock<Regex> = LazyLock::new(|| {
let d = "[0-9LMNPQRSTUV]";
re(&format!(
r"(?i)\b[A-Z]{{6}}{d}{{2}}[ABCDEHLMPRST]{d}{{2}}[A-Z]{d}{{3}}[A-Z]\b"
))
});
/// The Ministry's odd-position values for 0–9 and A–Z.
const CF_ODD: [u32; 36] = [
1, 0, 5, 7, 9, 13, 15, 17, 19, 21, // 0-9
1, 0, 5, 7, 9, 13, 15, 17, 19, 21, 2, 4, 18, 20, 11, 3, 6, 8, 12, 14, 16, 10, 22, 25, 24,
23, // A-Z
];
pub fn codice_fiscale_valid(cf: &str) -> bool {
let index = |c: u8| {
if c.is_ascii_digit() {
(c - b'0') as usize
} else {
(c - b'A') as usize + 10
}
};
let even = |c: u8| {
if c.is_ascii_digit() {
u32::from(c - b'0')
} else {
u32::from(c - b'A')
}
};
let bytes = cf.as_bytes();
let sum: u32 = bytes[..15]
.iter()
.enumerate()
.map(|(i, &c)| {
if i % 2 == 0 {
CF_ODD[index(c)]
} else {
even(c)
}
})
.sum();
u32::from(bytes[15] - b'A') == sum % 26
}
fn it_codice_fiscale(text: &str, findings: &mut Findings) {
for m in IT_CF.find_iter(text) {
let cf = m.as_str().to_ascii_uppercase();
if codice_fiscale_valid(&cf) {
findings.insert(cf);
}
}
}
// --- Netherlands ----------------------------------------------------------
/// Nine digits, sometimes written `1112.22.333`.
static NL_BSN: LazyLock<Regex> = LazyLock::new(|| re(r"\b(\d{4})(\.?)(\d{2})(\.?)(\d{3})\b"));
/// The eleven test: weights 9 down to 2, and −1 for the last digit.
pub fn bsn_valid(n: &str) -> bool {
let d = digit_values(n);
let sum: i64 = d[..8]
.iter()
.zip((2..=9).rev())
.map(|(a, w)| i64::from(a * w))
.sum::<i64>()
- i64::from(d[8]);
sum != 0 && sum % 11 == 0
}
const BSN_WORDS: &[&str] = &[
"bsn",
"burgerservicenummer",
"sofinummer",
"sofi-nummer",
"citizen service number",
];
fn nl_bsn(text: &str, findings: &mut Findings) {
for c in NL_BSN.captures_iter(text) {
let whole = c.get(0).unwrap();
let n = format!("{}{}{}", &c[1], &c[3], &c[5]);
let written = &c[2] == "." && &c[4] == ".";
if bsn_valid(&n) && (written || word_near(text, whole.start(), whole.end(), BSN_WORDS)) {
findings.insert(n);
}
}
}
// --- Belgium --------------------------------------------------------------
/// `YY.MM.DD-XXX.CC` or eleven digits.
static BE_NN: LazyLock<Regex> =
LazyLock::new(|| re(r"\b(\d{2})\.?(\d{2})\.?(\d{2})-?(\d{3})\.?(\d{2})\b"));
fn be_national_number(text: &str, findings: &mut Findings) {
for c in BE_NN.captures_iter(text) {
let (month, day) = (num(&c[2]), num(&c[3]));
// Month 0 and day 0 mean unknown; bis numbers add 20 or 40 to the month
if !(month <= 12 || (20..=32).contains(&month) || (40..=52).contains(&month)) || day > 31 {
continue;
}
let body = format!("{}{}{}{}", &c[1], &c[2], &c[3], &c[4]);
let check = u64::from(num(&c[5]));
let before_2000 = 97 - body.parse::<u64>().unwrap_or(0) % 97;
let since_2000 = 97 - format!("2{body}").parse::<u64>().unwrap_or(0) % 97;
if check == before_2000 || check == since_2000 {
findings.insert(format!("{body}{}", &c[5]));
}
}
}
// --- Poland ---------------------------------------------------------------
static ELEVEN: LazyLock<Regex> = LazyLock::new(|| re(r"\b\d{11}\b"));
/// Weights 1, 3, 7, 9 repeating; the birth date encodes the century in the
/// month (+80 for the 1800s, +20 for the 2000s, and so on).
pub fn pesel_valid(n: &str) -> bool {
let d = digit_values(n);
let sum: u32 = d[..10]
.iter()
.zip([1, 3, 7, 9].iter().cycle())
.map(|(a, w)| a * w)
.sum();
let month = d[2] * 10 + d[3];
let (century, month) = match month {
81..=92 => (1800, month - 80),
1..=12 => (1900, month),
21..=32 => (2000, month - 20),
41..=52 => (2100, month - 40),
_ => return false,
};
let year = century + d[0] * 10 + d[1];
(10 - sum % 10) % 10 == d[10] && (1..=super::days_in(year, month)).contains(&(d[4] * 10 + d[5]))
}
const PESEL_WORDS: &[&str] = &["pesel", "numer pesel", "nr pesel"];
fn pl_pesel(text: &str, findings: &mut Findings) {
for m in ELEVEN.find_iter(text) {
if pesel_valid(m.as_str()) && word_near(text, m.start(), m.end(), PESEL_WORDS) {
findings.insert(m.as_str());
}
}
}
// --- Sweden ---------------------------------------------------------------
/// `YYMMDD-NNNN`, `YYYYMMDD-NNNN` (`+` after 100), or the bare digits.
static SE_PNR: LazyLock<Regex> =
LazyLock::new(|| re(r"\b(?:\d{2})?(\d{2})(\d{2})(\d{2})([-+]?)(\d{4})\b"));
const SE_WORDS: &[&str] = &[
"personnummer",
"personnr",
"person nr",
"samordningsnummer",
"pnr",
];
fn se_personnummer(text: &str, findings: &mut Findings) {
for c in SE_PNR.captures_iter(text) {
let whole = c.get(0).unwrap();
let (yy, month, day) = (num(&c[1]), num(&c[2]), num(&c[3]));
// Coordination numbers add 60 to the day
let day = if day > 60 { day - 60 } else { day };
let ten = format!("{}{}{}{}", &c[1], &c[2], &c[3], &c[5]);
let written = !c[4].is_empty();
if valid_short_date(yy, month, day)
&& checks::luhn(&ten)
&& (written || word_near(text, whole.start(), whole.end(), SE_WORDS))
{
findings.insert(ten);
}
}
}
// --- Denmark --------------------------------------------------------------
static DK_CPR: LazyLock<Regex> = LazyLock::new(|| re(r"\b(\d{2})(\d{2})(\d{2})-?(\d{4})\b"));
const CPR_WORDS: &[&str] = &["cpr", "cpr-nr", "cpr nr", "cpr-nummer", "personnummer"];
fn dk_cpr(text: &str, findings: &mut Findings) {
for c in DK_CPR.captures_iter(text) {
let whole = c.get(0).unwrap();
if valid_short_date(num(&c[3]), num(&c[2]), num(&c[1]))
&& word_near(text, whole.start(), whole.end(), CPR_WORDS)
{
findings.insert(whole.as_str().replace('-', ""));
}
}
}
// --- Finland --------------------------------------------------------------
static FI_HETU: LazyLock<Regex> =
LazyLock::new(|| re(r"(?i)\b(\d{2})(\d{2})(\d{2})[-+ABCDEFYXWVU](\d{3})([0-9A-Y])\b"));
const HETU_CHECK: &[u8] = b"0123456789ABCDEFHJKLMNPRSTUVWXY";
fn fi_hetu(text: &str, findings: &mut Findings) {
for c in FI_HETU.captures_iter(text) {
let (day, month, yy) = (num(&c[1]), num(&c[2]), num(&c[3]));
let n: u64 = format!("{}{}{}{}", &c[1], &c[2], &c[3], &c[4])
.parse()
.unwrap_or(0);
let check = c[5].to_ascii_uppercase().as_bytes()[0];
if valid_short_date(yy, month, day) && HETU_CHECK[(n % 31) as usize] == check {
findings.insert(c[0].to_ascii_uppercase());
}
}
}
// --- Ireland --------------------------------------------------------------
static IE_PPS: LazyLock<Regex> = LazyLock::new(|| re(r"(?i)\b(\d{7})([A-W])([ABHW]?)\b"));
const PPS_CHECK: &[u8] = b"WABCDEFGHIJKLMNOPQRSTUV";
fn ie_pps(text: &str, findings: &mut Findings) {
for c in IE_PPS.captures_iter(text) {
let mut sum: u32 = digit_values(&c[1])
.iter()
.zip((2..=8).rev())
.map(|(a, w)| a * w)
.sum();
// The second letter counts, times 9; W (the old form) counts as 0
let second = c[3].to_ascii_uppercase();
if let Some(&letter) = second.as_bytes().first()
&& letter != b'W'
{
sum += u32::from(letter - b'A' + 1) * 9;
}
let check = c[2].to_ascii_uppercase().as_bytes()[0];
if PPS_CHECK[(sum % 23) as usize] == check {
findings.insert(c[0].to_ascii_uppercase());
}
}
}
// --- Portugal -------------------------------------------------------------
static NINE: LazyLock<Regex> = LazyLock::new(|| re(r"\b\d{9}\b"));
/// Mod 11 over weights 9 down to 2; a check of 10 or 11 becomes 0.
pub fn nif_valid(n: &str) -> bool {
let d = digit_values(n);
let sum: u32 = d[..8].iter().zip((2..=9).rev()).map(|(a, w)| a * w).sum();
let check = match 11 - sum % 11 {
10 | 11 => 0,
c => c,
};
matches!(d[0], 1 | 2 | 3 | 5 | 6 | 8 | 9) && check == d[8]
}
const NIF_WORDS: &[&str] = &[
"nif",
"contribuinte",
"número de identificação fiscal",
"numero de contribuinte",
];
fn pt_nif(text: &str, findings: &mut Findings) {
for m in NINE.find_iter(text) {
if nif_valid(m.as_str()) && word_near(text, m.start(), m.end(), NIF_WORDS) {
findings.insert(m.as_str());
}
}
}
// --- Austria --------------------------------------------------------------
/// A serial and check digit, then the birth date: `1237 010180`.
static AT_SVNR: LazyLock<Regex> = LazyLock::new(|| re(r"\b(\d{3})(\d)( ?)(\d{2})(\d{2})(\d{2})\b"));
const SVNR_WORDS: &[&str] = &[
"sozialversicherungsnummer",
"svnr",
"sv-nr",
"sv-nummer",
"versicherungsnummer",
];
fn at_svnr(text: &str, findings: &mut Findings) {
for c in AT_SVNR.captures_iter(text) {
let whole = c.get(0).unwrap();
let n = format!("{}{}{}{}{}", &c[1], &c[2], &c[4], &c[5], &c[6]);
let d = digit_values(&n);
let sum: u32 = d
.iter()
.zip([3, 7, 9, 0, 5, 8, 4, 2, 1, 6])
.map(|(a, w)| a * w)
.sum();
let written = &c[3] == " ";
if d[0] != 0
&& sum % 11 == d[3]
&& valid_short_date(num(&c[6]), num(&c[5]), num(&c[4]))
&& (written || word_near(text, whole.start(), whole.end(), SVNR_WORDS))
&& stands_alone(text, whole.start(), whole.end())
{
findings.insert(n);
}
}
}
#[cfg(test)]
mod tests {
use crate::mailflow::detectors::by_id;
fn count(id: &str, text: &str) -> usize {
by_id(id).unwrap().count(text)
}
#[test]
fn germany() {
assert_eq!(count("de-tax-id", "86 095 742 719"), 1);
assert_eq!(count("de-tax-id", "Steuer-ID: 86095742719"), 1);
assert_eq!(count("de-tax-id", "Rechnung 86095742719"), 0);
assert_eq!(count("de-tax-id", "86 095 742 718"), 0);
// ICAO 9303's German specimen card
assert_eq!(count("de-id-card", "Ausweis T220001293"), 1);
assert_eq!(count("de-id-card", "T220001294"), 0);
}
#[test]
fn france_spain_italy() {
assert_eq!(count("fr-nir", "2 55 08 14 168 025 38"), 1);
assert_eq!(count("fr-nir", "255081416802539"), 0);
assert_eq!(count("es-dni-nie", "DNI 12345678Z, NIE X-1234567-L"), 2);
assert_eq!(count("es-dni-nie", "12345678A"), 0);
assert_eq!(count("it-codice-fiscale", "CF: RSSMRA85T10A562S"), 1);
assert_eq!(count("it-codice-fiscale", "RSSMRA85T10A562T"), 0);
}
#[test]
fn benelux() {
assert_eq!(count("nl-bsn", "1112.22.333"), 1);
assert_eq!(count("nl-bsn", "BSN 111222333"), 1);
assert_eq!(count("nl-bsn", "order 111222333"), 0);
assert_eq!(count("nl-bsn", "BSN 111222334"), 0);
assert_eq!(count("be-national-number", "85.07.30-033.28"), 1);
assert_eq!(count("be-national-number", "85073003329"), 0);
}
#[test]
fn nordics() {
assert_eq!(count("se-personnummer", "811218-9876"), 1);
assert_eq!(count("se-personnummer", "811218-9875"), 0);
assert_eq!(count("se-personnummer", "order 8112189876"), 0);
assert_eq!(count("se-personnummer", "personnummer 198112189876"), 1);
assert_eq!(count("dk-cpr", "CPR-nr: 010170-1234"), 1);
assert_eq!(count("dk-cpr", "010170-1234"), 0);
assert_eq!(count("dk-cpr", "CPR 320170-1234"), 0);
assert_eq!(count("fi-hetu", "131052-308T"), 1);
assert_eq!(count("fi-hetu", "131052-308U"), 0);
}
#[test]
fn poland_ireland_portugal_austria() {
assert_eq!(count("pl-pesel", "PESEL 44051401359, pesel 02070803628"), 2);
assert_eq!(count("pl-pesel", "PESEL 44051401358"), 0);
assert_eq!(count("pl-pesel", "44051401359"), 0);
assert_eq!(count("ie-pps", "PPS 1234567T and 1234567FA"), 2);
assert_eq!(count("ie-pps", "1234567U"), 0);
assert_eq!(count("pt-nif", "NIF 123456789"), 1);
assert_eq!(count("pt-nif", "NIF 123456788"), 0);
assert_eq!(count("at-svnr", "1237 010180"), 1);
assert_eq!(count("at-svnr", "SVNR 1237010180"), 1);
assert_eq!(count("at-svnr", "1238 010180"), 0);
}
}
@@ -1,116 +0,0 @@
/*
* SPDX-FileCopyrightText: 2026 Coffey Labs LLC
*
* SPDX-License-Identifier: AGPL-3.0-only
*/
//! European identifiers outside the EU (§2.3): Norway's national identity
//! number and Switzerland's AHV number.
use super::{Detector, Findings, Region, Strength, digit_values, valid_short_date};
use regex::Regex;
use std::sync::LazyLock;
pub static DETECTORS: &[Detector] = &[
Detector::new(
"no-fnr",
"Norway: national identity number",
Region::Europe,
Strength::Checked,
no_fnr,
),
Detector::new(
"ch-ahv",
"Switzerland: AHV number",
Region::Europe,
Strength::Checked,
ch_ahv,
),
];
static ELEVEN: LazyLock<Regex> =
LazyLock::new(|| Regex::new(r"\b\d{6} ?\d{5}\b").expect("detector pattern"));
/// Two mod 11 check digits over a birth date (D-numbers add 40 to the day,
/// H-numbers 40 to the month): strong enough to count alone.
pub fn fnr_valid(n: &str) -> bool {
let d = digit_values(n);
if d.len() != 11 {
return false;
}
let check =
|weights: &[u32]| match 11 - d.iter().zip(weights).map(|(a, w)| a * w).sum::<u32>() % 11 {
11 => Some(0),
10 => None,
c => Some(c),
};
let day = d[0] * 10 + d[1];
let month = d[2] * 10 + d[3];
let day = if day > 40 { day - 40 } else { day };
let month = if month > 40 { month - 40 } else { month };
valid_short_date(d[4] * 10 + d[5], month, day)
&& check(&[3, 7, 6, 1, 8, 9, 4, 5, 2]) == Some(d[9])
&& check(&[5, 4, 3, 2, 7, 6, 5, 4, 3, 2]) == Some(d[10])
}
fn no_fnr(text: &str, findings: &mut Findings) {
for m in ELEVEN.find_iter(text) {
let n = m.as_str().replace(' ', "");
if fnr_valid(&n) {
findings.insert(n);
}
}
}
/// `756.1234.5678.97`: the country prefix, then an EAN-13 check digit.
static AHV: LazyLock<Regex> = LazyLock::new(|| {
Regex::new(r"\b756[. ]?\d{4}[. ]?\d{4}[. ]?\d{2}\b").expect("detector pattern")
});
pub fn ean13_valid(n: &str) -> bool {
let d = digit_values(n);
if d.len() != 13 {
return false;
}
let sum: u32 = d[..12]
.iter()
.enumerate()
.map(|(i, x)| if i % 2 == 0 { *x } else { x * 3 })
.sum();
(10 - sum % 10) % 10 == d[12]
}
fn ch_ahv(text: &str, findings: &mut Findings) {
for m in AHV.find_iter(text) {
let n: String = m.as_str().chars().filter(char::is_ascii_digit).collect();
if ean13_valid(&n) {
findings.insert(n);
}
}
}
#[cfg(test)]
mod tests {
use crate::mailflow::detectors::by_id;
fn count(id: &str, text: &str) -> usize {
by_id(id).unwrap().count(text)
}
#[test]
fn norway() {
assert_eq!(count("no-fnr", "01019000083"), 1);
assert_eq!(count("no-fnr", "010190 00083"), 1);
assert_eq!(count("no-fnr", "01019000084"), 0);
// Not a date
assert_eq!(count("no-fnr", "32019000083"), 0);
}
#[test]
fn switzerland() {
// The federal example
assert_eq!(count("ch-ahv", "AHV 756.9217.0769.85"), 1);
assert_eq!(count("ch-ahv", "7569217076985"), 1);
assert_eq!(count("ch-ahv", "756.9217.0769.86"), 0);
}
}
@@ -1,278 +0,0 @@
/*
* SPDX-FileCopyrightText: 2026 Coffey Labs LLC
*
* SPDX-License-Identifier: AGPL-3.0-only
*/
//! Detectors (dlp-and-mail-flow-rules spec, §2.3): each finds one kind of
//! identifier in text and reports the distinct ones it found.
//!
//! A detector is one of two strengths:
//!
//! - **Checked**: the identifier carries a published check digit or
//! checksum, so a random number rarely passes; found on its own.
//! - **Needs a word**: the format alone is too common, so a candidate counts
//! only with a corroborating word within [`WINDOW`] characters either
//! side.
//!
//! Findings are distinct normalized values (digits only, upper case), so the
//! same card number pasted twice counts once. They stay in memory: callers
//! read only [`Findings::len`].
pub mod africa;
pub mod americas;
pub mod any;
pub mod asia;
pub mod australia;
pub mod canada;
pub mod checks;
pub mod eu;
pub mod europe;
pub mod templates;
pub mod uk;
pub mod us;
use ahash::AHashSet;
/// How far, in characters, a corroborating word may be from a candidate.
pub const WINDOW: usize = 50;
#[derive(Debug, Clone, Copy, PartialEq, Eq)]
pub enum Strength {
Checked,
NeedsWord,
}
#[derive(Debug, Clone, Copy, PartialEq, Eq)]
pub enum Region {
Any,
Us,
Uk,
Canada,
Australia,
Eu,
Europe,
Asia,
Americas,
Africa,
}
/// The distinct values one detector found.
#[derive(Debug, Default)]
pub struct Findings(AHashSet<String>);
impl Findings {
pub fn insert(&mut self, value: impl Into<String>) {
self.0.insert(value.into());
}
pub fn len(&self) -> usize {
self.0.len()
}
pub fn is_empty(&self) -> bool {
self.0.is_empty()
}
}
pub struct Detector {
/// Stable id, stored in rules: `payment-card`, `iban`, `us-ssn`.
pub id: &'static str,
pub name: &'static str,
pub region: Region,
pub strength: Strength,
find: fn(&str, &mut Findings),
}
impl Detector {
pub const fn new(
id: &'static str,
name: &'static str,
region: Region,
strength: Strength,
find: fn(&str, &mut Findings),
) -> Self {
Self {
id,
name,
region,
strength,
find,
}
}
/// Adds what this detector finds in `text` to `findings`. Call once per
/// piece of text (subject, each part, each attachment) with the same
/// `findings`, then read its length.
pub fn find(&self, text: &str, findings: &mut Findings) {
(self.find)(text, findings)
}
/// The distinct values found in one text.
pub fn count(&self, text: &str) -> usize {
let mut findings = Findings::default();
self.find(text, &mut findings);
findings.len()
}
}
/// Every detector, in the order the console lists them.
pub fn all() -> impl Iterator<Item = &'static Detector> {
[
any::DETECTORS,
us::DETECTORS,
uk::DETECTORS,
canada::DETECTORS,
australia::DETECTORS,
eu::DETECTORS,
europe::DETECTORS,
asia::DETECTORS,
americas::DETECTORS,
africa::DETECTORS,
]
.into_iter()
.flatten()
}
pub fn by_id(id: &str) -> Option<&'static Detector> {
all().find(|detector| detector.id == id)
}
/// Whether one of `words` appears, as a whole word and ignoring case, within
/// [`WINDOW`] characters before `start` or after `end` (byte offsets of the
/// candidate in `text`). The window is widened by the longest word, so a
/// word that reaches into it still counts whole.
pub fn word_near(text: &str, start: usize, end: usize, words: &[&str]) -> bool {
let reach = WINDOW + words.iter().map(|w| w.chars().count()).max().unwrap_or(0);
let before = text[..start]
.char_indices()
.rev()
.nth(reach - 1)
.map_or(0, |(i, _)| i);
let after = text[end..]
.char_indices()
.nth(reach)
.map_or(text.len(), |(i, _)| end + i);
let window = text[before..after].to_lowercase();
words.iter().any(|word| contains_word(&window, word))
}
/// Whether `word` (lower case) appears in `haystack` (lower case) with no
/// letter or digit on either side.
pub fn contains_word(haystack: &str, word: &str) -> bool {
haystack.match_indices(word).any(|(i, _)| {
let before_ok = haystack[..i]
.chars()
.next_back()
.is_none_or(|c| !c.is_alphanumeric());
let after_ok = haystack[i + word.len()..]
.chars()
.next()
.is_none_or(|c| !c.is_alphanumeric());
before_ok && after_ok
})
}
/// Whether the match at `start..end` stands alone: no digit or letter
/// directly before or after it, so `123-45-6789` isn't found inside a
/// longer run of digits.
pub fn stands_alone(text: &str, start: usize, end: usize) -> bool {
let before = text[..start].chars().next_back();
let after = text[end..].chars().next();
before.is_none_or(|c| !c.is_alphanumeric()) && after.is_none_or(|c| !c.is_alphanumeric())
}
/// Days in `month` of `year` (0 for a month that doesn't exist).
pub fn days_in(year: u32, month: u32) -> u32 {
match month {
1 | 3 | 5 | 7 | 8 | 10 | 12 => 31,
4 | 6 | 9 | 11 => 30,
2 if year.is_multiple_of(4) && (!year.is_multiple_of(100) || year.is_multiple_of(400)) => {
29
}
2 => 28,
_ => 0,
}
}
/// Whether `year`-`month`-`day` is a real date between 1900 and 2100.
pub fn valid_date(year: u32, month: u32, day: u32) -> bool {
(1900..=2100).contains(&year) && (1..=days_in(year, month)).contains(&day)
}
/// Whether a two-digit year, month and day make a real date in either the
/// 1900s or the 2000s.
pub fn valid_short_date(yy: u32, month: u32, day: u32) -> bool {
valid_date(1900 + yy, month, day) || valid_date(2000 + yy, month, day)
}
/// The value of each digit in `s`.
pub fn digit_values(s: &str) -> Vec<u32> {
s.bytes()
.filter(u8::is_ascii_digit)
.map(|b| u32::from(b - b'0'))
.collect()
}
/// The ASCII digits of `s`.
pub fn digits(s: &str) -> String {
s.chars().filter(char::is_ascii_digit).collect()
}
#[cfg(test)]
mod tests {
use super::*;
#[test]
fn words_are_whole_and_near() {
let text = "Your passport number is X1234567, thanks";
let start = text.find("X123").unwrap();
assert!(word_near(text, start, start + 8, &["passport"]));
assert!(!word_near(text, start, start + 8, &["pass"]));
let far = format!("passport{}X1234567", " ".repeat(60));
let start = far.find("X123").unwrap();
assert!(!word_near(&far, start, start + 8, &["passport"]));
}
#[test]
fn near_counts_characters_not_bytes() {
// 45 two-byte characters between the word and the candidate: within
// 50 characters, though over 50 bytes
let text = format!("passport {} X1234567", "é".repeat(45));
let start = text.find("X123").unwrap();
assert!(word_near(&text, start, start + 8, &["passport"]));
}
/// An ordinary business email: order, invoice and tracking numbers,
/// dates, amounts, a street address. Nothing here is an identifier, so
/// no detector may fire, except the contact ones on the signature.
#[test]
fn ordinary_mail_finds_nothing() {
let text = "Hi Dana,\n\nThanks for order 4471-2290 placed 2026-09-14. Invoice INV-2026-00917 \
for $12,480.00 is due 10/31/2026; PO 7731902 covers lines 1-14. Tracking \
1Z999AA10123456784, parcel 3 of 5, 12.5 kg, box 40x30x20 cm. Meeting moved to \
Tuesday 9:30-10:15 in room 2B, building 1177. Ticket #5520318, case 20260914-0042. \
Version 2026.9.28.4, build 118822, commit 5a73a118. Serial SN-88213-X. \
Ship to 1600 Amphitheatre Pkwy, Mountain View, CA 94043. Revenue grew 18% to \
1,204,332 units; see figures 3.1-3.4 and table 12.\n\nBest,\nSam\n\
Sam Rivera | +1 (415) 555-2671 | sam@example.com";
let quiet = ["email-addresses", "phone-numbers"];
for detector in all().filter(|d| !quiet.contains(&d.id)) {
assert_eq!(
detector.count(text),
0,
"{} fired on ordinary mail",
detector.id
);
}
}
#[test]
fn ids_are_unique() {
let mut seen = AHashSet::new();
for detector in all() {
assert!(seen.insert(detector.id), "duplicate id {}", detector.id);
assert!(by_id(detector.id).is_some());
}
}
}
@@ -1,97 +0,0 @@
/*
* SPDX-FileCopyrightText: 2026 Coffey Labs LLC
*
* SPDX-License-Identifier: AGPL-3.0-only
*/
//! Templates (§2.3): named sets of detectors, so a policy doesn't pick forty
//! one at a time. Each is named for what it finds, never for a law, and is a
//! starting point: once added to a rule, its detectors can be changed.
pub struct Template {
pub id: &'static str,
pub name: &'static str,
pub detectors: &'static [&'static str],
}
pub static TEMPLATES: &[Template] = &[
Template {
id: "payment-and-bank",
name: "Payment cards and bank accounts",
detectors: &["payment-card", "iban", "swift-bic", "us-aba-routing"],
},
Template {
id: "us-personal",
name: "US personal identifiers",
detectors: &[
"us-ssn",
"us-itin",
"us-ein",
"us-drivers-license",
"passport",
"date-of-birth",
],
},
Template {
id: "uk-personal",
name: "UK personal identifiers",
detectors: &["uk-nino", "uk-utr", "uk-nhs", "passport", "date-of-birth"],
},
Template {
id: "eu-national",
name: "EU national identifiers",
detectors: &[
"de-tax-id",
"de-id-card",
"fr-nir",
"es-dni-nie",
"it-codice-fiscale",
"nl-bsn",
"be-national-number",
"pl-pesel",
"se-personnummer",
"dk-cpr",
"fi-hetu",
"ie-pps",
"pt-nif",
"at-svnr",
],
},
Template {
id: "health",
name: "Health identifiers",
detectors: &["uk-nhs", "us-mbi", "us-npi", "us-dea", "au-medicare"],
},
Template {
id: "credentials",
name: "Credentials and keys",
detectors: &["private-key", "credentials"],
},
Template {
id: "contact-lists",
name: "Contact lists",
detectors: &["email-addresses", "phone-numbers"],
},
];
pub fn by_id(id: &str) -> Option<&'static Template> {
TEMPLATES.iter().find(|template| template.id == id)
}
#[cfg(test)]
mod tests {
use super::*;
#[test]
fn every_template_names_real_detectors() {
for template in TEMPLATES {
for id in template.detectors {
assert!(
super::super::by_id(id).is_some(),
"{}: no detector {id}",
template.id
);
}
}
}
}
@@ -1,155 +0,0 @@
/*
* SPDX-FileCopyrightText: 2026 Coffey Labs LLC
*
* SPDX-License-Identifier: AGPL-3.0-only
*/
//! United Kingdom identifiers (§2.3): HMRC's National Insurance number and
//! Unique Taxpayer Reference, and the NHS number.
use super::{Detector, Findings, Region, Strength, word_near};
use regex::Regex;
use std::sync::LazyLock;
pub static DETECTORS: &[Detector] = &[
Detector::new(
"uk-nino",
"UK National Insurance number",
Region::Uk,
Strength::Checked,
nino,
),
Detector::new(
"uk-nhs",
"UK NHS number",
Region::Uk,
Strength::Checked,
nhs,
),
Detector::new(
"uk-utr",
"UK Unique Taxpayer Reference",
Region::Uk,
Strength::NeedsWord,
utr,
),
];
fn re(pattern: &str) -> Regex {
Regex::new(pattern).expect("detector pattern")
}
/// Two letters, six digits (often in pairs), a suffix A–D.
static NINO: LazyLock<Regex> =
LazyLock::new(|| re(r"(?i)\b([A-Z])([A-Z]) ?(\d{2}) ?(\d{2}) ?(\d{2}) ?([A-D])\b"));
/// HMRC's rules: D, F, I, Q, U and V are never used; O never second; and
/// BG, GB, KN, NK, NT, TN and ZZ are never allocated.
fn nino_prefix(first: char, second: char) -> bool {
const NEVER: &str = "DFIQUV";
let pair: String = [first, second].iter().collect();
!NEVER.contains(first)
&& !NEVER.contains(second)
&& second != 'O'
&& !["BG", "GB", "KN", "NK", "NT", "TN", "ZZ"].contains(&pair.as_str())
}
fn nino(text: &str, findings: &mut Findings) {
for c in NINO.captures_iter(text) {
let first = c[1].to_ascii_uppercase().chars().next().unwrap();
let second = c[2].to_ascii_uppercase().chars().next().unwrap();
if nino_prefix(first, second) {
findings.insert(format!(
"{first}{second}{}{}{}{}",
&c[3],
&c[4],
&c[5],
c[6].to_ascii_uppercase()
));
}
}
}
/// `NNN NNN NNNN` stands alone; ten bare digits need a word.
static NHS: LazyLock<Regex> = LazyLock::new(|| re(r"\b(\d{3})([ -]?)(\d{3})([ -]?)(\d{4})\b"));
/// Mod 11: weights 10 down to 2 over the first nine digits; the check digit
/// is 11 minus the remainder (11 becomes 0; 10 is never issued).
pub fn nhs_valid(n: &str) -> bool {
let d: Vec<u32> = n.bytes().map(|b| u32::from(b - b'0')).collect();
let sum: u32 = d[..9].iter().zip((2..=10).rev()).map(|(a, w)| a * w).sum();
match 11 - sum % 11 {
11 => d[9] == 0,
10 => false,
check => d[9] == check,
}
}
const NHS_WORDS: &[&str] = &["nhs", "nhs number", "nhs no"];
fn nhs(text: &str, findings: &mut Findings) {
for c in NHS.captures_iter(text) {
let whole = c.get(0).unwrap();
let n = format!("{}{}{}", &c[1], &c[3], &c[5]);
let written = !c[2].is_empty() && c[2] == c[4];
if nhs_valid(&n) && (written || word_near(text, whole.start(), whole.end(), NHS_WORDS)) {
findings.insert(n);
}
}
}
static UTR: LazyLock<Regex> = LazyLock::new(|| re(r"\b\d{5} ?\d{5}\b"));
const UTR_WORDS: &[&str] = &[
"utr",
"unique taxpayer reference",
"tax reference",
"self assessment",
];
fn utr(text: &str, findings: &mut Findings) {
for m in UTR.find_iter(text) {
if word_near(text, m.start(), m.end(), UTR_WORDS) {
findings.insert(m.as_str().replace(' ', ""));
}
}
}
#[cfg(test)]
mod tests {
use crate::mailflow::detectors::by_id;
fn count(id: &str, text: &str) -> usize {
by_id(id).unwrap().count(text)
}
#[test]
fn national_insurance() {
assert_eq!(count("uk-nino", "NI: AB 12 34 56 C, ce123456d"), 2);
// Letters never used, pairs never allocated, a suffix past D
for bad in [
"QQ123456C",
"AO123456C",
"GB123456A",
"AB123456E",
"DA123456A",
] {
assert_eq!(count("uk-nino", bad), 0, "{bad}");
}
}
#[test]
fn nhs_numbers() {
// The NHS's own example
assert_eq!(count("uk-nhs", "943 476 5919"), 1);
assert_eq!(count("uk-nhs", "943 476 5918"), 0);
assert_eq!(count("uk-nhs", "order 9434765919"), 0);
assert_eq!(count("uk-nhs", "NHS number 9434765919"), 1);
}
#[test]
fn utr() {
assert_eq!(count("uk-utr", "UTR 12345 67890"), 1);
assert_eq!(count("uk-utr", "order 1234567890"), 0);
}
}
Loaded 100 of 404 files, more files were not shown because too many files have changed in this diff. Show more