Compare commits

..
Author SHA1 Message Date
jcoffey-dev 77404353d5 Merge pull request 'Release 2026.9.28.6' (#107) from hotfix/2026.9.28.6 into release/2026.9.28.6
publish / version (push) Successful in 11s
publish / publish-amd64 (push) Successful in 31m53s
publish / release (push) Successful in 1s
publish / publish-arm64 (push) Successful in 50m13s
publish / binaries (push) Successful in 42s
publish / announce (push) Successful in 22s
2026-09-29 01:29:48 +00:00
jcoffey-dev d2b0750e1d Release 2026.9.28.6
ci / fork-checks (pull_request) Successful in 45s
ci / build (pull_request) Successful in 6m10s
2026-09-28 18:23:26 -07:00
jcoffey-dev 0232e59283 Ports: each node checks the others' ports from outside 2026-09-28 18:16:33 -07:00
jcoffey-dev 9fd1d32799 Explain: don't prepare answers for date fields 2026-09-28 18:16:33 -07:00
jcoffey-dev eafe8bfbfd Webhooks: send one sample event to a saved webhook 2026-09-28 18:16:33 -07:00
404 changed files with 1464 additions and 24745 deletions

No files matched your search

+1 -44
View File
@@ -7,12 +7,7 @@
# instance resolves short `uses:` against itself, never GitHub, so nothing # instance resolves short `uses:` against itself, never GitHub, so nothing
# unreviewed can be pulled in. # unreviewed can be pulled in.
# #
# BUILD_ON: when the Actions variable BUILD_ON is 'github' (org or repo), # Not ported, as on GitLab: publish.yml and release.yml still need doing.
# fork-checks and build skip here and the `github` job below waits for the
# same work done by .github/workflows/ci.yml on the GitHub mirror, passing or
# failing with it -- so this run still carries the answer pull requests and
# merges look at. Unset, everything builds here as before. If GitHub is
# unavailable, unset BUILD_ON and nothing else has to change.
name: ci name: ci
on: on:
@@ -30,7 +25,6 @@ jobs:
# without the AGPL 5(a) notice. Seconds, and needs no toolchain. The notice # without the AGPL 5(a) notice. Seconds, and needs no toolchain. The notice
# check diffs against the upstream snapshot branch, hence the full fetch. # check diffs against the upstream snapshot branch, hence the full fetch.
fork-checks: fork-checks:
if: ${{ vars.BUILD_ON != 'github' }}
runs-on: light runs-on: light
container: container:
image: python:3.13-slim@sha256:8d9d0b8bcf6506481eae4907c18f5e3e7902e629f5f6d684f9e7c32e85e3ddf0 # 3.13-slim image: python:3.13-slim@sha256:8d9d0b8bcf6506481eae4907c18f5e3e7902e629f5f6d684f9e7c32e85e3ddf0 # 3.13-slim
@@ -58,7 +52,6 @@ jobs:
run: python3 -m unittest discover -s tools/fork/tests run: python3 -m unittest discover -s tools/fork/tests
build: build:
if: ${{ vars.BUILD_ON != 'github' }}
# Either runner (host1 or host2): the build needs no docker socket. # Either runner (host1 or host2): the build needs no docker socket.
runs-on: light runs-on: light
container: container:
@@ -108,39 +101,3 @@ jobs:
used=$(du -s --block-size=1G /cache/target 2>/dev/null | cut -f1) used=$(du -s --block-size=1G /cache/target 2>/dev/null | cut -f1)
echo "target dir: ${used:-0} GB" echo "target dir: ${used:-0} GB"
if [ "${used:-0}" -gt 60 ]; then rm -rf /cache/target && echo "over 60 GB: target dir cleared"; fi if [ "${used:-0}" -gt 60 ]; then rm -rf /cache/target && echo "over 60 GB: target dir cleared"; fi
# BUILD_ON=github: the GitHub mirror builds this commit and posts the result
# back as the commit status "github/ci (branch)". This waits for that status
# and takes its answer. The mirror pushes on every commit, so a missing
# status means GitHub has not got the push or is not running: after the
# timeout this fails, which is the cue to unset BUILD_ON.
github:
if: ${{ vars.BUILD_ON == 'github' }}
# Its own runner label with plenty of slots: this job only polls, but holds a slot
# for as long as the GitHub build takes, and must not starve the build runners.
runs-on: wait
timeout-minutes: 150
container:
image: python:3.13-slim@sha256:8d9d0b8bcf6506481eae4907c18f5e3e7902e629f5f6d684f9e7c32e85e3ddf0 # 3.13-slim
steps:
- env:
TOKEN: ${{ secrets.GITHUB_TOKEN }}
SHA: ${{ github.event.pull_request.head.sha || github.sha }}
CONTEXT: github/ci (branch)
run: |
python3 - <<'EOF'
import json, os, time, urllib.request
url = (f"{os.environ['CI_SERVER_INTERNAL']}/api/v1/repos/{os.environ['GITHUB_REPOSITORY']}"
f"/commits/{os.environ['SHA']}/statuses?limit=50")
req = urllib.request.Request(url, headers={"Authorization": f"token {os.environ['TOKEN']}"})
ctx, last = os.environ["CONTEXT"], None
print(f"waiting for '{ctx}' on {os.environ['SHA']}", flush=True)
while True:
mine = [s for s in json.load(urllib.request.urlopen(req)) if s["context"] == ctx]
state = max(mine, key=lambda s: s["id"]) if mine else None
if state and state["status"] != last:
last = state["status"]; print(f"{ctx}: {last} {state.get('target_url', '')}", flush=True)
if last == "success": raise SystemExit(0)
if last in ("failure", "error"): raise SystemExit(1)
time.sleep(20)
EOF
+1 -54
View File
@@ -42,14 +42,6 @@
# #
# The push logs in with PACKAGE_TOKEN (jcoffey-dev, write:package): the job's # The push logs in with PACKAGE_TOKEN (jcoffey-dev, write:package): the job's
# own token is refused by the container registry. # own token is refused by the container registry.
#
# BUILD_ON: when the Actions variable BUILD_ON is 'github' (org or repo), every
# job here but the announcement skips, and the tag is published by
# .github/workflows/ci.yml on the GitHub mirror instead -- same guards, same
# tags, the same Release and binaries, created here through the API. The
# `github` job waits for that run's commit status, "github/ci (tag)", and the
# announcement follows it as it follows the binaries here. Unset, everything
# runs here as before.
name: publish name: publish
on: on:
@@ -58,7 +50,6 @@ on:
jobs: jobs:
version: version:
if: ${{ vars.BUILD_ON != 'github' }}
runs-on: light runs-on: light
container: container:
image: python:3.13-slim@sha256:8d9d0b8bcf6506481eae4907c18f5e3e7902e629f5f6d684f9e7c32e85e3ddf0 # 3.13-slim image: python:3.13-slim@sha256:8d9d0b8bcf6506481eae4907c18f5e3e7902e629f5f6d684f9e7c32e85e3ddf0 # 3.13-slim
@@ -97,7 +88,6 @@ jobs:
echo "version $V" echo "version $V"
publish-amd64: publish-amd64:
if: ${{ vars.BUILD_ON != 'github' }}
needs: [version] needs: [version]
runs-on: docker runs-on: docker
container: container:
@@ -138,7 +128,6 @@ jobs:
run: docker logout "$REGISTRY" || true run: docker logout "$REGISTRY" || true
publish-arm64: publish-arm64:
if: ${{ vars.BUILD_ON != 'github' }}
needs: [version, publish-amd64] needs: [version, publish-amd64]
runs-on: docker runs-on: docker
container: container:
@@ -173,46 +162,11 @@ jobs:
- if: always() - if: always()
run: docker logout "$REGISTRY" || true run: docker logout "$REGISTRY" || true
# BUILD_ON=github: waits for the GitHub mirror's run for this tag, which
# posts its result back as the commit status "github/ci (tag)", and takes
# its answer. Fails after the timeout if no answer comes.
github:
if: ${{ vars.BUILD_ON == 'github' }}
# Its own runner label with plenty of slots: this job only polls, but holds a slot
# for as long as the GitHub build takes, and must not starve the build runners.
runs-on: wait
timeout-minutes: 240
container:
image: python:3.13-slim@sha256:8d9d0b8bcf6506481eae4907c18f5e3e7902e629f5f6d684f9e7c32e85e3ddf0 # 3.13-slim
steps:
- env:
TOKEN: ${{ secrets.GITHUB_TOKEN }}
SHA: ${{ github.sha }}
CONTEXT: github/ci (tag)
run: |
python3 - <<'EOF'
import json, os, time, urllib.request
url = (f"{os.environ['CI_SERVER_INTERNAL']}/api/v1/repos/{os.environ['GITHUB_REPOSITORY']}"
f"/commits/{os.environ['SHA']}/statuses?limit=50")
req = urllib.request.Request(url, headers={"Authorization": f"token {os.environ['TOKEN']}"})
ctx, last = os.environ["CONTEXT"], None
print(f"waiting for '{ctx}' on {os.environ['SHA']}", flush=True)
while True:
mine = [s for s in json.load(urllib.request.urlopen(req)) if s["context"] == ctx]
state = max(mine, key=lambda s: s["id"]) if mine else None
if state and state["status"] != last:
last = state["status"]; print(f"{ctx}: {last} {state.get('target_url', '')}", flush=True)
if last == "success": raise SystemExit(0)
if last in ("failure", "error"): raise SystemExit(1)
time.sleep(20)
EOF
# The weekly release creates its Release (and so the tag) first; a tag # The weekly release creates its Release (and so the tag) first; a tag
# pushed by hand has none. Either way the tag ends up with exactly one # pushed by hand has none. Either way the tag ends up with exactly one
# Release, created once the amd64 image exists so its pull instructions # Release, created once the amd64 image exists so its pull instructions
# work; arm64 and the binaries follow. # work; arm64 and the binaries follow.
release: release:
if: ${{ vars.BUILD_ON != 'github' }}
needs: [version, publish-amd64] needs: [version, publish-amd64]
runs-on: light runs-on: light
container: container:
@@ -262,7 +216,6 @@ jobs:
# `docker create` does not start anything, so pulling an arm64 image on an # `docker create` does not start anything, so pulling an arm64 image on an
# amd64 runner and copying a file out of it needs no emulation. # amd64 runner and copying a file out of it needs no emulation.
binaries: binaries:
if: ${{ vars.BUILD_ON != 'github' }}
needs: [version, publish-arm64, release] needs: [version, publish-arm64, release]
runs-on: docker runs-on: docker
container: container:
@@ -336,14 +289,8 @@ jobs:
# The release above is made with the job's own token, and Gitea starts no # The release above is made with the job's own token, and Gitea starts no
# workflow for events the Actions bot causes -- announce.yml's # workflow for events the Actions bot causes -- announce.yml's
# 'on: release' never fires for it -- so announce it from here. # 'on: release' never fires for it -- so announce it from here.
#
# With BUILD_ON=github the release and binaries come from the GitHub run,
# so the announcement waits for the `github` job instead. The Release that
# run creates for a hand-pushed tag is made with a user token, so
# announce.yml fires for it too; discourse-release keeps one topic per tag.
announce: announce:
needs: [release, binaries, github] needs: [release, binaries]
if: ${{ always() && ((needs.release.result == 'success' && needs.binaries.result == 'success') || needs.github.result == 'success') }}
runs-on: light runs-on: light
steps: steps:
- uses: coffey-labs/actions/discourse-release@e9293996e2efa770839121fa8f8da93083f216be - uses: coffey-labs/actions/discourse-release@e9293996e2efa770839121fa8f8da93083f216be
+5
View File
@@ -5,6 +5,11 @@
version: 2 version: 2
updates: updates:
- package-ecosystem: "cargo" # See documentation for possible values
directory: "/" # Location of package manifests
schedule:
interval: "weekly"
# Enable version updates for GitHub Actions # Enable version updates for GitHub Actions
- package-ecosystem: "github-actions" - package-ecosystem: "github-actions"
# Workflow files stored in the default location of `.github/workflows` # Workflow files stored in the default location of `.github/workflows`
@@ -12,7 +12,7 @@ jobs:
runs-on: ubuntu-latest runs-on: ubuntu-latest
steps: steps:
- name: Close issues from non-allowed authors - name: Close issues from non-allowed authors
uses: actions/github-script@v9 uses: actions/github-script@v7
with: with:
script: | script: |
// Users allowed to open issues directly. All other authors will have // Users allowed to open issues directly. All other authors will have
@@ -18,7 +18,7 @@ jobs:
sparse-checkout-cone-mode: false sparse-checkout-cone-mode: false
- name: Close PRs from non-allowed authors - name: Close PRs from non-allowed authors
uses: actions/github-script@v9 uses: actions/github-script@v7
with: with:
script: | script: |
const fs = require('fs'); const fs = require('fs');
@@ -12,7 +12,7 @@ jobs:
runs-on: ubuntu-latest runs-on: ubuntu-latest
steps: steps:
- name: Post support portal redirect - name: Post support portal redirect
uses: actions/github-script@v9 uses: actions/github-script@v7
with: with:
script: | script: |
const discussion = context.payload.discussion; const discussion = context.payload.discussion;
+1 -1
View File
@@ -73,6 +73,6 @@ jobs:
# Upload the results to GitHub's code scanning dashboard (optional). # Upload the results to GitHub's code scanning dashboard (optional).
# Commenting out will disable upload of results to your repo's Code Scanning dashboard # Commenting out will disable upload of results to your repo's Code Scanning dashboard
- name: "Upload to code-scanning" - name: "Upload to code-scanning"
uses: github/codeql-action/[email protected]8.2 uses: github/codeql-action/[email protected]7.4
with: with:
sarif_file: results.sarif sarif_file: results.sarif
+1 -1
View File
@@ -36,6 +36,6 @@ jobs:
severity: 'CRITICAL,HIGH' severity: 'CRITICAL,HIGH'
- name: Upload Trivy scan results to GitHub Security tab - name: Upload Trivy scan results to GitHub Security tab
uses: github/codeql-action/[email protected]8.2 uses: github/codeql-action/[email protected]7.4
with: with:
sarif_file: 'trivy-results.sarif' sarif_file: 'trivy-results.sarif'
+42
View File
@@ -0,0 +1,42 @@
version: 2
updates:
# Cargo. One entry: the workspace has a single lockfile at the root, and
# ~30 manifests that upstream bumps on every release -- pointing entries at
# individual crates would find manifests with no lockfile beside them.
#
# Minor and patch arrive as one pull request a week. Majors are left out of
# the group on purpose: they are migrations rather than bumps, and each one
# deserves its own pull request and its own CI run.
- package-ecosystem: cargo
directory: "/"
schedule:
interval: weekly
day: tuesday
time: "09:00"
timezone: Etc/UTC
open-pull-requests-limit: 5
groups:
minor-and-patch:
update-types:
- minor
- patch
- package-ecosystem: github-actions
directory: "/"
schedule:
interval: weekly
day: tuesday
time: "09:00"
timezone: Etc/UTC
groups:
actions:
patterns:
- "*"
# The Dockerfiles pin their base images, so this is what keeps a published
# image off a stale base between releases.
- package-ecosystem: docker
directory: "/"
schedule:
interval: weekly
day: tuesday
time: "09:00"
timezone: Etc/UTC
+37 -468
View File
@@ -1,482 +1,51 @@
# CI and publishing on GitHub, for the repository Gitea mirrors here. # What CI can check without a mail server's worth of infrastructure.
# #
# Gitea (git.coffeylabs.org) is where this project lives: pull requests, # The build, and that every test target compiles. It deliberately does not
# issues, releases and the container registry are all there, and it pushes # *run* the test suites: the unit tests only build with the integration crate
# every branch and tag to this GitHub copy as it changes. GitHub's hosted # in the graph, because that is what switches on the `test_mode` features they
# runners are faster than the self-hosted ones -- and have native arm64 -- so # rely on (docs/spec/SPEC.md 2.2b), and the integration suites need a `STORE`,
# the building happens here, and the answer goes back to Gitea as a commit # fixed ports, and in most cases a container apiece (docs/spec/
# status that Gitea's own ci.yml / publish.yml wait on. # container-tests.md). Running them here would mean either a green tick that
# skipped everything, or a red one that means "the runner has no Redis".
# #
# One switch decides which side builds: the Actions variable BUILD_ON, set on # So this catches what it can honestly catch -- code that does not compile,
# both forges. BUILD_ON=github runs every job below and turns Gitea's heavy # including test code -- and the suites are run by hand, one at a time, as
# jobs into a wait for this one; anything else leaves Gitea building exactly # that page describes. If that changes, it changes because someone made the
# as before and every job here skips. If GitHub is ever unavailable, unset it # suites runnable unattended, not because CI started ignoring failures.
# on Gitea and nothing else has to change. name: CI
#
# Needs, as organization settings rather than anything in this file:
# variables BUILD_ON=github, REGISTRY (the Gitea container registry),
# GITEA_URL (the Gitea base URL)
# secret GITEA_TOKEN -- jcoffey-dev, write:repository + write:package:
# commit statuses, the release and its assets, the registry push
#
# There is no pull_request trigger: pull requests happen on Gitea, and their
# branch arrives here as an ordinary push. Branch pushes get what Gitea's
# ci.yml checks; v* tags get what its publish.yml does. Schedules (the weekly
# release, the upstream watch) and the release announcement stay on Gitea.
#
# Every `uses:` is pinned to a full commit SHA with the release in the
# trailing comment. A tag is a mutable pointer; do not "simplify" a pin back
# to one. Only GitHub's own actions and the three docker/* ones are used.
name: ci
on: on:
push: push:
branches: ['**'] branches: [main]
tags: ['**'] pull_request:
# Lets CI be run by hand against any ref, including one that predates a CI
# change, without pushing an empty commit to move it.
workflow_dispatch: workflow_dispatch:
# A newer push to a branch cancels the run for the older one, whose answer is # A second push to a branch cancels the run still going for the first: the
# about code nobody is looking at any more. A tag run is never cancelled: it # older run's answer is about code nobody is looking at any more.
# publishes.
concurrency: concurrency:
group: ci-${{ github.ref }} group: ci-${{ github.ref }}
cancel-in-progress: ${{ github.ref_type == 'branch' }} cancel-in-progress: true
permissions:
contents: read
env:
GITEA_URL: ${{ vars.GITEA_URL }}
# The Gitea status this run answers for. Gitea waits on the one matching
# its own event: "(branch)" from ci.yml, "(tag)" from publish.yml.
STATUS_CONTEXT: github/ci (${{ github.ref_type }})
jobs: jobs:
# Tells Gitea a run has started, so a pull request shows it as pending
# rather than missing while the build is still going.
start:
if: ${{ vars.BUILD_ON == 'github' }}
runs-on: ubuntu-latest
steps:
- env:
GITEA_TOKEN: ${{ secrets.GITEA_TOKEN }}
run: |
jq -n --arg c "$STATUS_CONTEXT" \
--arg u "$GITHUB_SERVER_URL/$GITHUB_REPOSITORY/actions/runs/$GITHUB_RUN_ID" \
'{state:"pending", context:$c, target_url:$u, description:"GitHub Actions"}' |
curl -fsS -o /dev/null -X POST -H "Authorization: token $GITEA_TOKEN" \
-H 'Content-Type: application/json' --data @- \
"$GITEA_URL/api/v1/repos/$GITHUB_REPOSITORY/statuses/$GITHUB_SHA"
# ----------------------------------------------------------- branches ------
# What an upstream merge can bring in or leave behind without a conflict:
# the upstream name in a new string literal, and a changed upstream file
# without the AGPL 5(a) notice. Seconds, and needs no toolchain. The notice
# check diffs against the upstream snapshot in the history, hence the full
# fetch.
fork-checks:
if: ${{ vars.BUILD_ON == 'github' && github.ref_type == 'branch' }}
runs-on: ubuntu-latest
steps:
- uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1
with:
fetch-depth: 0
- run: python3 tools/fork/name-check.py
- if: always()
run: python3 tools/fork/notice-check.py
# Cargo can patch a dependency to a directory in this repository, and
# the image builds from a context .dockerignore prunes to almost
# nothing. CI never sees the difference; a release does.
- if: always()
run: python3 tools/fork/context-check.py
# The personal-data catalog must classify every object and field the
# schema has, and name nothing that is gone.
- if: always()
run: python3 tools/fork/privacy-check.py
# The admin reads each expression field's allowed values and variables
# from the schema; they're generated from the registry and must match it.
- if: always()
run: python3 tools/fork/expr-schema.py --check
- if: always()
run: python3 -m unittest discover -s tools/fork/tests
# The build, and that every test target compiles. The suites are not run:
# they need a store, fixed ports and containers (docs/spec/
# container-tests.md), and are run by hand.
build: build:
if: ${{ vars.BUILD_ON == 'github' && github.ref_type == 'branch' }}
runs-on: ubuntu-latest runs-on: ubuntu-latest
env:
CARGO_INCREMENTAL: "0"
# Debug info is most of a dev target dir, and nothing here runs a
# debugger. Without it the dev and test builds fit the runner's disk and
# the cache below stays small enough to be worth restoring.
CARGO_PROFILE_DEV_DEBUG: "0"
CARGO_PROFILE_TEST_DEBUG: "0"
steps: steps:
# The hosted image carries toolchains this build never touches; a dev, # Every `uses:` here is pinned to a full commit SHA, with the release it
# test and release build of RocksDB and the workspace needs the room. # belongs to in the trailing comment. A tag is a mutable pointer, so
- run: | # trusting `@v7` is trusting every future version of that action,
sudo rm -rf /usr/share/dotnet /usr/local/lib/android /opt/ghc /opt/hostedtoolcache/CodeQL # including one pushed by whoever compromises the account. Dependabot
df -h / # updates both halves together -- do not "simplify" a pin back to a tag.
- uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1 - uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1
# Current stable, as Gitea's rust:1 image is. - uses: Swatinem/rust-cache@6323deb102c322ba6fcbdcafc7e3dddab59af2b6 # v2.9.2
- id: rust - name: System dependencies
run: | # foundationdb and the search backends are off by default, but the
rustup toolchain install stable --profile minimal # default feature set still links against the system's C libraries.
rustup default stable run: sudo apt-get update && sudo apt-get install -y --no-install-recommends clang
echo "version=$(rustc -V | cut -d' ' -f2)" >> "$GITHUB_OUTPUT" - name: Build the server
- run: sudo apt-get update -qq && sudo apt-get install -y -qq --no-install-recommends clang >/dev/null run: cargo build -p inbuxa --locked
# Cargo's download cache and the dev/test target dir, keyed on the - name: Compile every test target
# lockfile and the compiler. Saved from main only, so the one cache # `--no-run` is the point: it builds the unit tests and the integration
# every branch restores is main's, and branches cannot evict it. # crate together, which is the combination that resolves the test
- uses: actions/cache/restore@55cc8345863c7cc4c66a329aec7e433d2d1c52a9 # v6.1.0 # features, and stops short of running anything that wants a store.
with: run: cargo test --workspace --locked --no-run
path: |
~/.cargo/registry/index
~/.cargo/registry/cache
~/.cargo/git/db
target/debug
key: cargo-${{ steps.rust.outputs.version }}-${{ hashFiles('Cargo.lock') }}
restore-keys: cargo-${{ steps.rust.outputs.version }}-
- run: cargo build -p inbuxa --locked
# --no-run: compiles every test target without running them, which
# catches a test that no longer builds without needing a store.
- run: cargo test --workspace --locked --no-run
- if: github.ref == 'refs/heads/main'
uses: actions/cache/save@55cc8345863c7cc4c66a329aec7e433d2d1c52a9 # v6.1.0
with:
path: |
~/.cargo/registry/index
~/.cargo/registry/cache
~/.cargo/git/db
target/debug
key: cargo-${{ steps.rust.outputs.version }}-${{ hashFiles('Cargo.lock') }}
# The release profile, on main only. It is the profile the image is
# built with, and it fails in ways the dev profile does not: v2026.9.24
# was tagged on a commit whose CI was green and whose release build
# could not compile the scim crate at all.
- if: github.ref == 'refs/heads/main'
run: cargo build -p inbuxa --locked --release
# --------------------------------------------------------------- tags ------
# Two guards before anything is pushed, the same as Gitea's publish.yml:
# * the tag must be v<brand_version!>. The version is a string in
# crates/types/src/branding.rs, not Cargo.toml, and the image is tagged
# with it, so a tag beside an unbumped macro would publish an image that
# reports a different version from its tag.
# * the tag must be on main or on a release/* branch, so an image never
# describes code that was never reviewed onto one of them. A release/*
# branch carries a hotfix cut from an earlier release tag.
version:
if: ${{ vars.BUILD_ON == 'github' && github.ref_type == 'tag' && startsWith(github.ref_name, 'v') }}
runs-on: ubuntu-latest
outputs:
version: ${{ steps.v.outputs.version }}
steps:
# Full history, and every branch as origin/*: the ancestry check cannot
# be answered from a shallow clone.
- uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1
with:
fetch-depth: 0
- id: v
env:
TAG: ${{ github.ref_name }}
run: |
set -euo pipefail
# Scoped to the macro body: branding.rs holds other string literals,
# and tagging an image from one of those would be worse than failing.
V="$(awk '/macro_rules! brand_version /,/^}/' crates/types/src/branding.rs \
| grep -om1 '"[0-9][^"]*"' | tr -d '"')"
[ -n "$V" ] || { echo "could not read brand_version! from branding.rs" >&2; exit 1; }
if [ "$TAG" != "v$V" ]; then
echo "Tag $TAG names a commit whose brand_version! says $V." >&2
echo "Refusing to publish an image that would report the wrong version." >&2
exit 1
fi
commit="$(git rev-parse "${TAG}^{commit}")"
on=""
for ref in origin/main $(git for-each-ref --format='%(refname:short)' 'refs/remotes/origin/release/*'); do
if git merge-base --is-ancestor "$commit" "$ref"; then on="$ref"; break; fi
done
[ -n "$on" ] || { echo "$TAG is not on main or a release/* branch" >&2; exit 1; }
echo "$TAG is on $on"
echo "version=$V" >> "$GITHUB_OUTPUT"
# Each architecture on its own native runner, side by side. The Dockerfile
# cross-compiles from the build platform, and on the self-hosted runners one
# machine built both one after the other; here two machines build at once,
# each natively (the builder stage picks the matching target, and the
# aarch64 toolchain it installs exists on arm64 too), and the small final
# stage needs no QEMU. amd64 also moves :<version> as soon as it is done, so
# a production deploy can start from it; :latest waits for the index below,
# so it never names an image without arm64.
publish:
needs: [version]
runs-on: ${{ matrix.runner }}
strategy:
fail-fast: false
matrix:
include:
- arch: amd64
runner: ubuntu-latest
- arch: arm64
runner: ubuntu-24.04-arm
env:
VERSION: ${{ needs.version.outputs.version }}
steps:
- run: |
sudo rm -rf /usr/share/dotnet /usr/local/lib/android /opt/ghc /opt/hostedtoolcache/CodeQL
echo "IMAGE=${{ vars.REGISTRY }}/${GITHUB_REPOSITORY,,}" >> "$GITHUB_ENV"
# The release link (fat LTO, one codegen unit) outgrows the runner's
# 16 GB: v2026.9.30's arm64 link was killed for memory. Swap gives it
# room; buildx's container has no memory limit of its own, so it
# reaches the host's swap.
- run: |
sudo fallocate -l 16G /swap.release
sudo chmod 600 /swap.release
sudo mkswap /swap.release >/dev/null
sudo swapon /swap.release
free -g
df -h /
- uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1
- uses: docker/setup-buildx-action@594f3bf4285d9ea8dc53c9a0c9c4092420091003 # v4.4.0
- uses: docker/login-action@dbcb813823bdd20940b903addbd779551569679f # v4.6.0
with:
registry: ${{ vars.REGISTRY }}
username: jcoffey-dev
password: ${{ secrets.GITEA_TOKEN }}
# Attestations off: they add manifests of their own, and the index
# should hold the two images and nothing else. No build cache: GitHub
# scopes a tag run's cache to that tag, so the next release could never
# read it, and each one would park several GB in the repository's 10 GB
# cache and evict main's cargo cache.
- uses: docker/build-push-action@c3c9e263c25d99ce0380d002d59b67737d91b0dc # v7.4.0
with:
context: .
platforms: linux/${{ matrix.arch }}
provenance: false
sbom: false
push: true
tags: |
${{ env.IMAGE }}:${{ env.VERSION }}-${{ matrix.arch }}
${{ matrix.arch == 'amd64' && format('{0}:{1}', env.IMAGE, env.VERSION) || '' }}
# Joins the two per-architecture tags into :<version> and :latest. Built
# from the per-architecture tags rather than :<version>, which by now is
# the amd64 image and would be read as such.
index:
needs: [version, publish]
runs-on: ubuntu-latest
env:
VERSION: ${{ needs.version.outputs.version }}
steps:
- run: echo "IMAGE=${{ vars.REGISTRY }}/${GITHUB_REPOSITORY,,}" >> "$GITHUB_ENV"
- uses: docker/setup-buildx-action@594f3bf4285d9ea8dc53c9a0c9c4092420091003 # v4.4.0
- uses: docker/login-action@dbcb813823bdd20940b903addbd779551569679f # v4.6.0
with:
registry: ${{ vars.REGISTRY }}
username: jcoffey-dev
password: ${{ secrets.GITEA_TOKEN }}
- run: |
docker buildx imagetools create \
--tag "$IMAGE:$VERSION" \
--tag "$IMAGE:latest" \
"$IMAGE:$VERSION-amd64" "$IMAGE:$VERSION-arm64"
docker buildx imagetools inspect "$IMAGE:$VERSION"
# Gitea keeps a container package on its owner; linking it shows it on
# the repository's Packages tab. Idempotent.
- env:
GITEA_TOKEN: ${{ secrets.GITEA_TOKEN }}
run: |
owner="${GITHUB_REPOSITORY%%/*}"; name="${GITHUB_REPOSITORY#*/}"
curl -fsS -o /dev/null -X POST -H "Authorization: token $GITEA_TOKEN" \
"$GITEA_URL/api/v1/packages/${owner,,}/container/$name/-/link/$name" \
|| echo "package already linked (or link refused); not fatal"
# The weekly release creates its Release (and so the tag) on Gitea first; a
# tag pushed by hand has none. Either way the tag ends up with exactly one
# Release there, created once the image exists so its pull instructions
# work.
release:
needs: [version, index]
runs-on: ubuntu-latest
steps:
- env:
TAG: ${{ github.ref_name }}
VERSION: ${{ needs.version.outputs.version }}
GITEA_TOKEN: ${{ secrets.GITEA_TOKEN }}
REGISTRY: ${{ vars.REGISTRY }}
run: |
set -euo pipefail
api="$GITEA_URL/api/v1/repos/$GITHUB_REPOSITORY"
code="$(curl -sS -o /dev/null -w '%{http_code}' -H "Authorization: token $GITEA_TOKEN" "$api/releases/tags/$TAG")"
if [ "$code" = 200 ]; then echo "$TAG already has a release"; exit 0; fi
[ "$code" = 404 ] || { echo "looking up the release for $TAG answered $code" >&2; exit 1; }
image="$REGISTRY/${GITHUB_REPOSITORY,,}:$VERSION"
body="Container image: \`$image\` (linux/amd64, linux/arm64); also \`:latest\`.
Binaries for a host install are attached: \`inbuxa-linux-amd64.tar.gz\` and \`inbuxa-linux-arm64.tar.gz\`, with \`SHA256SUMS\`. Each is the binary out of this release's image for that architecture, so it is the same build. The image grants it \`cap_net_bind_service\`; a host install has to grant that itself (\`setcap\`, or \`AmbientCapabilities\` in the unit) to bind port 25."
jq -n --arg tag "$TAG" --arg name "INBUXA $VERSION" --arg body "$body" \
'{tag_name:$tag, name:$name, body:$body}' |
curl -fsS -X POST -H "Authorization: token $GITEA_TOKEN" -H 'Content-Type: application/json' \
--data @- "$api/releases" | jq -r '"created release " + .tag_name'
# The binaries for a host install, taken out of the image that was just
# pushed rather than compiled again: the binary in the tarball is the file
# the image runs. `docker create` starts nothing, so copying a file out of
# the arm64 image on an amd64 runner needs no emulation.
binaries:
needs: [version, index, release]
runs-on: ubuntu-latest
env:
VERSION: ${{ needs.version.outputs.version }}
TAG: ${{ github.ref_name }}
GITEA_TOKEN: ${{ secrets.GITEA_TOKEN }}
steps:
- run: echo "IMAGE=${{ vars.REGISTRY }}/${GITHUB_REPOSITORY,,}" >> "$GITHUB_ENV"
- uses: docker/login-action@dbcb813823bdd20940b903addbd779551569679f # v4.6.0
with:
registry: ${{ vars.REGISTRY }}
username: jcoffey-dev
password: ${{ secrets.GITEA_TOKEN }}
- name: take the binaries out of the image
run: |
set -euo pipefail
mkdir -p out && cd out
for arch in amd64 arm64; do
docker pull -q --platform "linux/$arch" "$IMAGE:$VERSION"
id="$(docker create --platform "linux/$arch" "$IMAGE:$VERSION")"
docker cp "$id:/usr/local/bin/inbuxa" inbuxa
docker rm -f "$id" >/dev/null
chmod 0755 inbuxa
tar -czf "inbuxa-linux-$arch.tar.gz" inbuxa
rm inbuxa
done
sha256sum inbuxa-linux-*.tar.gz > SHA256SUMS
cat SHA256SUMS
# A re-run of a tag replaces its assets rather than leaving two files
# with the same name and different contents.
#
# The uploads cross Cloudflare, which dropped 50 MB HTTP/2 uploads
# part-way for v2026.9.30.1 (curl 92, PROTOCOL_ERROR; origin logged
# 400), once on each of two runs. Uploads go over HTTP/1.1 and retry.
- name: attach them to the release
run: |
set -euo pipefail
api="$GITEA_URL/api/v1/repos/$GITHUB_REPOSITORY"
auth="Authorization: token $GITEA_TOKEN"
retry=(--retry 5 --retry-all-errors --retry-delay 15)
rel="$(curl -fsS "${retry[@]}" -H "$auth" "$api/releases/tags/$TAG" | jq -r .id)"
assets="$(curl -fsS "${retry[@]}" -H "$auth" "$api/releases/$rel/assets")"
for f in out/inbuxa-linux-amd64.tar.gz out/inbuxa-linux-arm64.tar.gz out/SHA256SUMS; do
name="$(basename "$f")"
old="$(jq -r --arg n "$name" '.[] | select(.name == $n) | .id' <<<"$assets")"
for id in $old; do curl -fsS "${retry[@]}" -o /dev/null -X DELETE -H "$auth" "$api/releases/$rel/assets/$id"; done
curl -fsS --http1.1 "${retry[@]}" -o /dev/null -X POST -H "$auth" -F "attachment=@$f" "$api/releases/$rel/assets?name=$name"
echo "attached $name"
done
# ------------------------------------------------------ ghcr replica ------
# Copies the release image from the Gitea registry, which stays the
# authoritative one, to ghcr.io under the same version tag and :latest. It is
# a copy, not a second build: the digest on GHCR is the digest on the
# registry, so `docker pull ghcr.io/...` gets exactly the same image. Left
# out of the report to Gitea, like the release copy, so a GHCR problem
# cannot fail a release.
ghcr:
if: ${{ vars.BUILD_ON == 'github' && github.ref_type == 'tag' }}
needs: [version, index]
runs-on: ubuntu-latest
permissions:
contents: read
packages: write
steps:
- env:
GH_TOKEN: ${{ github.token }}
TAG: ${{ needs.version.outputs.version }}
run: |
set -euo pipefail
src="${{ vars.REGISTRY }}/${GITHUB_REPOSITORY,,}"
dst="ghcr.io/${GITHUB_REPOSITORY,,}"
tag="$TAG"
echo "$GH_TOKEN" | docker login ghcr.io -u "$GITHUB_ACTOR" --password-stdin
docker buildx imagetools create -t "$dst:$tag" -t "$dst:latest" "$src:$tag"
want="$(docker buildx imagetools inspect "$src:$tag" --format '{{json .Manifest.Digest}}')"
got="$(docker buildx imagetools inspect "$dst:$tag" --format '{{json .Manifest.Digest}}')"
echo "registry $src:$tag = $want"
echo "ghcr $dst:$tag = $got"
[ "$want" = "$got" ] || echo "::warning::GHCR digest differs from the registry's"
docker logout ghcr.io
# ---------------------------------------------------- github release ------
# Copies this tag's Gitea release -- notes and files -- to a GitHub release,
# so the replica's Releases page, and anyone watching it, keeps up. Gitea's
# release is the real one; this is left out of the report to Gitea, so a
# failure here cannot fail a release. PR and issue numbers in the notes are
# rewritten to Gitea links: on GitHub a bare #16 is some other PR.
github-release:
if: ${{ vars.BUILD_ON == 'github' && github.ref_type == 'tag' }}
needs: [binaries]
runs-on: ubuntu-latest
permissions:
contents: write
env:
GITEA_URL: ${{ vars.GITEA_URL }}
GH_TOKEN: ${{ github.token }}
TAG: ${{ github.ref_name }}
steps:
- run: |
set -euo pipefail
if gh release view "$TAG" --repo "$GITHUB_REPOSITORY" >/dev/null 2>&1; then
echo "GitHub already has a release for $TAG"; exit 0
fi
# The Gitea release exists by now if this run made it; if the weekly
# release job made it, it came before the tag. Allow a few minutes.
code=0
for _ in $(seq 1 15); do
code="$(curl -sS -o rel.json -w '%{http_code}' "$GITEA_URL/api/v1/repos/$GITHUB_REPOSITORY/releases/tags/$TAG")"
[ "$code" = 200 ] && break
sleep 20
done
if [ "$code" != 200 ]; then echo "No Gitea release for $TAG; nothing to copy"; exit 0; fi
if [ "$(jq -r .draft rel.json)" = true ]; then echo "The Gitea release is a draft; not copying"; exit 0; fi
export BASE="$(jq -r '.html_url | sub("/releases/tag/.*$"; "")' rel.json)"
jq -r '.body // ""' rel.json | perl -pe 's{(?<![\w/&\[])#(\d+)\b}{[#$1]($ENV{BASE}/pulls/$1)}g' > notes.md
printf '\n\n_Mirrored from [the Gitea release](%s); report issues on [Gitea](%s/issues)._\n' \
"$(jq -r .html_url rel.json)" "$BASE" >> notes.md
files=()
mkdir -p files
while IFS=$'\t' read -r name url; do
curl -fsSL -o "files/$name" "$url"; files+=("files/$name")
done < <(jq -r '.assets[]? | [.name, .browser_download_url] | @tsv' rel.json)
title="$(jq -r '.name // ""' rel.json)"; [ -n "$title" ] || title="$TAG"
if [ "$(jq -r .prerelease rel.json)" = true ]; then kind=--prerelease; else kind=--latest; fi
gh release create "$TAG" --repo "$GITHUB_REPOSITORY" --verify-tag --title "$title" \
--notes-file notes.md "$kind" "${files[@]}"
echo "created the GitHub release for $TAG with ${#files[@]} file(s)"
# ------------------------------------------------------------- report ------
# One commit status on Gitea for the whole run: what Gitea's ci.yml and
# publish.yml wait on. Skipped jobs (the tag jobs on a branch, and the other
# way round) count as passing; a failed or cancelled one does not.
report:
if: ${{ always() && vars.BUILD_ON == 'github' }}
needs: [start, fork-checks, build, version, publish, index, release, binaries]
runs-on: ubuntu-latest
steps:
- env:
GITEA_TOKEN: ${{ secrets.GITEA_TOKEN }}
STATE: ${{ contains(needs.*.result, 'failure') && 'failure' || (contains(needs.*.result, 'cancelled') && 'cancelled' || 'success') }}
run: |
# A cancelled run was superseded by a newer run for the same commit (the
# mirror can push one commit twice); that run reports. Posting "failure"
# here would fail the Gitea check while the real build is still going.
if [ "$STATE" = cancelled ]; then echo "cancelled: leaving the result to the newer run"; exit 0; fi
jq -n --arg s "$STATE" --arg c "$STATUS_CONTEXT" \
--arg u "$GITHUB_SERVER_URL/$GITHUB_REPOSITORY/actions/runs/$GITHUB_RUN_ID" \
'{state:$s, context:$c, target_url:$u, description:"GitHub Actions"}' |
curl -fsS -o /dev/null -X POST -H "Authorization: token $GITEA_TOKEN" \
-H 'Content-Type: application/json' --data @- \
"$GITEA_URL/api/v1/repos/$GITHUB_REPOSITORY/statuses/$GITHUB_SHA"
echo "$STATUS_CONTEXT: $STATE"
+69
View File
@@ -0,0 +1,69 @@
# Prune old image versions from GHCR.
#
# Releases are kept forever -- they carry no assets and their generated notes
# are this project's only changelog, so deleting one destroys history that
# cannot be reconstructed for nothing saved. Images are the opposite: a
# multi-arch build a week, and the by-digest push in publish.yml leaves two
# untagged per-architecture manifests behind each time on top of the tagged
# index. Those accumulate and nobody wants fifty of them.
#
# THE FOOTGUN: the obvious tool for this -- delete-package-versions with
# `delete-only-untagged-versions` -- will happily delete the per-architecture
# manifests that a multi-arch tag points *at*, because they are untagged by
# design. Nothing appears to break: the tag still exists, and pulls simply
# start failing for one architecture. This action understands manifest lists
# and will not orphan a retained index, and `validate` re-checks every
# multi-arch manifest against the registry afterwards.
#
# Separate from publish.yml, and dispatchable on its own, so `dry_run` can show
# exactly what would be deleted without rebuilding and re-pushing an image to
# find out.
name: Prune images
on:
workflow_call:
inputs:
dry_run:
type: boolean
default: false
workflow_dispatch:
inputs:
dry_run:
description: "List what would be deleted, delete nothing"
type: boolean
default: true
jobs:
prune:
runs-on: ubuntu-latest
permissions:
packages: write
steps:
# The only third-party action here that is not published by GitHub or
# Docker, and the one with the most to lose: it is handed
# `packages: write` and its whole job is deletion, so a ref repointed at
# something else -- by a compromise or a mistake upstream -- is a bad
# day. It was pinned to a commit long before the rest of them were.
- uses: dataaxiom/ghcr-cleanup-action@d52806a0dc70b430571a37da1fde39733ffd640f # v1.2.2
with:
owner: inbuxa
package: inbuxa-server
token: ${{ secrets.GITHUB_TOKEN }}
# Ten weekly releases is roughly a quarter of history, which is more
# than enough to roll back to and far less than the year's worth that
# would otherwise pile up. Older *releases* stay either way; this
# only removes the images.
keep-n-tagged: 10
# Belt and braces on top of the action's own manifest awareness:
# `latest` is never a candidate for deletion under any counting.
exclude-tags: latest
delete-untagged: true
# Sweeps the wreckage of a half-failed run: an index whose platform
# images did not all land, and referrers whose parent is gone.
delete-partial-images: true
delete-orphaned-images: true
# Checks every remaining multi-architecture manifest still resolves
# in the registry. This is the step that would catch the footgun
# above rather than leaving a reader to discover it on `docker pull`.
validate: true
dry-run: ${{ inputs.dry_run }}
+198
View File
@@ -0,0 +1,198 @@
# Publish the container image to GHCR.
#
# The README and the docs site have told people to run
# `ghcr.io/inbuxa/inbuxa-server:latest` for a long time, and nothing ever
# pushed it: `docker pull` answered `denied`, because the package did not
# exist. This is the workflow that makes those instructions true. It is also
# the prerequisite for the self-hosted app catalogs -- TrueNAS and Unraid
# both install by pulling an image and neither builds from source.
#
# FIRST RUN: a package GHCR creates for the first time is **private**, even in
# a public repository, and an anonymous `docker pull` will still answer
# `denied`. Nothing in a workflow can change that -- the visibility is set once
# by hand under the package's settings, and until it is, this looks like it
# worked while the docs stay just as wrong as before. Check with a logged-out
# pull, not with one from a machine that has credentials.
#
# Two architectures, each built on its own native runner rather than under
# QEMU. Emulated arm64 has to run `npm ci` and the Vite build through
# instruction translation, which takes tens of minutes and occasionally runs
# out of memory; `ubuntu-24.04-arm` is free for public repositories and does
# the same work at native speed. The cost is the by-digest dance below: each
# runner pushes an untagged image, and a final job joins the two digests into
# one multi-arch tag.
name: Publish image
on:
release:
types: [published]
# Callable, so release.yml can build the release it just cut. This is not a
# stylistic choice: a release created with GITHUB_TOKEN does **not** raise a
# `release` event -- GitHub refuses to let a token trigger another workflow,
# to stop a workflow looping on its own output. A scheduled job that cut a
# release and expected this file to notice would silently never publish. The
# alternatives are a personal access token kept as a secret, or calling the
# workflow directly. This is the one that needs no credential.
workflow_call:
inputs:
ref:
description: "Tag, branch or SHA to build"
required: true
type: string
tag_latest:
description: "Also move :latest to this build"
type: boolean
default: false
# Same reasoning as ci.yml's dispatch trigger: a run GitHub queues and then
# orphans can be neither rerun nor canceled, and this workflow otherwise
# only fires on a release -- which is not something to cut twice because a
# runner died. `ref` also allows publishing an image for a tag that predates
# this workflow, which is how the first one gets built.
workflow_dispatch:
inputs:
ref:
description: "Tag, branch or SHA to build"
required: true
default: main
tag_latest:
description: "Also move :latest to this build"
type: boolean
default: false
env:
# Hardcoded rather than derived from github.repository, which would have to
# be lowercased to be a legal registry path. This is the string the docs name.
IMAGE: ghcr.io/inbuxa/inbuxa-server
jobs:
# The version is read once and handed to both builds, so the two
# architectures cannot disagree about what they are. It is read from the
# macro the binary itself compiles in, which the weekly release commits
# before this runs -- so the image is tagged with the version it reports.
version:
runs-on: ubuntu-latest
outputs:
version: ${{ steps.v.outputs.version }}
steps:
- uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1
with:
ref: ${{ inputs.ref || github.ref }}
- id: v
run: |
set -euo pipefail
# Scoped to the macro body: branding.rs holds other string literals,
# and tagging an image from one of those would be worse than failing.
V="$(awk '/macro_rules! brand_version/,/^}/' crates/types/src/branding.rs \
| grep -om1 '"[0-9][^"]*"' | tr -d '"')"
[ -n "$V" ] || { echo "could not read brand_version! from branding.rs" >&2; exit 1; }
# A date version carries nothing a Docker tag objects to, so there is
# no second, sanitized form of it here.
echo "version=$V" >> "$GITHUB_OUTPUT"
echo "version $V"
build:
needs: version
runs-on: ${{ matrix.runner }}
permissions:
contents: read
packages: write
strategy:
fail-fast: false
matrix:
include:
- platform: linux/amd64
runner: ubuntu-latest
- platform: linux/arm64
runner: ubuntu-24.04-arm
steps:
- uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1
with:
ref: ${{ inputs.ref || github.ref }}
- uses: docker/setup-buildx-action@594f3bf4285d9ea8dc53c9a0c9c4092420091003 # v4.4.0
- uses: docker/login-action@dbcb813823bdd20940b903addbd779551569679f # v4.6.0
with:
registry: ghcr.io
username: ${{ github.actor }}
password: ${{ secrets.GITHUB_TOKEN }}
- name: Build and push by digest
id: push
uses: docker/build-push-action@c3c9e263c25d99ce0380d002d59b67737d91b0dc # v7.4.0
with:
context: .
platforms: ${{ matrix.platform }}
# Attestations are off deliberately: they add manifests of their own
# to the index, and `imagetools create` below expects the two entries
# it pushed rather than four.
provenance: false
sbom: false
cache-from: type=gha,scope=${{ matrix.platform }}
cache-to: type=gha,mode=max,scope=${{ matrix.platform }}
outputs: type=image,name=${{ env.IMAGE }},push-by-digest=true,name-canonical=true,push=true
- name: Save the digest
run: |
mkdir -p /tmp/digests
# The prefix is stripped here and put back in the merge job, so the
# filename is the bare hash. Leaving it on produces
# `image@sha256:sha256:...` when the reference is rebuilt.
digest="${{ steps.push.outputs.digest }}"
touch "/tmp/digests/${digest#sha256:}"
- uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7.0.1
with:
# One artifact per platform; the merge job globs them back together.
name: digest-${{ strategy.job-index }}
path: /tmp/digests/*
retention-days: 1
if-no-files-found: error
# Joins the per-architecture digests into a single tagged manifest, so
# `docker pull ghcr.io/inbuxa/inbuxa-server:<tag>` resolves on both.
publish:
needs: [version, build]
runs-on: ubuntu-latest
permissions:
contents: read
packages: write
steps:
- uses: actions/download-artifact@3e5f45b2cfb9172054b4087a40e8e0b5a5461e7c # v8.0.1
with:
path: /tmp/digests
pattern: digest-*
merge-multiple: true
- uses: docker/setup-buildx-action@594f3bf4285d9ea8dc53c9a0c9c4092420091003 # v4.4.0
- uses: docker/login-action@dbcb813823bdd20940b903addbd779551569679f # v4.6.0
with:
registry: ghcr.io
username: ${{ github.actor }}
password: ${{ secrets.GITHUB_TOKEN }}
- name: Create the manifest
run: |
# Arrays rather than a string: the tags and the digest references
# have to reach docker as separate arguments, and building them by
# word-splitting an unquoted variable is the version of this that
# breaks the day a value contains a space.
tags=(-t "${IMAGE}:${{ needs.version.outputs.version }}")
# :latest follows real releases only. A prerelease that moved it
# would hand every `:latest` deployment an unfinished build, and a
# dispatch run has to ask for it on purpose.
if [ "${{ github.event_name }}" = "release" ] && [ "${{ github.event.release.prerelease }}" = "false" ]; then
tags+=(-t "${IMAGE}:latest")
elif [ "${{ inputs.tag_latest }}" = "true" ]; then
tags+=(-t "${IMAGE}:latest")
fi
refs=()
for f in /tmp/digests/*; do
refs+=("${IMAGE}@sha256:$(basename "$f")")
done
echo "tags: ${tags[*]}"
echo "refs: ${refs[*]}"
docker buildx imagetools create "${tags[@]}" "${refs[@]}"
- name: Show what landed
run: docker buildx imagetools inspect "${IMAGE}:${{ needs.version.outputs.version }}"
# Runs only after a successful publish, because that is the only moment the
# package grows. See cleanup.yml for why this is not the obvious one-liner.
prune:
needs: publish
permissions:
packages: write
uses: ./.github/workflows/cleanup.yml
+246
View File
@@ -0,0 +1,246 @@
# Cut a release once a week, but only if there is something in it.
#
# It does nothing on a quiet week. A release with no commits in it is worse
# than no release: it moves `:latest` to an identical build, spends a version
# number, and mails everybody watching the repository about nothing.
#
# INBUXA's version is a string in crates/types/src/branding.rs, deliberately
# not in Cargo.toml so that upstream's version bumps merge without conflicts.
# So this writes it: the bump is committed to main, and the tag names that
# commit. The tree a tag points at therefore reports the version the tag
# claims, which a tag placed beside an unbumped macro cannot promise.
name: Weekly release
on:
schedule:
# Mondays, 10:07 UTC, and last of the three: INBUXA Admin and the webmail
# release ahead of the server they talk to. Staggered rather than
# simultaneous so three releases do not compete for runners, and so a bad
# Monday names one repository instead of three. GitHub runs scheduled jobs
# best-effort and can delay a run considerably, so the exact minute is not
# a promise; the odd minute keeps it off the crowded top of the hour.
#
# Note also that GitHub disables scheduled workflows in a repository with
# no activity for 60 days, which is worth checking for before assuming
# this file is broken.
- cron: "7 10 * * 1"
workflow_dispatch:
inputs:
dry_run:
description: "Work out what would be released, then stop"
type: boolean
default: false
# One at a time. Two overlapping runs would race to write the same version and
# create the same tag, and the loser fails noisily for a reason that has
# nothing to do with the code.
concurrency:
group: weekly-release
cancel-in-progress: false
jobs:
check:
runs-on: ubuntu-latest
permissions:
contents: read
outputs:
should_release: ${{ steps.decide.outputs.should_release }}
version: ${{ steps.decide.outputs.version }}
tag: ${{ steps.decide.outputs.tag }}
previous: ${{ steps.decide.outputs.previous }}
count: ${{ steps.decide.outputs.count }}
steps:
- uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1
with:
ref: main
fetch-depth: 0
- id: decide
env:
GH_TOKEN: ${{ secrets.GITHUB_TOKEN }}
run: |
set -euo pipefail
# The newest published release, or empty on a repository that has
# never had one -- in which case everything counts as new. Drafts are
# excluded: an unpublished draft is not a release anybody has, so
# counting from it would hide commits that have never shipped.
previous="$(gh release list --limit 1 --exclude-drafts --json tagName --jq '.[0].tagName // ""')"
# A tag named by a release is normally present after a full checkout,
# but a release can outlive its tag. Falling back to the whole
# history is the safe direction to be wrong in: it over-counts, which
# cuts a release that was due anyway, where under-counting would skip
# one that was.
if [ -n "$previous" ] && git rev-parse -q --verify "refs/tags/${previous}" >/dev/null; then
count="$(git rev-list --count "${previous}..HEAD")"
else
count="$(git rev-list --count HEAD)"
fi
# INBUXA's version is the date: YYYY.M.D, unpadded, as branding.rs
# documents. A second release on one day takes a `.N` suffix,
# counting from 2, which is why this asks the tags rather than
# assuming today is free.
today="$(date -u +%Y.%-m.%-d)"
version="$today"
n=2
while git rev-parse -q --verify "refs/tags/v${version}" >/dev/null; do
version="${today}.${n}"
n=$((n + 1))
done
should_release=true
reason=""
if [ "$count" -eq 0 ]; then
should_release=false
reason="no commits since ${previous}"
fi
{
echo "should_release=$should_release"
echo "version=$version"
echo "tag=v${version}"
echo "previous=$previous"
echo "count=$count"
} >> "$GITHUB_OUTPUT"
# Written to the run summary so a skipped week reads as a decision
# rather than as a workflow that quietly did nothing.
{
echo "### Weekly release"
echo
if [ "$should_release" = "true" ]; then
echo "Releasing **v${version}** — ${count} commit(s) since ${previous:-the beginning}."
else
echo "Nothing to release: ${reason}."
fi
} >> "$GITHUB_STEP_SUMMARY"
cut:
needs: check
if: needs.check.outputs.should_release == 'true' && !inputs.dry_run
runs-on: ubuntu-latest
permissions:
contents: write
pull-requests: write
outputs:
sha: ${{ steps.land.outputs.sha }}
steps:
- uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1
with:
ref: main
fetch-depth: 0
- id: bump
env:
VERSION: ${{ needs.check.outputs.version }}
BRANCH: release/v${{ needs.check.outputs.version }}
run: |
set -euo pipefail
# Scoped to the macro body rather than replacing the first quoted
# string in the file, and asserted to have matched exactly once.
# branding.rs holds other string literals, and a bump that silently
# edited one of those -- or none -- would ship a build whose version
# disagrees with its tag.
python3 - <<'PY'
import os, re
path = "crates/types/src/branding.rs"
src = open(path, encoding="utf-8").read()
pattern = re.compile(r'(macro_rules! brand_version \{\s*\(\) => \{\s*")[^"]+(")')
out, n = pattern.subn(lambda m: m.group(1) + os.environ["VERSION"] + m.group(2), src, count=1)
assert n == 1, f"brand_version! not found in {path}"
open(path, "w", encoding="utf-8").write(out)
PY
git config user.name "github-actions[bot]"
git config user.email "41898282+github-actions[bot]@users.noreply.github.com"
git add crates/types/src/branding.rs
git commit -m "Version ${VERSION}"
git push origin "HEAD:refs/heads/${BRANCH}"
# main is protected: it takes a pull request with a green build, and
# GITHUB_TOKEN is not among the bypass actors. So the bump lands the way
# every other change does. The alternative was to hand the release a
# credential that outranks the rule, which is a worse thing to own than
# a slower Monday.
- id: land
env:
VERSION: ${{ needs.check.outputs.version }}
BRANCH: release/v${{ needs.check.outputs.version }}
GH_TOKEN: ${{ secrets.GITHUB_TOKEN }}
run: |
set -euo pipefail
url="$(gh pr create --base main --head "${BRANCH}" \
--title "Version ${VERSION}" \
--body "Weekly release. Bumps \`brand_version!\` to ${VERSION} so the tag names a tree that reports the version the tag claims.")"
# The number, not the branch: the branch is deleted on merge, and a
# deleted branch no longer resolves to its pull request.
pr="${url##*/}"
echo "Opened #${pr}"
# The build is what the rule actually requires, and it is also the
# thing worth waiting for: a release cut from a tree that does not
# compile is the failure this whole arrangement exists to prevent.
# A full build of this tree is long, so the deadline is generous.
deadline=$(( SECONDS + 3600 ))
while :; do
state="$(gh pr view "${pr}" --json statusCheckRollup \
--jq '[.statusCheckRollup[]? | .conclusion // "PENDING"] | join(",")')"
case "${state}" in
*FAILURE*|*CANCELLED*|*TIMED_OUT*)
echo "::error::CI failed on ${BRANCH} (${state}); no release cut. PR #${pr} is left open."
exit 1 ;;
*SUCCESS*) break ;;
esac
if [ "${SECONDS}" -ge "${deadline}" ]; then
echo "::error::timed out waiting for CI on ${BRANCH}. PR #${pr} is left open."
exit 1
fi
sleep 30
done
gh pr merge "${pr}" --rebase --delete-branch
# A rebase merge rewrites the commit, so the sha to tag is the one
# GitHub recorded for the merge, not the tip that was pushed. It can
# take a moment to appear.
sha=""
for _ in $(seq 1 30); do
sha="$(gh pr view "${pr}" --json mergeCommit --jq '.mergeCommit.oid // ""')"
[ -n "${sha}" ] && break
sleep 5
done
if [ -z "${sha}" ]; then
echo "::error::#${pr} merged but GitHub reported no merge commit; nothing safe to tag."
exit 1
fi
echo "sha=${sha}" >> "$GITHUB_OUTPUT"
- env:
GH_TOKEN: ${{ secrets.GITHUB_TOKEN }}
run: |
set -euo pipefail
args=(--target "${{ steps.land.outputs.sha }}"
--title "INBUXA ${{ needs.check.outputs.version }}"
--generate-notes)
# Bound the notes to what is actually new. Without a start tag the
# generator reaches back to whatever it decides is previous, which on
# a repository carrying upstream's tag shapes is not always the last
# release.
if [ -n "${{ needs.check.outputs.previous }}" ]; then
args+=(--notes-start-tag "${{ needs.check.outputs.previous }}")
fi
gh release create "${{ needs.check.outputs.tag }}" "${args[@]}"
# Called rather than left to the `release` trigger on purpose: see the note
# at the top of publish.yml. A release created with GITHUB_TOKEN raises no
# event, so without this the tag would exist and no image would follow it.
publish:
needs: [check, cut]
permissions:
contents: read
packages: write
uses: ./.github/workflows/publish.yml
with:
ref: ${{ needs.cut.outputs.sha }}
tag_latest: true
-27
View File
@@ -2,33 +2,6 @@
All notable changes to this project will be documented in this file. This project adheres to [Semantic Versioning](http://semver.org/). All notable changes to this project will be documented in this file. This project adheres to [Semantic Versioning](http://semver.org/).
## [0.16.25] - 2026-10-05
If you are upgrading from v0.16.x, replace the binary (or run `docker pull`). If you are upgrading from v0.15.x and below, please read the [upgrading documentation](https://github.com/stalwartlabs/stalwart/blob/main/UPGRADING/v0_16.md) for more information on how to upgrade from previous versions.
## Added
## Changed
## Fixed
- JMAP: Creating a `MaskedEmail` with `emailDomain` fails with `forbidden` for every domain when the account has addresses on more than one domain.
- Autodiscover: Requests for a response schema other than Outlook's, such as ActiveSync (`mobilesync`), are answered with the Outlook settings instead of error 601.
- IMAP:
- `LOGIN` and `AUTHENTICATE` with a wrong, expired or unknown app password or API key are answered with an untagged `NO`, so clients keep waiting for the command to complete until the connection times out.
- The failed login that exceeds the maximum number of authentication failures is answered with an untagged `NO` before the connection is closed.
- DKIM:
- A rotation moves the active key to retiring even when its successor fails to publish or propagate, so outgoing mail is sent unsigned until a retry publishes the new key. The DNS write failure is also not logged and the task reports success.
- Keys created while DNS management was manual, or before DKIM was added to the published records, are never rotated after DNS management becomes automatic. Domains already affected start rotating once a `DkimManagement` task is created for them.
- After switching DNS management from automatic to manual, a due rotation activates a new key that was never published in DNS, so signatures fail verification, and retiring the old key is retried forever.
- Spam filter:
- Messages with no text line long enough for a Pyzor digest are checked with the digest of empty input and tagged `PYZOR`.
- DNSBL answers with several return codes, such as a Spamhaus ZEN listing in both SBL and PBL, are scored for only the first code returned.
- DNSBL lookups that return "not listed" are cached for 24 hours regardless of the zone's negative TTL.
- Removing a duplicate training sample of a message reclassified on the same day clears the blob link of the sample that is kept.
- MTA: Queue quotas with an empty `match` expression are never enforced, including the global queue quota created on first start.
- RocksDB: The info log (`LOG`, `LOG.old.*`) grows without limit because log rotation and retention are left at RocksDB defaults.
- WebUI: A blob store read error at startup, such as an S3 authentication failure, stops the web interface from being downloaded.
## [0.16.24] - 2026-09-27 ## [0.16.24] - 2026-09-27
If you are upgrading from v0.16.x, replace the binary (or run `docker pull`). If you are upgrading from v0.15.x and below, please read the [upgrading documentation](https://github.com/stalwartlabs/stalwart/blob/main/UPGRADING/v0_16.md) for more information on how to upgrade from previous versions. If you are upgrading from v0.16.x, replace the binary (or run `docker pull`). If you are upgrading from v0.15.x and below, please read the [upgrading documentation](https://github.com/stalwartlabs/stalwart/blob/main/UPGRADING/v0_16.md) for more information on how to upgrade from previous versions.
+1 -1
View File
@@ -60,7 +60,7 @@ representative at an online or offline event.
Instances of abusive, harassing, or otherwise unacceptable behavior may be Instances of abusive, harassing, or otherwise unacceptable behavior may be
reported to the community leaders responsible for enforcement at reported to the community leaders responsible for enforcement at
**communityATcoffeylabsDOTorg**. **johnellisATlinuxDOTcom**.
All complaints will be reviewed and investigated promptly and fairly. All complaints will be reviewed and investigated promptly and fairly.
All community leaders are obligated to respect the privacy and security of the All community leaders are obligated to respect the privacy and security of the
+1 -1
View File
@@ -54,7 +54,7 @@ Coffey Labs" line in place. New files carry:
``` ```
/* /*
* SPDX-FileCopyrightText: 2026 Coffey Labs LLC * SPDX-FileCopyrightText: 2026 Coffey Labs
* *
* SPDX-License-Identifier: AGPL-3.0-only * SPDX-License-Identifier: AGPL-3.0-only
*/ */
Generated
+69 -87
View File
@@ -277,9 +277,9 @@ dependencies = [
[[package]] [[package]]
name = "async-compression" name = "async-compression"
version = "0.4.50" version = "0.4.48"
source = "registry+https://github.com/rust-lang/crates.io-index" source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "ee19bd99b43e3691acbad4e840420a4881cea6c0b66a208125a824f8fd53f5a1" checksum = "fb61aea1a7def73ee7c350a184f0e70b32c182344e2e75bf70c9b621b83417fd"
dependencies = [ dependencies = [
"compression-codecs", "compression-codecs",
"compression-core", "compression-core",
@@ -1292,7 +1292,7 @@ dependencies = [
[[package]] [[package]]
name = "common" name = "common"
version = "0.16.25" version = "0.16.24"
dependencies = [ dependencies = [
"aes-gcm-siv", "aes-gcm-siv",
"ahash", "ahash",
@@ -1392,9 +1392,9 @@ dependencies = [
[[package]] [[package]]
name = "compression-codecs" name = "compression-codecs"
version = "0.4.45" version = "0.4.43"
source = "registry+https://github.com/rust-lang/crates.io-index" source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "98fc98460ba0ad5317075d3632b8dfc45d0be8c4a49347c2a38272019717614a" checksum = "bef16c47ba2797aa6a909cc37d39911f3a6743811fe7408ac0b0cc0276b656e9"
dependencies = [ dependencies = [
"compression-core", "compression-core",
"flate2", "flate2",
@@ -1477,7 +1477,7 @@ checksum = "3d52eff69cd5e647efe296129160853a42795992097e8af39800e1060caeea9b"
[[package]] [[package]]
name = "coordinator" name = "coordinator"
version = "0.16.25" version = "0.16.24"
dependencies = [ dependencies = [
"async-nats", "async-nats",
"futures", "futures",
@@ -1889,7 +1889,7 @@ checksum = "4583a4551df46e2792f82ceeac45e850d2e2d5debba0b91f102385cda5b11f06"
[[package]] [[package]]
name = "dav" name = "dav"
version = "0.16.25" version = "0.16.24"
dependencies = [ dependencies = [
"calcard", "calcard",
"chrono", "chrono",
@@ -1912,7 +1912,7 @@ dependencies = [
[[package]] [[package]]
name = "dav-proto" name = "dav-proto"
version = "0.16.25" version = "0.16.24"
dependencies = [ dependencies = [
"calcard", "calcard",
"chrono", "chrono",
@@ -2125,7 +2125,7 @@ dependencies = [
[[package]] [[package]]
name = "directory" name = "directory"
version = "0.16.25" version = "0.16.24"
dependencies = [ dependencies = [
"ahash", "ahash",
"argon2 0.6.0", "argon2 0.6.0",
@@ -2366,7 +2366,7 @@ dependencies = [
[[package]] [[package]]
name = "email" name = "email"
version = "0.16.25" version = "0.16.24"
dependencies = [ dependencies = [
"aes 0.9.3", "aes 0.9.3",
"aes-gcm 0.11.1", "aes-gcm 0.11.1",
@@ -2406,16 +2406,6 @@ dependencies = [
"log", "log",
] ]
[[package]]
name = "encodify"
version = "1.0.1"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "798c447647dd23f673748f2b868ef309a01dd86aaae999182559d36f06ac82f0"
dependencies = [
"memchr",
"simdutf8",
]
[[package]] [[package]]
name = "encoding_rs" name = "encoding_rs"
version = "0.8.42" version = "0.8.42"
@@ -2484,7 +2474,7 @@ dependencies = [
[[package]] [[package]]
name = "event_macro" name = "event_macro"
version = "0.16.25" version = "0.16.24"
dependencies = [ dependencies = [
"quote", "quote",
"syn 3.0.6", "syn 3.0.6",
@@ -3012,7 +3002,7 @@ dependencies = [
[[package]] [[package]]
name = "groupware" name = "groupware"
version = "0.16.25" version = "0.16.24"
dependencies = [ dependencies = [
"ahash", "ahash",
"calcard", "calcard",
@@ -3299,7 +3289,7 @@ dependencies = [
[[package]] [[package]]
name = "http" name = "http"
version = "0.16.25" version = "0.16.24"
dependencies = [ dependencies = [
"async-stream", "async-stream",
"base64 0.23.1", "base64 0.23.1",
@@ -3395,7 +3385,7 @@ dependencies = [
[[package]] [[package]]
name = "http_proto" name = "http_proto"
version = "0.16.25" version = "0.16.24"
dependencies = [ dependencies = [
"common", "common",
"compact_str", "compact_str",
@@ -3882,7 +3872,7 @@ checksum = "65b27460c2c92b037f3f94c538ed9a3342f3fdf923606781629ccb35f82d042a"
[[package]] [[package]]
name = "imap" name = "imap"
version = "0.16.25" version = "0.16.24"
dependencies = [ dependencies = [
"ahash", "ahash",
"common", "common",
@@ -3907,7 +3897,7 @@ dependencies = [
[[package]] [[package]]
name = "imap_proto" name = "imap_proto"
version = "0.16.25" version = "0.16.24"
dependencies = [ dependencies = [
"ahash", "ahash",
"base64 0.23.1", "base64 0.23.1",
@@ -3922,7 +3912,7 @@ dependencies = [
[[package]] [[package]]
name = "inbuxa" name = "inbuxa"
version = "0.16.25" version = "0.16.24"
dependencies = [ dependencies = [
"common", "common",
"coordinator", "coordinator",
@@ -3930,7 +3920,7 @@ dependencies = [
"directory", "directory",
"email", "email",
"groupware", "groupware",
"http 0.16.25", "http 0.16.24",
"http_proto", "http_proto",
"imap", "imap",
"jmap", "jmap",
@@ -3957,14 +3947,9 @@ name = "inbuxa-features"
version = "0.16.22" version = "0.16.22"
dependencies = [ dependencies = [
"ahash", "ahash",
"aho-corasick",
"base64 0.23.1", "base64 0.23.1",
"flate2", "flate2",
"jmap_proto", "jmap_proto",
"mail-builder 1.0.0",
"mail-parser",
"quick-xml 0.41.0",
"regex",
"registry", "registry",
"serde", "serde",
"serde_json", "serde_json",
@@ -3976,7 +3961,6 @@ dependencies = [
"types", "types",
"utils", "utils",
"xxhash-rust", "xxhash-rust",
"zip",
] ]
[[package]] [[package]]
@@ -4219,7 +4203,7 @@ dependencies = [
[[package]] [[package]]
name = "jmap" name = "jmap"
version = "0.16.25" version = "0.16.24"
dependencies = [ dependencies = [
"async-stream", "async-stream",
"base64 0.23.1", "base64 0.23.1",
@@ -4268,14 +4252,14 @@ dependencies = [
[[package]] [[package]]
name = "jmap-client" name = "jmap-client"
version = "0.4.3" version = "0.4.2"
source = "registry+https://github.com/rust-lang/crates.io-index" source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "f5b5bc66252cc8e779ef1238f40ab93971d54ad00b5d7afb5c1d447a9647ed62" checksum = "4deab22e057d24e32122f0fc6e2d667a124fdd6a0d8ef3ed4f8a89923c11084f"
dependencies = [ dependencies = [
"ahash", "ahash",
"async-stream", "async-stream",
"base64 0.22.1",
"chrono", "chrono",
"encodify",
"futures-util", "futures-util",
"maybe-async", "maybe-async",
"parking_lot", "parking_lot",
@@ -4302,7 +4286,7 @@ dependencies = [
[[package]] [[package]]
name = "jmap_proto" name = "jmap_proto"
version = "0.16.25" version = "0.16.24"
dependencies = [ dependencies = [
"ahash", "ahash",
"calcard", "calcard",
@@ -4512,9 +4496,9 @@ dependencies = [
[[package]] [[package]]
name = "lazy_static" name = "lazy_static"
version = "1.5.1" version = "1.5.0"
source = "registry+https://github.com/rust-lang/crates.io-index" source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "20870f649af7073d53e38067b2a84312175d56ea15217e1b15bc83506ec50afb" checksum = "bbd2bcb4c963f2ddae06a2efc7e9f3591312473c50c6685e1f298068316e66fe"
dependencies = [ dependencies = [
"spin 0.9.9", "spin 0.9.9",
] ]
@@ -4808,7 +4792,7 @@ dependencies = [
[[package]] [[package]]
name = "managesieve" name = "managesieve"
version = "0.16.25" version = "0.16.24"
dependencies = [ dependencies = [
"common", "common",
"compact_str", "compact_str",
@@ -4943,7 +4927,7 @@ checksum = "c797b9d6bb23aab2fc369c65f871be49214f5c759af65bde26ffaaa2b646b492"
[[package]] [[package]]
name = "migration" name = "migration"
version = "0.16.25" version = "0.16.24"
dependencies = [ dependencies = [
"common", "common",
"email", "email",
@@ -5193,7 +5177,7 @@ dependencies = [
[[package]] [[package]]
name = "nlp" name = "nlp"
version = "0.16.25" version = "0.16.24"
dependencies = [ dependencies = [
"ahash", "ahash",
"hashify", "hashify",
@@ -5513,8 +5497,8 @@ dependencies = [
[[package]] [[package]]
name = "opentelemetry" name = "opentelemetry"
version = "0.33.0" version = "0.32.0"
source = "git+https://github.com/stalwartlabs/opentelemetry-rust#ae66e97b140f70e477ab710686aafce665cc2f8b" source = "git+https://github.com/stalwartlabs/opentelemetry-rust#80a14a3b6846f62f85506d68d2600c948fccc9d2"
dependencies = [ dependencies = [
"futures-core", "futures-core",
"futures-sink", "futures-sink",
@@ -5526,8 +5510,8 @@ dependencies = [
[[package]] [[package]]
name = "opentelemetry-http" name = "opentelemetry-http"
version = "0.33.0" version = "0.32.0"
source = "git+https://github.com/stalwartlabs/opentelemetry-rust#ae66e97b140f70e477ab710686aafce665cc2f8b" source = "git+https://github.com/stalwartlabs/opentelemetry-rust#80a14a3b6846f62f85506d68d2600c948fccc9d2"
dependencies = [ dependencies = [
"async-trait", "async-trait",
"bytes", "bytes",
@@ -5538,8 +5522,8 @@ dependencies = [
[[package]] [[package]]
name = "opentelemetry-otlp" name = "opentelemetry-otlp"
version = "0.33.0" version = "0.32.0"
source = "git+https://github.com/stalwartlabs/opentelemetry-rust#ae66e97b140f70e477ab710686aafce665cc2f8b" source = "git+https://github.com/stalwartlabs/opentelemetry-rust#80a14a3b6846f62f85506d68d2600c948fccc9d2"
dependencies = [ dependencies = [
"http 1.5.0", "http 1.5.0",
"httpdate", "httpdate",
@@ -5557,8 +5541,8 @@ dependencies = [
[[package]] [[package]]
name = "opentelemetry-proto" name = "opentelemetry-proto"
version = "0.33.0" version = "0.32.0"
source = "git+https://github.com/stalwartlabs/opentelemetry-rust#ae66e97b140f70e477ab710686aafce665cc2f8b" source = "git+https://github.com/stalwartlabs/opentelemetry-rust#80a14a3b6846f62f85506d68d2600c948fccc9d2"
dependencies = [ dependencies = [
"opentelemetry", "opentelemetry",
"opentelemetry_sdk", "opentelemetry_sdk",
@@ -5569,13 +5553,13 @@ dependencies = [
[[package]] [[package]]
name = "opentelemetry-semantic-conventions" name = "opentelemetry-semantic-conventions"
version = "0.33.0" version = "0.32.1"
source = "git+https://github.com/stalwartlabs/opentelemetry-rust#ae66e97b140f70e477ab710686aafce665cc2f8b" source = "git+https://github.com/stalwartlabs/opentelemetry-rust#80a14a3b6846f62f85506d68d2600c948fccc9d2"
[[package]] [[package]]
name = "opentelemetry_sdk" name = "opentelemetry_sdk"
version = "0.33.0" version = "0.32.1"
source = "git+https://github.com/stalwartlabs/opentelemetry-rust#ae66e97b140f70e477ab710686aafce665cc2f8b" source = "git+https://github.com/stalwartlabs/opentelemetry-rust#80a14a3b6846f62f85506d68d2600c948fccc9d2"
dependencies = [ dependencies = [
"futures-channel", "futures-channel",
"futures-executor", "futures-executor",
@@ -6025,7 +6009,7 @@ dependencies = [
[[package]] [[package]]
name = "pop3" name = "pop3"
version = "0.16.25" version = "0.16.24"
dependencies = [ dependencies = [
"common", "common",
"directory", "directory",
@@ -6395,9 +6379,9 @@ dependencies = [
[[package]] [[package]]
name = "quinn-proto" name = "quinn-proto"
version = "0.11.19" version = "0.11.18"
source = "registry+https://github.com/rust-lang/crates.io-index" source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "0e750cca55fe4f0439a15d0bb529da9651e79993e8e72c61a899a36d462befbe" checksum = "a9746dbde176634f4f2f1faf2404e30a31b2bc1e9cafb5329c95d8177a18c9fc"
dependencies = [ dependencies = [
"aws-lc-rs", "aws-lc-rs",
"bytes", "bytes",
@@ -6420,9 +6404,9 @@ dependencies = [
[[package]] [[package]]
name = "quinn-udp" name = "quinn-udp"
version = "0.5.16" version = "0.5.15"
source = "registry+https://github.com/rust-lang/crates.io-index" source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "af66907df18639dcf4db56ca65490cabc4b27a97dbadd96f2926cca73298f016" checksum = "35a133f956daabe89a61a685c2649f13d82d5aa4bd5d12d1277e1072a21c0694"
dependencies = [ dependencies = [
"cfg_aliases", "cfg_aliases",
"libc", "libc",
@@ -6845,7 +6829,7 @@ checksum = "d6f6ff9a378485b298a5286656da665ba74413d36db0979633275d2e708145d4"
[[package]] [[package]]
name = "registry" name = "registry"
version = "0.16.25" version = "0.16.24"
dependencies = [ dependencies = [
"ahash", "ahash",
"hashify", "hashify",
@@ -7402,7 +7386,7 @@ dependencies = [
[[package]] [[package]]
name = "scim" name = "scim"
version = "0.16.25" version = "0.16.24"
dependencies = [ dependencies = [
"ahash", "ahash",
"base64 0.23.1", "base64 0.23.1",
@@ -7428,7 +7412,7 @@ dependencies = [
[[package]] [[package]]
name = "scim-proto" name = "scim-proto"
version = "0.16.25" version = "0.16.24"
dependencies = [ dependencies = [
"hashify", "hashify",
"serde", "serde",
@@ -7682,9 +7666,9 @@ dependencies = [
[[package]] [[package]]
name = "serde_with" name = "serde_with"
version = "3.24.0" version = "3.23.0"
source = "registry+https://github.com/rust-lang/crates.io-index" source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "df9adc193c780ef8f159aee8b61e2d5801aaa555e6eb0947fe45530ec506296f" checksum = "935177bb8c0cd8ca1a4e6d1a2ac8988bea69cab4f9d3a31311e012ad27868ea4"
dependencies = [ dependencies = [
"base64 0.23.1", "base64 0.23.1",
"bs58", "bs58",
@@ -7703,9 +7687,9 @@ dependencies = [
[[package]] [[package]]
name = "serde_with_macros" name = "serde_with_macros"
version = "3.24.0" version = "3.23.0"
source = "registry+https://github.com/rust-lang/crates.io-index" source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "3e17bbc68e28663bbbb90df47e058aa7eda4fb445b89fe70457bb94fbccf6e49" checksum = "1d607aa01a3cb0ad757d6fd216136910db3c97b102fe686585689615a02dbcdc"
dependencies = [ dependencies = [
"darling 0.24.1", "darling 0.24.1",
"proc-macro2", "proc-macro2",
@@ -7763,7 +7747,7 @@ dependencies = [
[[package]] [[package]]
name = "services" name = "services"
version = "0.16.25" version = "0.16.24"
dependencies = [ dependencies = [
"aes-gcm 0.11.1", "aes-gcm 0.11.1",
"aho-corasick", "aho-corasick",
@@ -7772,13 +7756,11 @@ dependencies = [
"common", "common",
"dns-update", "dns-update",
"email", "email",
"futures",
"groupware", "groupware",
"hkdf 0.13.0", "hkdf 0.13.0",
"inbuxa-features", "inbuxa-features",
"jmap-tools", "jmap-tools",
"jmap_proto", "jmap_proto",
"mail-auth",
"mail-builder 1.0.0", "mail-builder 1.0.0",
"mail-parser", "mail-parser",
"memory-stats", "memory-stats",
@@ -8078,7 +8060,7 @@ checksum = "f9395f0f0eee849a9b707b2f06bb92a6a422090e2123bb2ef8e87a0e61892a8e"
[[package]] [[package]]
name = "smtp" name = "smtp"
version = "0.16.25" version = "0.16.24"
dependencies = [ dependencies = [
"ahash", "ahash",
"base64 0.23.1", "base64 0.23.1",
@@ -8117,9 +8099,9 @@ dependencies = [
[[package]] [[package]]
name = "smtp-proto" name = "smtp-proto"
version = "0.2.5" version = "0.2.4"
source = "registry+https://github.com/rust-lang/crates.io-index" source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "142a5a642c6bd7ffd7e1b525ad6a6e9921ccef992dba4663974f8519209a00c1" checksum = "707104487221ff447b5b796b5049e5c09cf52ff9fc1c8abba4695b89ac4b0f37"
dependencies = [ dependencies = [
"memchr", "memchr",
"rkyv", "rkyv",
@@ -8169,7 +8151,7 @@ dependencies = [
[[package]] [[package]]
name = "spam-filter" name = "spam-filter"
version = "0.16.25" version = "0.16.24"
dependencies = [ dependencies = [
"common", "common",
"compact_str", "compact_str",
@@ -8289,7 +8271,7 @@ checksum = "a2eb9349b6444b326872e140eb1cf5e7c522154d69e7a0ffb0fb81c06b37543f"
[[package]] [[package]]
name = "store" name = "store"
version = "0.16.25" version = "0.16.24"
dependencies = [ dependencies = [
"ahash", "ahash",
"arc-swap", "arc-swap",
@@ -8549,7 +8531,7 @@ dependencies = [
[[package]] [[package]]
name = "tests" name = "tests"
version = "0.16.25" version = "0.16.24"
dependencies = [ dependencies = [
"ahash", "ahash",
"aws-lc-rs", "aws-lc-rs",
@@ -8571,7 +8553,7 @@ dependencies = [
"form_urlencoded", "form_urlencoded",
"futures", "futures",
"groupware", "groupware",
"http 0.16.25", "http 0.16.24",
"http_proto", "http_proto",
"hyper", "hyper",
"hyper-util", "hyper-util",
@@ -8828,9 +8810,9 @@ dependencies = [
[[package]] [[package]]
name = "tokio-rustls" name = "tokio-rustls"
version = "0.26.6" version = "0.26.5"
source = "registry+https://github.com/rust-lang/crates.io-index" source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "c9cc2678c2cdd569ef8215e2afd7954ada2ae20b4fdd2c5fe6139a3b02d105db" checksum = "b0c85f2c3ef0b1cd58b36682f4b17aaa995f0e5db534d85692b4903abce21f67"
dependencies = [ dependencies = [
"rustls", "rustls",
"tokio", "tokio",
@@ -9164,7 +9146,7 @@ dependencies = [
[[package]] [[package]]
name = "trc" name = "trc"
version = "0.16.25" version = "0.16.24"
dependencies = [ dependencies = [
"ahash", "ahash",
"base64 0.23.1", "base64 0.23.1",
@@ -9273,7 +9255,7 @@ checksum = "b6f5e870be6c3b371b77fe0ee0bafb859fa4964b4404c27de1d380043c4dda20"
[[package]] [[package]]
name = "types" name = "types"
version = "0.16.25" version = "0.16.24"
dependencies = [ dependencies = [
"blake3", "blake3",
"compact_str", "compact_str",
@@ -9442,7 +9424,7 @@ checksum = "b6c140620e7ffbb22c2dee59cafe6084a59b5ffc27a8859a5f0d494b5d52b6be"
[[package]] [[package]]
name = "utils" name = "utils"
version = "0.16.25" version = "0.16.24"
dependencies = [ dependencies = [
"ahash", "ahash",
"arcstr", "arcstr",
@@ -10127,9 +10109,9 @@ dependencies = [
[[package]] [[package]]
name = "xxhash-rust" name = "xxhash-rust"
version = "0.8.19" version = "0.8.18"
source = "registry+https://github.com/rust-lang/crates.io-index" source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "550a2b930b62486a393c52d5c3b84bff264b28aa437ed64694d31e93b1757af7" checksum = "aee1b19627c7c60102ab80d3a9cbe18de90bfe03bfa6c3715447681f0e8c8af6"
[[package]] [[package]]
name = "yasna" name = "yasna"
@@ -10154,9 +10136,9 @@ dependencies = [
[[package]] [[package]]
name = "yoke-derive" name = "yoke-derive"
version = "0.8.4" version = "0.8.3"
source = "registry+https://github.com/rust-lang/crates.io-index" source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "ec8ebde2db3681e8c9980cc27822030e68752690ddfa9473e739aeb4dbde6d71" checksum = "33811428bee40dbceb6d545e95754741d17a6aef9a4849f0fd62e2ba4f412a78"
dependencies = [ dependencies = [
"proc-macro2", "proc-macro2",
"quote", "quote",
+1 -1
View File
@@ -4,7 +4,7 @@
# ***************** # *****************
# Base image for planner & builder # Base image for planner & builder
# ***************** # *****************
FROM --platform=$BUILDPLATFORM rust:1.98.1-slim-trixie AS base FROM --platform=$BUILDPLATFORM rust:slim-trixie AS base
ENV DEBIAN_FRONTEND="noninteractive" \ ENV DEBIAN_FRONTEND="noninteractive" \
BINSTALL_DISABLE_TELEMETRY=true \ BINSTALL_DISABLE_TELEMETRY=true \
-4
View File
@@ -8,10 +8,6 @@
--- ---
> [!NOTE]
> Development happens on [git.coffeylabs.org/inbuxa/inbuxa-server](https://git.coffeylabs.org/inbuxa/inbuxa-server); the copy on GitHub is a read-only mirror.
> Report issues at **[git.coffeylabs.org/inbuxa/inbuxa-server/issues](https://git.coffeylabs.org/inbuxa/inbuxa-server/issues)**, and join discussions at **[community.coffeylabs.org](https://community.coffeylabs.org)**.
**inbuxa** is a mail and collaboration server: JMAP, IMAP, POP3, SMTP, **inbuxa** is a mail and collaboration server: JMAP, IMAP, POP3, SMTP,
CalDAV, CardDAV and WebDAV, in one Rust binary, with ihasmail as its web front CalDAV, CardDAV and WebDAV, in one Rust binary, with ihasmail as its web front
end. It is a fork of [Stalwart](https://github.com/stalwartlabs/stalwart). end. It is a fork of [Stalwart](https://github.com/stalwartlabs/stalwart).
+2 -2
View File
@@ -17,7 +17,7 @@ visible to everyone, including whoever would use it, before there is a fix.
Report it privately by email to: Report it privately by email to:
**securityATcoffeylabsDOTorg** **johnellisATlinuxDOTcom**
Include as much as you can of: Include as much as you can of:
@@ -36,7 +36,7 @@ to Stalwart Labs with credit to you, and you'll be told that has happened.
This repository is the mail server. The web front ends have their own: This repository is the mail server. The web front ends have their own:
- [inbuxa-admin](https://git.coffeylabs.org/inbuxa/inbuxa-admin) - [inbuxa-admin](https://git.coffeylabs.org/inbuxa/inbuxa-admin)
- [inbuxa-webmail](https://git.coffeylabs.org/inbuxa/inbuxa-webmail) - [ihasmail-inbuxa](https://git.coffeylabs.org/inbuxa/ihasmail-inbuxa)
Upstream's own security documents are kept in `.github-upstream/` for Upstream's own security documents are kept in `.github-upstream/` for
reference. They describe Stalwart Labs' process, not this project's. reference. They describe Stalwart Labs' process, not this project's.
+1 -1
View File
@@ -1,6 +1,6 @@
[package] [package]
name = "common" name = "common"
version = "0.16.25" version = "0.16.24"
edition = "2024" edition = "2024"
build = "build.rs" build = "build.rs"
+1 -58
View File
@@ -1,5 +1,5 @@
/* /*
* SPDX-FileCopyrightText: 2026 Coffey Labs LLC * SPDX-FileCopyrightText: 2026 Coffey Labs
* *
* SPDX-License-Identifier: AGPL-3.0-only * SPDX-License-Identifier: AGPL-3.0-only
*/ */
@@ -445,63 +445,6 @@ impl Server {
} }
} }
/// MA-D0a: a message sent from an address that isn't the sender's own:
/// a group's or a shared mailbox's. The message itself only says
/// `From:` that address, so the audit log is where the person who sent
/// it is named. A locked account's delegate's send is AL-9's record, not
/// this one.
pub async fn audit_send_as(
&self,
token: &AccessToken,
submission_account_id: u32,
submission_id: u32,
address: &str,
) {
let Ok(Some(as_account_id)) = self.account_id_from_email(address, true).await else {
return;
};
if as_account_id == token.account_id()
|| token
.delegation(as_account_id)
.is_some_and(|delegation| delegation.kind.is_lock())
{
return;
}
let actor = self.audit_actor(token).await;
let tenant_id = self
.account(as_account_id)
.await
.ok()
.and_then(|account| account.id_tenant);
let details = if submission_account_id == as_account_id {
format!("Sent as {address}")
} else {
format!(
"Sent as {address}, from {}",
self.audit_account_name(submission_account_id).await
)
};
self.audit_note(Record {
at: ms(),
actor,
via: token.origin().cloned(),
remote_ip: None,
action: Action::Create,
target: Target {
kind: "EmailSubmission".into(),
id: Some(Id::from(submission_id).to_string()),
name: Some(address.to_string()),
account_id: Some(as_account_id),
tenant_id,
},
changes: vec![],
details: Some(details),
reason: None,
outcome: Outcome::success(),
})
.await;
}
/// AU-7: removes entries past the retention period. /// AU-7: removes entries past the retention period.
pub async fn audit_purge(&self) -> trc::Result<usize> { pub async fn audit_purge(&self) -> trc::Result<usize> {
let settings = log::settings(self.store()).await?; let settings = log::settings(self.store()).await?;
+4 -84
View File
@@ -36,32 +36,6 @@ use utils::map::bitmap::{Bitmap, BitmapItem};
use xxhash_rust::xxh3; use xxhash_rust::xxh3;
impl Server { impl Server {
/// inbuxa: MA-C: whether people in `owner`'s tenant may share their mail
/// (the server's switch, narrowed by the tenant's).
pub async fn mail_sharing_allowed(&self, owner: u32) -> trc::Result<bool> {
let tenant_id = self.account(owner).await.ok().and_then(|account| account.id_tenant);
Ok(
inbuxa_features::security::sharing_policy::effective_for(self.store(), tenant_id)
.await
.caused_by(trc::location!())?
.mail_sharing,
)
}
/// inbuxa: MA-C: whether `owner`'s mail shares give access now. A locked
/// account's or shared mailbox's grants are an administrator's and always
/// do; anyone else's only while their tenant allows mail sharing.
pub async fn mail_shares_honored(&self, owner: u32) -> trc::Result<bool> {
if inbuxa_features::lock::get(self.store(), owner)
.await
.caused_by(trc::location!())?
.is_some()
{
return Ok(true);
}
self.mail_sharing_allowed(owner).await
}
async fn build_access_token( async fn build_access_token(
&self, &self,
account: Account, account: Account,
@@ -72,22 +46,19 @@ impl Server {
// inbuxa: AL-2, AL-5: whether this account is locked, and which // inbuxa: AL-2, AL-5: whether this account is locked, and which
// locked accounts are handed to it. The token is their cache: every // locked accounts are handed to it. The token is their cache: every
// change to a lock invalidates the tokens it touches. // change to a lock invalidates the tokens it touches.
let lock_kind = inbuxa_features::lock::get(self.store(), account_id) let locked = inbuxa_features::lock::get(self.store(), account_id)
.await .await
.caused_by(trc::location!())? .caused_by(trc::location!())?
.map(|lock| lock.kind); .is_some();
let locked = lock_kind.is_some();
let shared_mailbox = lock_kind == Some(inbuxa_features::lock::Kind::SharedMailbox);
let now_secs = now(); let now_secs = now();
let delegations: Box<[super::Delegation]> = let delegations: Box<[super::Delegation]> =
inbuxa_features::lock::delegated_to(self.store(), account_id) inbuxa_features::lock::delegated_to(self.store(), account_id)
.await .await
.caused_by(trc::location!())? .caused_by(trc::location!())?
.into_iter() .into_iter()
.filter(|(_, delegate, _)| delegate.is_current(now_secs)) .filter(|(_, delegate)| delegate.is_current(now_secs))
.map(|(locked_id, delegate, kind)| super::Delegation { .map(|(locked_id, delegate)| super::Delegation {
account_id: locked_id, account_id: locked_id,
kind,
access: delegate.access, access: delegate.access,
send_as: delegate.send_as, send_as: delegate.send_as,
until: delegate.until, until: delegate.until,
@@ -126,9 +97,6 @@ impl Server {
.map(|m| m.id() as u32) .map(|m| m.id() as u32)
.collect::<TinyVec<[u32; 3]>>(); .collect::<TinyVec<[u32; 3]>>();
let mut access_to: Vec<AccessTo> = Vec::new(); let mut access_to: Vec<AccessTo> = Vec::new();
// inbuxa: MA-C: whether an owner's mail shares are honored,
// looked up once per owner
let mut mail_shares_honored: Vec<(u32, bool)> = Vec::new();
for grant_account_id in [account_id].into_iter().chain(member_of.iter().copied()) { for grant_account_id in [account_id].into_iter().chain(member_of.iter().copied()) {
for acl_item in self for acl_item in self
.store() .store()
@@ -149,27 +117,6 @@ impl Server {
.caused_by(trc::location!())); .caused_by(trc::location!()));
} }
// inbuxa: MA-C: a mail share from an account whose
// tenant (or server) has mail sharing off gives
// nothing while it is off. It stays stored, so it
// comes back when sharing does. A lock's and a
// shared mailbox's grants are an administrator's,
// and always count.
if collection == Collection::Mailbox {
let owner = acl_item.to_account_id;
let honored = match mail_shares_honored.iter().find(|(id, _)| *id == owner) {
Some((_, honored)) => *honored,
None => {
let honored = self.mail_shares_honored(owner).await?;
mail_shares_honored.push((owner, honored));
honored
}
};
if !honored {
continue;
}
}
let mut collections: Bitmap<Collection> = Bitmap::new(); let mut collections: Bitmap<Collection> = Bitmap::new();
if acl.contains(Acl::Read) { if acl.contains(Acl::Read) {
collections.insert(collection); collections.insert(collection);
@@ -300,7 +247,6 @@ impl Server {
.map(ConcurrencyLimiter::new), .map(ConcurrencyLimiter::new),
obj_size: 0, obj_size: 0,
locked, locked,
shared_mailbox,
delegations: delegations.clone(), delegations: delegations.clone(),
revision, revision,
revision_account, revision_account,
@@ -354,7 +300,6 @@ impl Server {
.map(ConcurrencyLimiter::new), .map(ConcurrencyLimiter::new),
obj_size: 0, obj_size: 0,
locked, locked,
shared_mailbox,
delegations: delegations.clone(), delegations: delegations.clone(),
revision, revision,
revision_account, revision_account,
@@ -608,16 +553,6 @@ impl AccessToken {
|| self.inner.access_to.iter().any(|a| a.account_id == account_id) || self.inner.access_to.iter().any(|a| a.account_id == account_id)
} }
/// inbuxa: MA-D0: in the account only because it is a group this token
/// belongs to. Such a member has the group's mailbox but may not share it
/// on: who is in a group is an administrator's decision, and a share
/// would let anyone in.
pub fn is_group_member_only(&self, account_id: u32) -> bool {
self.inner.account_id != account_id
&& self.inner.member_of.contains(&account_id)
&& !self.has_permission(Permission::Impersonate)
}
pub fn is_account_id(&self, account_id: u32) -> bool { pub fn is_account_id(&self, account_id: u32) -> bool {
self.inner.account_id == account_id self.inner.account_id == account_id
} }
@@ -713,7 +648,6 @@ impl AccessToken {
credential_version: old_inner.credential_version, credential_version: old_inner.credential_version,
obj_size: old_inner.obj_size, obj_size: old_inner.obj_size,
locked: old_inner.locked, locked: old_inner.locked,
shared_mailbox: old_inner.shared_mailbox,
delegations: old_inner.delegations.clone(), delegations: old_inner.delegations.clone(),
}; };
@@ -904,18 +838,6 @@ impl AccessToken {
self.inner.locked self.inner.locked
} }
/// inbuxa: MA-S: the account is a shared mailbox (a lock of that kind).
pub fn is_shared_mailbox(&self) -> bool {
self.inner.shared_mailbox
}
/// inbuxa: MA-S: this account's delegation into `account_id` is to a
/// shared mailbox, not a locked account.
pub fn delegated_shared_mailbox(&self, account_id: u32) -> bool {
self.delegation(account_id)
.is_some_and(|d| d.kind == inbuxa_features::lock::Kind::SharedMailbox)
}
/// inbuxa: AL-5: this account's delegation into a locked account, if it /// inbuxa: AL-5: this account's delegation into a locked account, if it
/// has one that hasn't ended. /// has one that hasn't ended.
/// inbuxa: AL-6, AL-7: a delegate at organize or full, who may add to /// inbuxa: AL-6, AL-7: a delegate at organize or full, who may add to
@@ -996,7 +918,6 @@ impl AccessToken {
credential_version: Default::default(), credential_version: Default::default(),
obj_size: Default::default(), obj_size: Default::default(),
locked: false, locked: false,
shared_mailbox: false,
delegations: Default::default(), delegations: Default::default(),
}), }),
} }
@@ -1057,7 +978,6 @@ impl AccessTokenInner {
credential_version: Default::default(), credential_version: Default::default(),
obj_size: Default::default(), obj_size: Default::default(),
locked: false, locked: false,
shared_mailbox: false,
delegations: Default::default(), delegations: Default::default(),
} }
} }
-4
View File
@@ -152,8 +152,6 @@ pub struct AccessTokenInner {
pub(crate) obj_size: u64, pub(crate) obj_size: u64,
// inbuxa: AL-2: the account is locked; it may not authenticate // inbuxa: AL-2: the account is locked; it may not authenticate
pub(crate) locked: bool, pub(crate) locked: bool,
// inbuxa: MA-S: the lock is a shared mailbox
pub(crate) shared_mailbox: bool,
// inbuxa: AL-5: locked accounts handed to this one // inbuxa: AL-5: locked accounts handed to this one
pub(crate) delegations: Box<[Delegation]>, pub(crate) delegations: Box<[Delegation]>,
} }
@@ -167,8 +165,6 @@ pub struct Delegation {
pub send_as: bool, pub send_as: bool,
/// Seconds since the epoch. /// Seconds since the epoch.
pub until: Option<u64>, pub until: Option<u64>,
/// MA-S: a locked account, or a shared mailbox.
pub kind: inbuxa_features::lock::Kind,
} }
#[derive(Debug, Default, Hash, Clone)] #[derive(Debug, Default, Hash, Clone)]
-42
View File
@@ -111,9 +111,6 @@ impl Server {
Permission::SysLegalHoldCreate, Permission::SysLegalHoldCreate,
Permission::SysLegalHoldUpdate, Permission::SysLegalHoldUpdate,
Permission::SysLegalHoldExport, Permission::SysLegalHoldExport,
// inbuxa: DL-20: the lists and the check are the server's
Permission::SysDeliverabilityUpdate,
Permission::SysDeliverabilityCheck,
] { ] {
permissions.disabled.set(permission as usize); permissions.disabled.set(permission as usize);
} }
@@ -168,14 +165,6 @@ impl AccessToken {
mut requested_permissions: Permissions, mut requested_permissions: Permissions,
) -> Result<(), Vec<Permission>> { ) -> Result<(), Vec<Permission>> {
requested_permissions.difference(self.permissions_bits()); requested_permissions.difference(self.permissions_bits());
// inbuxa: journaling, JR-18: whoever sets up journals may give
// others (or, through a role, themselves) the reading of them,
// which administrators don't hold by default; the role change is
// in the audit log
if self.has_permission(Permission::SysJournalUpdate) {
requested_permissions.clear(Permission::SysJournalSearch as usize);
requested_permissions.clear(Permission::SysJournalExport as usize);
}
if requested_permissions.is_empty() { if requested_permissions.is_empty() {
Ok(()) Ok(())
} else { } else {
@@ -307,37 +296,6 @@ impl Default for DefaultPermissions {
default.superuser.push(permission); default.superuser.push(permission);
default.tenant.push(permission); default.tenant.push(permission);
} }
// inbuxa: deliverability spec, DL-20: a tenant administrator
// reads its own domains' findings; the lists and the check
// itself are the server's
Permission::SysDeliverabilityGet => {
default.superuser.push(permission);
default.tenant.push(permission);
}
Permission::SysDeliverabilityUpdate | Permission::SysDeliverabilityCheck => {
default.superuser.push(permission);
}
// inbuxa: DLP and mail flow rules, and held mail, are the
// server's: never a tenant's (dlp-and-mail-flow-rules spec,
// settled answer 3)
Permission::SysMailRuleGet
| Permission::SysMailRuleUpdate
| Permission::SysDlpPolicyGet
| Permission::SysDlpPolicyUpdate
| Permission::SysDlpReviewGet
| Permission::SysDlpReviewUpdate
// inbuxa: every security check is server-wide (security
// to-do list spec)
| Permission::SysSecurityAccept => {
default.superuser.push(permission);
}
// inbuxa: journals are the server's; administrators set them
// up but read what's journaled only if granted it
// (journaling spec, JR-18, settled answer 5)
Permission::SysJournalGet | Permission::SysJournalUpdate => {
default.superuser.push(permission);
}
Permission::SysJournalSearch | Permission::SysJournalExport => {}
// inbuxa: AL-12: tenant administrators lock and delegate // inbuxa: AL-12: tenant administrators lock and delegate
// within their tenant // within their tenant
Permission::SysAccountLockGet Permission::SysAccountLockGet
-34
View File
@@ -72,10 +72,6 @@ pub struct Http {
pub cors_origins: Vec<hyper::header::HeaderValue>, pub cors_origins: Vec<hyper::header::HeaderValue>,
pub use_forwarded: bool, pub use_forwarded: bool,
pub redirect_root: Option<String>, pub redirect_root: Option<String>,
/// inbuxa: HTTP Basic accepted on every endpoint, not only DAV (contract
/// C-23). True in bootstrap and recovery mode, or with
/// `INBUXA_HTTP_BASIC_AUTH=all`.
pub basic_auth_everywhere: bool,
} }
#[derive(Clone)] #[derive(Clone)]
@@ -457,35 +453,6 @@ impl Http {
.collect() .collect()
}; };
// inbuxa: outside DAV, HTTP sign-in is a token unless the operator
// says otherwise (contract C-23). The integration suites sign in with
// passwords over JMAP and the API, so test builds accept Basic
// everywhere.
#[cfg(feature = "test_mode")]
let basic_auth_everywhere = true;
#[cfg(not(feature = "test_mode"))]
let basic_auth_everywhere = bp.registry.is_recovery_mode()
|| bp.registry.is_bootstrap_mode()
|| match types::branding::env_var("HTTP_BASIC_AUTH") {
Ok(value) if value.trim().eq_ignore_ascii_case("all") => true,
Ok(value)
if value.trim().is_empty() || value.trim().eq_ignore_ascii_case("dav") =>
{
false
}
Ok(value) => {
bp.build_warning(
ObjectType::Http.singleton(),
format!(
"INBUXA_HTTP_BASIC_AUTH is {value:?}; expected \"dav\" or \"all\". Basic authentication stays on DAV only."
),
);
false
}
Err(_) => false,
};
if use_permissive_cors { if use_permissive_cors {
http_headers.push(( http_headers.push((
hyper::header::ACCESS_CONTROL_ALLOW_ORIGIN, hyper::header::ACCESS_CONTROL_ALLOW_ORIGIN,
@@ -545,7 +512,6 @@ impl Http {
cors_origins, cors_origins,
use_forwarded: http.use_x_forwarded, use_forwarded: http.use_x_forwarded,
redirect_root: http.redirect_root, redirect_root: http.redirect_root,
basic_auth_everywhere,
} }
} }
} }
+1 -1
View File
@@ -1,5 +1,5 @@
/* /*
* SPDX-FileCopyrightText: 2026 Coffey Labs LLC * SPDX-FileCopyrightText: 2026 Coffey Labs
* *
* SPDX-License-Identifier: AGPL-3.0-only * SPDX-License-Identifier: AGPL-3.0-only
*/ */
+1 -1
View File
@@ -1,5 +1,5 @@
/* /*
* SPDX-FileCopyrightText: 2026 Coffey Labs LLC * SPDX-FileCopyrightText: 2026 Coffey Labs
* *
* SPDX-License-Identifier: AGPL-3.0-only * SPDX-License-Identifier: AGPL-3.0-only
*/ */
+1 -1
View File
@@ -1,5 +1,5 @@
/* /*
* SPDX-FileCopyrightText: 2026 Coffey Labs LLC * SPDX-FileCopyrightText: 2026 Coffey Labs
* *
* SPDX-License-Identifier: AGPL-3.0-only * SPDX-License-Identifier: AGPL-3.0-only
*/ */
-2
View File
@@ -88,8 +88,6 @@ pub enum BroadcastEvent {
QueueRefresh, QueueRefresh,
// inbuxa: AL-3: end an account's open sessions on every node // inbuxa: AL-3: end an account's open sessions on every node
EndSessions(u32), EndSessions(u32),
// inbuxa: deliverability spec, DL-15: every node checks itself now
DeliverabilityCheck,
} }
#[derive(Debug, Clone, Copy)] #[derive(Debug, Clone, Copy)]
+1 -1
View File
@@ -225,7 +225,7 @@ pub struct Caches {
pub dns_ipv6: CacheWithTtl<Box<str>, RecordSet<Ipv6Addr>>, pub dns_ipv6: CacheWithTtl<Box<str>, RecordSet<Ipv6Addr>>,
pub dns_tlsa: CacheWithTtl<Box<str>, Arc<Tlsa>>, pub dns_tlsa: CacheWithTtl<Box<str>, Arc<Tlsa>>,
pub dns_mta_sts: CacheWithTtl<Box<str>, Arc<Policy>>, pub dns_mta_sts: CacheWithTtl<Box<str>, Arc<Policy>>,
pub dns_rbl: CacheWithTtl<Box<str>, Option<Arc<[IpResolver]>>>, pub dns_rbl: CacheWithTtl<Box<str>, Option<Arc<IpResolver>>>,
pub negative_cache_ttl: Duration, pub negative_cache_ttl: Duration,
} }
+2 -14
View File
@@ -227,22 +227,10 @@ impl WebApplicationManager {
let cached = if force_refresh { let cached = if force_refresh {
None None
} else { } else {
match server server
.blob_store() .blob_store()
.get_blob(self.blob_key.as_slice(), 0..usize::MAX) .get_blob(self.blob_key.as_slice(), 0..usize::MAX)
.await .await?
{
Ok(cached) => cached,
Err(err) => {
trc::event!(
Resource(trc::ResourceEvent::Error),
Reason = err,
Url = self.url.clone(),
Details = "Failed to read cached application bundle, downloading it again"
);
None
}
}
}; };
let is_cached = cached.is_some(); let is_cached = cached.is_some();
let bundle = match cached { let bundle = match cached {
+8 -39
View File
@@ -1,5 +1,5 @@
/* /*
* SPDX-FileCopyrightText: 2026 Coffey Labs LLC * SPDX-FileCopyrightText: 2026 Coffey Labs
* *
* SPDX-License-Identifier: AGPL-3.0-only * SPDX-License-Identifier: AGPL-3.0-only
*/ */
@@ -65,14 +65,6 @@ const OFFICER: &[Permission] = &[
Permission::SysLegalHoldUpdate, Permission::SysLegalHoldUpdate,
Permission::SysLegalHoldExport, Permission::SysLegalHoldExport,
Permission::SysAccountLockGet, Permission::SysAccountLockGet,
// dlp-and-mail-flow-rules spec, §2.8: see DLP rules, review held mail
Permission::SysDlpPolicyGet,
Permission::SysDlpReviewGet,
Permission::SysDlpReviewUpdate,
// journaling spec, JR-18: see journals, search and export them
Permission::SysJournalGet,
Permission::SysJournalSearch,
Permission::SysJournalExport,
]; ];
/// What a tenant's officer holds besides [`READS`]. /// What a tenant's officer holds besides [`READS`].
@@ -121,11 +113,6 @@ fn created_key(tenant: Option<Id>) -> ValueClass {
}) })
} }
/// The server-level Compliance Officer role the server made, if it has.
pub async fn server_role(data: &Store) -> trc::Result<Option<Id>> {
recorded(data, None).await
}
async fn recorded(data: &Store, tenant: Option<Id>) -> trc::Result<Option<Id>> { async fn recorded(data: &Store, tenant: Option<Id>) -> trc::Result<Option<Id>> {
Ok(data Ok(data
.get_value::<u64>(ValueKey::from(created_key(tenant))) .get_value::<u64>(ValueKey::from(created_key(tenant)))
@@ -185,19 +172,13 @@ pub async fn ensure_compliance_roles(registry: &RegistryStore, data: &Store) ->
/// A new tenant gets its Compliance Officer role. /// A new tenant gets its Compliance Officer role.
pub async fn tenant_created(registry: &RegistryStore, data: &Store, tenant: Id) -> trc::Result<()> { pub async fn tenant_created(registry: &RegistryStore, data: &Store, tenant: Id) -> trc::Result<()> {
create_once(registry, data, Some(tenant), tenant_role(tenant)) create_once(registry, data, Some(tenant), tenant_role(tenant)).await.map(|_| ())
.await
.map(|_| ())
} }
/// Before a tenant is deleted: removes its Compliance Officer role if nobody /// Before a tenant is deleted: removes its Compliance Officer role if nobody
/// holds it, so the role doesn't block the delete. Returns whether it did, /// holds it, so the role doesn't block the delete. Returns whether it did,
/// so a delete refused for another reason can put it back. /// so a delete refused for another reason can put it back.
pub async fn tenant_deleting( pub async fn tenant_deleting(registry: &RegistryStore, data: &Store, tenant: Id) -> trc::Result<bool> {
registry: &RegistryStore,
data: &Store,
tenant: Id,
) -> trc::Result<bool> {
let Some(role) = recorded(data, Some(tenant)).await? else { let Some(role) = recorded(data, Some(tenant)).await? else {
return Ok(false); return Ok(false);
}; };
@@ -239,9 +220,7 @@ mod tests {
// Beyond what any user holds for their own account // Beyond what any user holds for their own account
for permission in all.into_iter().filter(|p| !user.contains(p)) { for permission in all.into_iter().filter(|p| !user.contains(p)) {
let name = permission.as_str(); let name = permission.as_str();
// Placing holds and reviewing held mail are the officer's let holds = name.starts_with("sysLegalHold");
// job, not settings (settled answers 2 and 4)
let holds = name.starts_with("sysLegalHold") || name.starts_with("sysDlpReview");
assert!( assert!(
!(name.ends_with("Update") && !holds) !(name.ends_with("Update") && !holds)
&& !(name.ends_with("Create") && !holds) && !(name.ends_with("Create") && !holds)
@@ -270,11 +249,7 @@ mod tests {
assert!(officer.contains(&hold)); assert!(officer.contains(&hold));
assert!(!tenant.contains(&hold)); assert!(!tenant.contains(&hold));
} }
for both in [ for both in [Permission::SysComplianceGet, Permission::SysAuditGet, Permission::SysAccountGet] {
Permission::SysComplianceGet,
Permission::SysAuditGet,
Permission::SysAccountGet,
] {
assert!(officer.contains(&both) && tenant.contains(&both)); assert!(officer.contains(&both) && tenant.contains(&both));
} }
assert!(!officer.contains(&Permission::SysAuditSettingsUpdate)); assert!(!officer.contains(&Permission::SysAuditSettingsUpdate));
@@ -282,15 +257,9 @@ mod tests {
#[test] #[test]
fn records_are_per_place() { fn records_are_per_place() {
let ValueClass::Any(server) = created_key(None) else { let ValueClass::Any(server) = created_key(None) else { panic!() };
panic!() let ValueClass::Any(a) = created_key(Some(Id::from(1u64))) else { panic!() };
}; let ValueClass::Any(b) = created_key(Some(Id::from(2u64))) else { panic!() };
let ValueClass::Any(a) = created_key(Some(Id::from(1u64))) else {
panic!()
};
let ValueClass::Any(b) = created_key(Some(Id::from(2u64))) else {
panic!()
};
assert_eq!(server.key, b"Pc"); assert_eq!(server.key, b"Pc");
assert_ne!(a.key, b.key); assert_ne!(a.key, b.key);
assert!(a.key.starts_with(b"Pc")); assert!(a.key.starts_with(b"Pc"));
+3 -3
View File
@@ -1,5 +1,5 @@
/* /*
* SPDX-FileCopyrightText: 2026 Coffey Labs LLC * SPDX-FileCopyrightText: 2026 Coffey Labs
* *
* SPDX-License-Identifier: AGPL-3.0-only * SPDX-License-Identifier: AGPL-3.0-only
*/ */
@@ -14,7 +14,7 @@
//! application names another; //! application names another;
//! - INBUXA Admin hosted elsewhere, as `inbuxa-admin`, when `INBUXA_ADMIN_URL` //! - INBUXA Admin hosted elsewhere, as `inbuxa-admin`, when `INBUXA_ADMIN_URL`
//! is set; //! is set;
//! - inbuxa-webmail, as the confidential client `ihasmail-inbuxa`, when //! - ihasmail-inbuxa, as the confidential client `ihasmail-inbuxa`, when
//! `INBUXA_WEBMAIL_URL` and `INBUXA_WEBMAIL_CLIENT_SECRET` are set. //! `INBUXA_WEBMAIL_URL` and `INBUXA_WEBMAIL_CLIENT_SECRET` are set.
//! //!
//! inbuxa: the environment variables stand in for `x:FrontEnds` (C-4) until //! inbuxa: the environment variables stand in for `x:FrontEnds` (C-4) until
@@ -22,7 +22,7 @@
//! it instead. //! it instead.
//! //!
//! A missing client is created. An existing one gains any redirect URI it //! A missing client is created. An existing one gains any redirect URI it
//! lacks and, for inbuxa-webmail, the configured secret; nothing an operator //! lacks and, for ihasmail-inbuxa, the configured secret; nothing an operator
//! added is removed. //! added is removed.
use directory::core::secret::{hash_secret, verify_secret_hash}; use directory::core::secret::{hash_secret, verify_secret_hash};
@@ -1,5 +1,5 @@
/* /*
* SPDX-FileCopyrightText: 2026 Coffey Labs LLC * SPDX-FileCopyrightText: 2026 Coffey Labs
* *
* SPDX-License-Identifier: AGPL-3.0-only * SPDX-License-Identifier: AGPL-3.0-only
*/ */
@@ -31,9 +31,7 @@ use types::id::Id;
/// Granted to the default administrator roles: "Explain this" /// Granted to the default administrator roles: "Explain this"
/// (ai-explain spec, EX-4: superuser by default), the audit log, account /// (ai-explain spec, EX-4: superuser by default), the audit log, account
/// locks and legal holds (audit-hold-lock spec, AU-9, AL-12, LH-13), and /// locks and legal holds (audit-hold-lock spec, AU-9, AL-12, LH-13), and
/// the data inventory (personal-data catalog spec), accepting security /// the data inventory (personal-data catalog spec).
/// to-do items (security to-do list spec), and the deliverability check
/// (deliverability spec).
const ADMIN_GRANTS: &[Permission] = &[ const ADMIN_GRANTS: &[Permission] = &[
Permission::SysAiExplain, Permission::SysAiExplain,
Permission::SysAuditGet, Permission::SysAuditGet,
@@ -48,37 +46,11 @@ const ADMIN_GRANTS: &[Permission] = &[
Permission::SysLegalHoldUpdate, Permission::SysLegalHoldUpdate,
Permission::SysLegalHoldExport, Permission::SysLegalHoldExport,
Permission::SysComplianceGet, Permission::SysComplianceGet,
Permission::SysMailRuleGet,
Permission::SysMailRuleUpdate,
Permission::SysDlpPolicyGet,
Permission::SysDlpPolicyUpdate,
Permission::SysDlpReviewGet,
Permission::SysDlpReviewUpdate,
Permission::SysJournalGet,
Permission::SysJournalUpdate,
Permission::SysSecurityAccept,
Permission::SysDeliverabilityGet,
Permission::SysDeliverabilityUpdate,
Permission::SysDeliverabilityCheck,
];
/// Granted to the server-level Compliance Officer role once it exists:
/// seeing DLP rules and reviewing held mail (dlp-and-mail-flow-rules spec,
/// §2.8, settled answer 4). A new install's role has them from the start.
const OFFICER_GRANTS: &[Permission] = &[
Permission::SysDlpPolicyGet,
Permission::SysDlpReviewGet,
Permission::SysDlpReviewUpdate,
// journaling spec, JR-18: see journals, search and export them
Permission::SysJournalGet,
Permission::SysJournalSearch,
Permission::SysJournalExport,
]; ];
/// Granted to the default tenant administrator roles: reading and exporting /// Granted to the default tenant administrator roles: reading and exporting
/// the tenant's audit log (AU-9), locking and delegating its accounts /// the tenant's audit log (AU-9), locking and delegating its accounts
/// (AL-12), the tenant's slice of the data inventory, and its own domains' /// (AL-12), and the tenant's slice of the data inventory.
/// deliverability findings (DL-20).
const TENANT_GRANTS: &[Permission] = &[ const TENANT_GRANTS: &[Permission] = &[
Permission::SysAuditGet, Permission::SysAuditGet,
Permission::SysAuditExport, Permission::SysAuditExport,
@@ -87,23 +59,19 @@ const TENANT_GRANTS: &[Permission] = &[
Permission::SysAccountLockUpdate, Permission::SysAccountLockUpdate,
Permission::SysAccountLockDestroy, Permission::SysAccountLockDestroy,
Permission::SysComplianceGet, Permission::SysComplianceGet,
Permission::SysDeliverabilityGet,
]; ];
#[derive(Clone, Copy, PartialEq, Eq)] #[derive(Clone, Copy, PartialEq, Eq)]
enum Audience { enum Audience {
Admin, Admin,
Tenant, Tenant,
Officer,
} }
fn granted_key(permission: Permission, audience: Audience) -> ValueClass { fn granted_key(permission: Permission, audience: Audience) -> ValueClass {
let mut key = b"Pg".to_vec(); let mut key = b"Pg".to_vec();
// Admin grants keep the key they were first recorded under // Admin grants keep the key they were first recorded under
match audience { if audience == Audience::Tenant {
Audience::Admin => {} key.extend_from_slice(b"tenant:");
Audience::Tenant => key.extend_from_slice(b"tenant:"),
Audience::Officer => key.extend_from_slice(b"officer:"),
} }
key.extend_from_slice(permission.as_str().as_bytes()); key.extend_from_slice(permission.as_str().as_bytes());
ValueClass::Any(AnyClass { ValueClass::Any(AnyClass {
@@ -114,8 +82,7 @@ fn granted_key(permission: Permission, audience: Audience) -> ValueClass {
pub(crate) async fn grant_new_admin_permissions(bp: &mut Bootstrap) -> trc::Result<()> { pub(crate) async fn grant_new_admin_permissions(bp: &mut Bootstrap) -> trc::Result<()> {
grant(bp, Audience::Admin, ADMIN_GRANTS).await?; grant(bp, Audience::Admin, ADMIN_GRANTS).await?;
grant(bp, Audience::Tenant, TENANT_GRANTS).await?; grant(bp, Audience::Tenant, TENANT_GRANTS).await
grant(bp, Audience::Officer, OFFICER_GRANTS).await
} }
async fn grant(bp: &mut Bootstrap, audience: Audience, grants: &[Permission]) -> trc::Result<()> { async fn grant(bp: &mut Bootstrap, audience: Audience, grants: &[Permission]) -> trc::Result<()> {
@@ -134,47 +101,39 @@ async fn grant(bp: &mut Bootstrap, audience: Audience, grants: &[Permission]) ->
if pending.is_empty() { if pending.is_empty() {
return Ok(()); return Ok(());
} }
// The officer role is the one the server made, if it has made it yet: a // An administrator's default roles include the plain User role, which
// new install makes it after this, with the permissions already in it // every user also holds; only roles that are the audience's alone get it
let admin_roles: Vec<Id> = if audience == Audience::Officer { let admin_roles: Vec<Id> = bp
super::compliance_roles::server_role(&bp.data_store) .registry
.await? .object::<Authentication>(Id::singleton())
.into_iter() .await?
.collect() .map(|auth| {
} else { let (own, shared) = match audience {
// An administrator's default roles include the plain User role, which Audience::Admin => (
// every user also holds; only roles that are the audience's alone get it auth.default_admin_role_ids.as_slice(),
bp.registry [
.object::<Authentication>(Id::singleton()) auth.default_user_role_ids.as_slice(),
.await? auth.default_group_role_ids.as_slice(),
.map(|auth| {
let (own, shared) = match audience {
Audience::Admin => (
auth.default_admin_role_ids.as_slice(),
[
auth.default_user_role_ids.as_slice(),
auth.default_group_role_ids.as_slice(),
auth.default_tenant_role_ids.as_slice(),
]
.concat(),
),
Audience::Tenant | Audience::Officer => (
auth.default_tenant_role_ids.as_slice(), auth.default_tenant_role_ids.as_slice(),
[ ]
auth.default_user_role_ids.as_slice(), .concat(),
auth.default_group_role_ids.as_slice(), ),
auth.default_admin_role_ids.as_slice(), Audience::Tenant => (
] auth.default_tenant_role_ids.as_slice(),
.concat(), [
), auth.default_user_role_ids.as_slice(),
}; auth.default_group_role_ids.as_slice(),
own.iter() auth.default_admin_role_ids.as_slice(),
.filter(|id| !shared.contains(id)) ]
.copied() .concat(),
.collect() ),
}) };
.unwrap_or_default() own.iter()
}; .filter(|id| !shared.contains(id))
.copied()
.collect()
})
.unwrap_or_default();
// Fetched by id: the registry's listing doesn't reach stored roles // Fetched by id: the registry's listing doesn't reach stored roles
for role_id in admin_roles { for role_id in admin_roles {
let Some(stored) = bp let Some(stored) = bp
+1 -1
View File
@@ -1,5 +1,5 @@
/* /*
* SPDX-FileCopyrightText: 2026 Coffey Labs LLC * SPDX-FileCopyrightText: 2026 Coffey Labs
* *
* SPDX-License-Identifier: AGPL-3.0-only * SPDX-License-Identifier: AGPL-3.0-only
*/ */
+1 -1
View File
@@ -1,6 +1,6 @@
/* /*
* SPDX-FileCopyrightText: 2020 Stalwart Labs LLC <hello@stalw.art> * SPDX-FileCopyrightText: 2020 Stalwart Labs LLC <hello@stalw.art>
* SPDX-FileCopyrightText: 2026 Coffey Labs LLC * SPDX-FileCopyrightText: 2026 Coffey Labs
* *
* SPDX-License-Identifier: AGPL-3.0-only OR LicenseRef-SEL * SPDX-License-Identifier: AGPL-3.0-only OR LicenseRef-SEL
* *
@@ -11,7 +11,7 @@ use quick_xml::Reader;
use quick_xml::XmlVersion; use quick_xml::XmlVersion;
use quick_xml::events::Event; use quick_xml::events::Event;
use registry::schema::{enums::ServiceProtocol, structs::Service}; use registry::schema::{enums::ServiceProtocol, structs::Service};
use std::{borrow::Cow, fmt::Write}; use std::fmt::Write;
use utils::map::vec_map::VecMap; use utils::map::vec_map::VecMap;
impl Server { impl Server {
@@ -20,78 +20,32 @@ impl Server {
body: Option<Vec<u8>>, body: Option<Vec<u8>>,
) -> trc::Result<Resource<Vec<u8>>> { ) -> trc::Result<Resource<Vec<u8>>> {
// Obtain parameters // Obtain parameters
let request = let emailaddress = parse_autodiscover_request(body.as_deref().unwrap_or_default())
parse_autodiscover_request(body.as_deref().unwrap_or_default()).map_err(|err| { .map_err(|err| {
trc::ResourceEvent::BadParameters trc::ResourceEvent::BadParameters
.into_err() .into_err()
.details("Failed to parse autodiscover request") .details("Failed to parse autodiscover request")
.ctx(trc::Key::Reason, err) .ctx(trc::Key::Reason, err)
})?; })?;
// inbuxa: legacy-protocols LP-7, LP-14a // inbuxa: legacy-protocols LP-7, LP-14a
let legacy_off = match request.email.rsplit_once('@') { let legacy_off = match emailaddress.rsplit_once('@') {
Some((_, domain)) => self.legacy_off_for(domain).await?, Some((_, domain)) => self.legacy_off_for(domain).await?,
None => self.legacy_off_for("").await?, None => self.legacy_off_for("").await?,
}; };
let response = match request.response_schema { Ok(Resource::new(
ResponseSchema::Outlook => build_autodiscover_response( "application/xml; charset=utf-8",
&request.email, build_autodiscover_response(
&emailaddress,
&self.core.network.server_name, &self.core.network.server_name,
&self.core.network.info.services, &self.core.network.info.services,
|protocol| legacy_off.service(protocol), |protocol| legacy_off.service(protocol),
) )
.into_bytes(), .into_bytes(),
ResponseSchema::Unsupported => PROVIDER_NOT_AVAILABLE_RESPONSE.as_bytes().to_vec(), ))
};
Ok(Resource::new("application/xml; charset=utf-8", response))
} }
} }
const OUTLOOK_RESPONSE_SCHEMA: &str =
"http://schemas.microsoft.com/exchange/autodiscover/outlook/responseschema/2006a";
const PROVIDER_NOT_AVAILABLE_RESPONSE: &str = concat!(
"<?xml version=\"1.0\" encoding=\"UTF-8\"?>\n",
"<Autodiscover xmlns=\"http://schemas.microsoft.com/exchange/autodiscover/responseschema/2006\">\n",
"\t<Response>\n",
"\t\t<Error>\n",
"\t\t\t<ErrorCode>601</ErrorCode>\n",
"\t\t\t<Message>Provider is not available</Message>\n",
"\t\t\t<DebugData />\n",
"\t\t</Error>\n",
"\t</Response>\n",
"</Autodiscover>\n",
);
#[derive(Debug, Clone, Copy, PartialEq, Eq)]
enum ResponseSchema {
Outlook,
Unsupported,
}
impl ResponseSchema {
fn parse(value: &str) -> Self {
if value.trim().eq_ignore_ascii_case(OUTLOOK_RESPONSE_SCHEMA) {
ResponseSchema::Outlook
} else {
ResponseSchema::Unsupported
}
}
}
#[derive(Debug, PartialEq, Eq)]
struct AutodiscoverRequest {
email: String,
response_schema: ResponseSchema,
}
#[derive(Clone, Copy)]
enum RequestField {
EmailAddress,
ResponseSchema,
}
fn build_autodiscover_response( fn build_autodiscover_response(
emailaddress: &str, emailaddress: &str,
default_host: &str, default_host: &str,
@@ -170,7 +124,7 @@ fn build_autodiscover_response(
config config
} }
fn parse_autodiscover_request(bytes: &[u8]) -> Result<AutodiscoverRequest, String> { fn parse_autodiscover_request(bytes: &[u8]) -> Result<String, String> {
if bytes.is_empty() { if bytes.is_empty() {
return Err("Empty request body".to_string()); return Err("Empty request body".to_string());
} }
@@ -178,9 +132,8 @@ fn parse_autodiscover_request(bytes: &[u8]) -> Result<AutodiscoverRequest, Strin
let mut reader = Reader::from_reader(bytes); let mut reader = Reader::from_reader(bytes);
reader.config_mut().trim_text(true); reader.config_mut().trim_text(true);
let mut buf = Vec::with_capacity(128); let mut buf = Vec::with_capacity(128);
let mut value_buf = Vec::with_capacity(128);
'outer: for tag_name in ["Autodiscover", "Request"] { 'outer: for tag_name in ["Autodiscover", "Request", "EMailAddress"] {
loop { loop {
match reader.read_event_into(&mut buf) { match reader.read_event_into(&mut buf) {
Ok(Event::Start(e)) => { Ok(Event::Start(e)) => {
@@ -190,6 +143,30 @@ fn parse_autodiscover_request(bytes: &[u8]) -> Result<AutodiscoverRequest, Strin
.eq_ignore_ascii_case(found_tag_name.as_ref()) .eq_ignore_ascii_case(found_tag_name.as_ref())
{ {
continue 'outer; continue 'outer;
} else if tag_name == "EMailAddress" {
// Skip unsupported tags under Request, such as AcceptableResponseSchema
let mut tag_count = 0;
loop {
match reader.read_event_into(&mut buf) {
Ok(Event::End(_)) => {
if tag_count == 0 {
break;
} else {
tag_count -= 1;
}
}
Ok(Event::Start(_)) => {
tag_count += 1;
}
Ok(Event::Eof) => {
return Err(format!(
"Expected value, found unexpected EOF at position {}.",
reader.buffer_position()
));
}
_ => (),
}
}
} else { } else {
return Err(format!( return Err(format!(
"Expected tag {}, found unexpected tag {} at position {}.", "Expected tag {}, found unexpected tag {} at position {}.",
@@ -218,170 +195,36 @@ fn parse_autodiscover_request(bytes: &[u8]) -> Result<AutodiscoverRequest, Strin
} }
} }
let mut email = None; if let Ok(Event::Text(text)) = reader.read_event_into(&mut buf)
let mut response_schema = ResponseSchema::Outlook; && let Ok(text) = text.xml_content(XmlVersion::Implicit1_0)
&& text.contains('@')
loop { {
match reader.read_event_into(&mut buf) { return Ok(text.trim().to_lowercase());
Ok(Event::Start(e)) => {
let local_name = e.local_name();
let field = hashify::tiny_map_ignore_case!(local_name.as_ref(),
b"EMailAddress" => RequestField::EmailAddress,
b"AcceptableResponseSchema" => RequestField::ResponseSchema,
);
let value = match reader.read_event_into(&mut value_buf) {
Ok(Event::End(_)) => None,
Ok(event) => {
let value = match event {
Event::Text(text) => text
.xml_content(XmlVersion::Implicit1_0)
.ok()
.map(Cow::into_owned),
_ => None,
};
reader
.read_to_end_into(e.name(), &mut value_buf)
.map_err(|err| {
format!("Error at position {}: {:?}", reader.buffer_position(), err)
})?;
value
}
Err(err) => {
return Err(format!(
"Error at position {}: {:?}",
reader.buffer_position(),
err
));
}
};
match (field, value) {
(Some(RequestField::EmailAddress), Some(value)) => {
email = Some(value);
}
(Some(RequestField::ResponseSchema), Some(value)) => {
response_schema = ResponseSchema::parse(&value);
}
_ => (),
}
}
Ok(Event::End(_) | Event::Eof) => break,
Ok(_) => (),
Err(e) => {
return Err(format!(
"Error at position {}: {:?}",
reader.buffer_position(),
e
));
}
}
} }
match email { Err(format!(
Some(email) if email.contains('@') => Ok(AutodiscoverRequest { "Expected email address, found unexpected value at position {}.",
email: email.trim().to_lowercase(), reader.buffer_position()
response_schema, ))
}),
_ => Err(format!(
"Expected email address, found unexpected value at position {}.",
reader.buffer_position()
)),
}
} }
#[cfg(test)] #[cfg(test)]
mod tests { mod tests {
use super::{AutodiscoverRequest, ResponseSchema, parse_autodiscover_request};
#[test] #[test]
fn parse_autodiscover() { fn parse_autodiscover() {
const OUTLOOK: &str = let r = r#"<?xml version="1.0" encoding="utf-8"?>
"http://schemas.microsoft.com/exchange/autodiscover/outlook/responseschema/2006a";
const MOBILESYNC: &str =
"http://schemas.microsoft.com/exchange/autodiscover/mobilesync/responseschema/2006";
for (request, expected) in [
(
format!(
r#"<?xml version="1.0" encoding="utf-8"?>
<Autodiscover xmlns="http://schemas.microsoft.com/exchange/autodiscover/outlook/requestschema/2006"> <Autodiscover xmlns="http://schemas.microsoft.com/exchange/autodiscover/outlook/requestschema/2006">
<Request> <Request>
<EMailAddress>Email@Example.com</EMailAddress>
<AcceptableResponseSchema>{OUTLOOK}</AcceptableResponseSchema>
</Request>
</Autodiscover>"#
),
ResponseSchema::Outlook,
),
(
format!(
r#"<Autodiscover xmlns="http://schemas.microsoft.com/exchange/autodiscover/outlook/requestschema/2006">
<Request>
<AcceptableResponseSchema>{OUTLOOK}</AcceptableResponseSchema>
<EMailAddress>email@example.com</EMailAddress> <EMailAddress>email@example.com</EMailAddress>
<AcceptableResponseSchema>http://schemas.microsoft.com/exchange/autodiscover/outlook/responseschema/2006a</AcceptableResponseSchema>
</Request> </Request>
</Autodiscover>"# </Autodiscover>"#;
),
ResponseSchema::Outlook,
),
(
r#"<Autodiscover>
<Request>
<EMailAddress>email@example.com</EMailAddress>
</Request>
</Autodiscover>"#
.to_string(),
ResponseSchema::Outlook,
),
(
format!(
r#"<?xml version="1.0" encoding="utf-8"?>
<Autodiscover xmlns="http://schemas.microsoft.com/exchange/autodiscover/mobilesync/requestschema/2006">
<Request>
<EMailAddress>email@example.com</EMailAddress>
<AcceptableResponseSchema>{MOBILESYNC}</AcceptableResponseSchema>
</Request>
</Autodiscover>"#
),
ResponseSchema::Unsupported,
),
(
format!(
r#"<Autodiscover>
<Request>
<LegacyDN>/o=Example/ou=Users/cn=email</LegacyDN>
<Unknown><Nested>value</Nested><Empty/></Unknown>
<AcceptableResponseSchema>{MOBILESYNC}</AcceptableResponseSchema>
<EMailAddress>email@example.com</EMailAddress>
</Request>
</Autodiscover>"#
),
ResponseSchema::Unsupported,
),
] {
assert_eq!(
parse_autodiscover_request(request.as_bytes()).expect("valid request"),
AutodiscoverRequest {
email: "[email protected]".to_string(),
response_schema: expected,
},
"{request}"
);
}
for request in [ assert_eq!(
"", super::parse_autodiscover_request(r.as_bytes()).unwrap(),
"<Autodiscover><Request></Request></Autodiscover>", "[email protected]"
"<Autodiscover><Request><EMailAddress>no-domain</EMailAddress></Request></Autodiscover>", );
"<Autodiscover><Request><EMailAddress>[email protected]</Request></Autodiscover>",
"<Request><EMailAddress>[email protected]</EMailAddress></Request>",
] {
assert!(
parse_autodiscover_request(request.as_bytes()).is_err(),
"{request}"
);
}
} }
#[test] #[test]
+1 -1
View File
@@ -1,5 +1,5 @@
/* /*
* SPDX-FileCopyrightText: 2026 Coffey Labs LLC * SPDX-FileCopyrightText: 2026 Coffey Labs
* *
* SPDX-License-Identifier: AGPL-3.0-only * SPDX-License-Identifier: AGPL-3.0-only
*/ */
+1 -1
View File
@@ -1,5 +1,5 @@
/* /*
* SPDX-FileCopyrightText: 2026 Coffey Labs LLC * SPDX-FileCopyrightText: 2026 Coffey Labs
* *
* SPDX-License-Identifier: AGPL-3.0-only * SPDX-License-Identifier: AGPL-3.0-only
*/ */
+1 -1
View File
@@ -1,5 +1,5 @@
/* /*
* SPDX-FileCopyrightText: 2026 Coffey Labs LLC * SPDX-FileCopyrightText: 2026 Coffey Labs
* *
* SPDX-License-Identifier: AGPL-3.0-only * SPDX-License-Identifier: AGPL-3.0-only
*/ */
+1 -1
View File
@@ -1,5 +1,5 @@
/* /*
* SPDX-FileCopyrightText: 2026 Coffey Labs LLC * SPDX-FileCopyrightText: 2026 Coffey Labs
* *
* SPDX-License-Identifier: AGPL-3.0-only * SPDX-License-Identifier: AGPL-3.0-only
*/ */
+1 -1
View File
@@ -1,5 +1,5 @@
/* /*
* SPDX-FileCopyrightText: 2026 Coffey Labs LLC * SPDX-FileCopyrightText: 2026 Coffey Labs
* *
* SPDX-License-Identifier: AGPL-3.0-only * SPDX-License-Identifier: AGPL-3.0-only
*/ */
+1 -1
View File
@@ -1,5 +1,5 @@
/* /*
* SPDX-FileCopyrightText: 2026 Coffey Labs LLC * SPDX-FileCopyrightText: 2026 Coffey Labs
* *
* SPDX-License-Identifier: AGPL-3.0-only * SPDX-License-Identifier: AGPL-3.0-only
*/ */
+1 -1
View File
@@ -1,5 +1,5 @@
/* /*
* SPDX-FileCopyrightText: 2026 Coffey Labs LLC * SPDX-FileCopyrightText: 2026 Coffey Labs
* *
* SPDX-License-Identifier: AGPL-3.0-only * SPDX-License-Identifier: AGPL-3.0-only
*/ */
+4 -62
View File
@@ -1,5 +1,5 @@
/* /*
* SPDX-FileCopyrightText: 2026 Coffey Labs LLC * SPDX-FileCopyrightText: 2026 Coffey Labs
* *
* SPDX-License-Identifier: AGPL-3.0-only * SPDX-License-Identifier: AGPL-3.0-only
*/ */
@@ -104,31 +104,14 @@ impl StoredMetric {
pub fn timestamp(&self) -> u64 { pub fn timestamp(&self) -> u64 {
SnowflakeIdGenerator::to_timestamp(self.id) SnowflakeIdGenerator::to_timestamp(self.id)
} }
/// The node that wrote the sample. Histogram totals are per node, so a
/// reader diffs them per node.
pub fn node_id(&self) -> u64 {
SnowflakeIdGenerator::to_node_id(self.id)
}
} }
/// What the node wrote last, so counters and histograms are written as /// What the node wrote last, so counters and histograms are written as
/// changes (MON-4). Per process: a restart counts from the start. /// changes (MON-4). Per process: a restart counts from the start.
static LAST: Mutex<Option<AHashMap<MetricType, (u64, u64)>>> = Mutex::new(None); static LAST: Mutex<Option<AHashMap<MetricType, (u64, u64)>>> = Mutex::new(None);
/// Gauges that count the whole cluster's data, not this node's. Only the node /// One tick's samples (MON-4 to MON-6).
/// that computes them (the metrics-calculation role) has a true reading; on pub fn sample() -> Vec<Metric> {
/// the others the queue gauge only moves with local queue events and drifts
/// below zero, and the account and domain counts stay at 0.
const CLUSTER_GAUGES: [MetricType; 3] = [
MetricType::QueueCount,
MetricType::UserCount,
MetricType::DomainCount,
];
/// One tick's samples (MON-4 to MON-6). `calculates` is whether this node
/// computes the cluster-wide gauges; a node that doesn't leaves them out.
pub fn sample(calculates: bool) -> Vec<Metric> {
let mut last_guard = LAST.lock().unwrap(); let mut last_guard = LAST.lock().unwrap();
let last = last_guard.get_or_insert_with(AHashMap::new); let last = last_guard.get_or_insert_with(AHashMap::new);
let mut samples = Vec::new(); let mut samples = Vec::new();
@@ -151,9 +134,6 @@ pub fn sample(calculates: bool) -> Vec<Metric> {
// Gauges: the reading, always (MON-5) // Gauges: the reading, always (MON-5)
for gauge in Collector::collect_gauges() { for gauge in Collector::collect_gauges() {
if !calculates && CLUSTER_GAUGES.contains(&gauge.id()) {
continue;
}
samples.push(Metric::Gauge(MetricCount { samples.push(Metric::Gauge(MetricCount {
count: gauge.get(), count: gauge.get(),
metric: gauge.id(), metric: gauge.id(),
@@ -195,7 +175,7 @@ impl Server {
if store.is_none() { if store.is_none() {
return; return;
} }
let samples = sample(self.core.network.roles.metrics_calculate); let samples = sample();
let count = samples.len(); let count = samples.len();
let started = std::time::Instant::now(); let started = std::time::Instant::now();
match store.write_metrics(samples, now()).await { match store.write_metrics(samples, now()).await {
@@ -285,41 +265,3 @@ impl Server {
} }
} }
} }
#[cfg(test)]
mod tests {
use super::*;
fn gauges(samples: &[Metric]) -> Vec<MetricType> {
samples
.iter()
.filter_map(|m| match m {
Metric::Gauge(g) => Some(g.metric),
_ => None,
})
.collect()
}
#[test]
fn only_the_calculating_node_stores_cluster_gauges() {
let all = gauges(&sample(true));
let local = gauges(&sample(false));
for metric in CLUSTER_GAUGES {
assert!(
all.contains(&metric),
"{metric:?} missing on the calculating node"
);
assert!(
!local.contains(&metric),
"{metric:?} stored by a node that doesn't compute it"
);
}
// Per-node gauges are stored either way
for metric in [MetricType::ServerMemory, MetricType::HttpActiveConnections] {
assert!(
all.contains(&metric) && local.contains(&metric),
"{metric:?}"
);
}
}
}
+1 -1
View File
@@ -1,5 +1,5 @@
/* /*
* SPDX-FileCopyrightText: 2026 Coffey Labs LLC * SPDX-FileCopyrightText: 2026 Coffey Labs
* *
* SPDX-License-Identifier: AGPL-3.0-only * SPDX-License-Identifier: AGPL-3.0-only
*/ */
+1 -1
View File
@@ -1,6 +1,6 @@
[package] [package]
name = "coordinator" name = "coordinator"
version = "0.16.25" version = "0.16.24"
edition = "2024" edition = "2024"
[dependencies] [dependencies]
+1 -1
View File
@@ -1,6 +1,6 @@
[package] [package]
name = "dav-proto" name = "dav-proto"
version = "0.16.25" version = "0.16.24"
edition = "2024" edition = "2024"
[dependencies] [dependencies]
+1 -1
View File
@@ -1,6 +1,6 @@
[package] [package]
name = "dav" name = "dav"
version = "0.16.25" version = "0.16.24"
edition = "2024" edition = "2024"
[dependencies] [dependencies]
+1 -11
View File
@@ -133,10 +133,6 @@ impl DavAclHandler for Server {
{ {
return Err(DavError::Code(StatusCode::FORBIDDEN)); return Err(DavError::Code(StatusCode::FORBIDDEN));
} }
// inbuxa: MA-D0: a group's members don't share what it owns on.
if access_token.is_group_member_only(account_id) {
return Err(DavError::Code(StatusCode::FORBIDDEN));
}
// Validate ACEs // Validate ACEs
let grants = self let grants = self
@@ -569,13 +565,7 @@ impl Privileges for AccessToken {
grants: &ArchivedVec<ArchivedAclGrant>, grants: &ArchivedVec<ArchivedAclGrant>,
is_calendar: bool, is_calendar: bool,
) -> Vec<Privilege> { ) -> Vec<Privilege> {
if self.is_group_member_only(account_id) { if self.is_member(account_id) {
// inbuxa: MA-D0: everything but sharing it on.
Privilege::all(is_calendar)
.into_iter()
.filter(|privilege| !matches!(privilege, Privilege::All | Privilege::WriteAcl))
.collect()
} else if self.is_member(account_id) {
Privilege::all(is_calendar) Privilege::all(is_calendar)
} else { } else {
current_user_privilege_set(grants.effective_acl(self)) current_user_privilege_set(grants.effective_acl(self))
+1 -1
View File
@@ -1,6 +1,6 @@
[package] [package]
name = "directory" name = "directory"
version = "0.16.25" version = "0.16.24"
edition = "2024" edition = "2024"
[dependencies] [dependencies]
+1 -1
View File
@@ -1,6 +1,6 @@
[package] [package]
name = "email" name = "email"
version = "0.16.25" version = "0.16.24"
edition = "2024" edition = "2024"
[dependencies] [dependencies]
+1 -1
View File
@@ -1,5 +1,5 @@
/* /*
* SPDX-FileCopyrightText: 2026 Coffey Labs LLC * SPDX-FileCopyrightText: 2026 Coffey Labs
* *
* SPDX-License-Identifier: AGPL-3.0-only * SPDX-License-Identifier: AGPL-3.0-only
*/ */
+6 -2
View File
@@ -20,7 +20,7 @@ use groupware::{
scheduling::{ItipError, ItipMessages}, scheduling::{ItipError, ItipMessages},
}; };
use mail_parser::{ use mail_parser::{
Header, HeaderName, HeaderValue, Message, MessageParser, MimeHeaders, PartType, DateTime, Header, HeaderName, HeaderValue, Message, MessageParser, MimeHeaders, PartType,
parsers::fields::thread::thread_name, parsers::fields::thread::thread_name,
}; };
use registry::{ use registry::{
@@ -924,7 +924,11 @@ impl EmailIngest for Server {
span_id: u64, span_id: u64,
) { ) {
if let Some(config) = &self.core.spam.classifier { if let Some(config) = &self.core.spam.classifier {
let until = now() + config.hold_samples_for; let mut dt = DateTime::from_timestamp(now() as i64);
dt.hour = 0;
dt.minute = 0;
dt.second = 0;
let until = dt.to_timestamp() as u64 + config.hold_samples_for;
let sample = SpamTrainingSample { let sample = SpamTrainingSample {
account_id: Some(Id::from(account_id)), account_id: Some(Id::from(account_id)),
+2 -10
View File
@@ -290,11 +290,7 @@ impl SieveScriptIngest for Server {
// inbuxa: AL-4: a locked account answers no sender, so a // inbuxa: AL-4: a locked account answers no sender, so a
// rejection is kept instead; sieve has already cleared // rejection is kept instead; sieve has already cleared
// the implicit keep, so it is filed here // the implicit keep, so it is filed here
// A shared mailbox (MA-S) is a role address and answers Event::Reject { .. } if access_token.is_locked() => {
// as one: its Sieve script runs as written
Event::Reject { .. }
if access_token.is_locked() && !access_token.is_shared_mailbox() =>
{
if let Some(message) = messages.get_mut(0) if let Some(message) = messages.get_mut(0)
&& !message.file_into.contains(&INBOX_ID) && !message.file_into.contains(&INBOX_ID)
{ {
@@ -407,11 +403,7 @@ impl SieveScriptIngest for Server {
// inbuxa: AL-4: a locked account sends nothing on its // inbuxa: AL-4: a locked account sends nothing on its
// own: no redirect, vacation reply or notification. An // own: no redirect, vacation reply or notification. An
// unsent redirect leaves the message to be kept. // unsent redirect leaves the message to be kept.
// A shared mailbox's acknowledgements and redirects go Event::SendMessage { .. } if access_token.is_locked() => {
// out (MA-S).
Event::SendMessage { .. }
if access_token.is_locked() && !access_token.is_shared_mailbox() =>
{
trc::event!( trc::event!(
Sieve(SieveEvent::ActionReject), Sieve(SieveEvent::ActionReject),
Details = "Account is locked: nothing is sent", Details = "Account is locked: nothing is sent",
-7
View File
@@ -21,13 +21,6 @@ base64 = "0.23"
sha2 = "0.11" sha2 = "0.11"
flate2 = "1.1" flate2 = "1.1"
tokio = { version = "1.53", features = ["sync", "rt"] } tokio = { version = "1.53", features = ["sync", "rt"] }
# inbuxa: DLP detectors and attachment text (dlp-and-mail-flow-rules spec)
regex = "1.13.1"
aho-corasick = "1.1"
zip = "8.6"
quick-xml = "0.41"
mail-parser = { version = "0.11", features = ["full_encoding"] }
mail-builder = { version = "1.0" }
[dev-dependencies] [dev-dependencies]
tokio = { version = "1.53", features = ["macros", "rt"] } tokio = { version = "1.53", features = ["macros", "rt"] }
+1 -1
View File
@@ -1,5 +1,5 @@
/* /*
* SPDX-FileCopyrightText: 2026 Coffey Labs LLC * SPDX-FileCopyrightText: 2026 Coffey Labs
* *
* SPDX-License-Identifier: AGPL-3.0-only * SPDX-License-Identifier: AGPL-3.0-only
*/ */
+1 -1
View File
@@ -1,5 +1,5 @@
/* /*
* SPDX-FileCopyrightText: 2026 Coffey Labs LLC * SPDX-FileCopyrightText: 2026 Coffey Labs
* *
* SPDX-License-Identifier: AGPL-3.0-only * SPDX-License-Identifier: AGPL-3.0-only
*/ */
+1 -1
View File
@@ -1,5 +1,5 @@
/* /*
* SPDX-FileCopyrightText: 2026 Coffey Labs LLC * SPDX-FileCopyrightText: 2026 Coffey Labs
* *
* SPDX-License-Identifier: AGPL-3.0-only * SPDX-License-Identifier: AGPL-3.0-only
*/ */
+1 -1
View File
@@ -1,5 +1,5 @@
/* /*
* SPDX-FileCopyrightText: 2026 Coffey Labs LLC * SPDX-FileCopyrightText: 2026 Coffey Labs
* *
* SPDX-License-Identifier: AGPL-3.0-only * SPDX-License-Identifier: AGPL-3.0-only
*/ */
+1 -1
View File
@@ -1,5 +1,5 @@
/* /*
* SPDX-FileCopyrightText: 2026 Coffey Labs LLC * SPDX-FileCopyrightText: 2026 Coffey Labs
* *
* SPDX-License-Identifier: AGPL-3.0-only * SPDX-License-Identifier: AGPL-3.0-only
*/ */
+1 -1
View File
@@ -1,5 +1,5 @@
/* /*
* SPDX-FileCopyrightText: 2026 Coffey Labs LLC * SPDX-FileCopyrightText: 2026 Coffey Labs
* *
* SPDX-License-Identifier: AGPL-3.0-only * SPDX-License-Identifier: AGPL-3.0-only
*/ */
+1 -1
View File
@@ -1,5 +1,5 @@
/* /*
* SPDX-FileCopyrightText: 2026 Coffey Labs LLC * SPDX-FileCopyrightText: 2026 Coffey Labs
* *
* SPDX-License-Identifier: AGPL-3.0-only * SPDX-License-Identifier: AGPL-3.0-only
*/ */
+1 -1
View File
@@ -1,5 +1,5 @@
/* /*
* SPDX-FileCopyrightText: 2026 Coffey Labs LLC * SPDX-FileCopyrightText: 2026 Coffey Labs
* *
* SPDX-License-Identifier: AGPL-3.0-only * SPDX-License-Identifier: AGPL-3.0-only
*/ */
+1 -1
View File
@@ -1,5 +1,5 @@
/* /*
* SPDX-FileCopyrightText: 2026 Coffey Labs LLC * SPDX-FileCopyrightText: 2026 Coffey Labs
* *
* SPDX-License-Identifier: AGPL-3.0-only * SPDX-License-Identifier: AGPL-3.0-only
*/ */
+1 -1
View File
@@ -1,5 +1,5 @@
/* /*
* SPDX-FileCopyrightText: 2026 Coffey Labs LLC * SPDX-FileCopyrightText: 2026 Coffey Labs
* *
* SPDX-License-Identifier: AGPL-3.0-only * SPDX-License-Identifier: AGPL-3.0-only
*/ */
+1 -1
View File
@@ -1,5 +1,5 @@
/* /*
* SPDX-FileCopyrightText: 2026 Coffey Labs LLC * SPDX-FileCopyrightText: 2026 Coffey Labs
* *
* SPDX-License-Identifier: AGPL-3.0-only * SPDX-License-Identifier: AGPL-3.0-only
*/ */
+1 -1
View File
@@ -1,5 +1,5 @@
/* /*
* SPDX-FileCopyrightText: 2026 Coffey Labs LLC * SPDX-FileCopyrightText: 2026 Coffey Labs
* *
* SPDX-License-Identifier: AGPL-3.0-only * SPDX-License-Identifier: AGPL-3.0-only
*/ */
+1 -1
View File
@@ -1,5 +1,5 @@
/* /*
* SPDX-FileCopyrightText: 2026 Coffey Labs LLC * SPDX-FileCopyrightText: 2026 Coffey Labs
* *
* SPDX-License-Identifier: AGPL-3.0-only * SPDX-License-Identifier: AGPL-3.0-only
*/ */
+1 -1
View File
@@ -1,5 +1,5 @@
/* /*
* SPDX-FileCopyrightText: 2026 Coffey Labs LLC * SPDX-FileCopyrightText: 2026 Coffey Labs
* *
* SPDX-License-Identifier: AGPL-3.0-only * SPDX-License-Identifier: AGPL-3.0-only
*/ */
+1 -1
View File
@@ -1,5 +1,5 @@
/* /*
* SPDX-FileCopyrightText: 2026 Coffey Labs LLC * SPDX-FileCopyrightText: 2026 Coffey Labs
* *
* SPDX-License-Identifier: AGPL-3.0-only * SPDX-License-Identifier: AGPL-3.0-only
*/ */
+1 -1
View File
@@ -1,5 +1,5 @@
/* /*
* SPDX-FileCopyrightText: 2026 Coffey Labs LLC * SPDX-FileCopyrightText: 2026 Coffey Labs
* *
* SPDX-License-Identifier: AGPL-3.0-only * SPDX-License-Identifier: AGPL-3.0-only
*/ */
+1 -1
View File
@@ -1,5 +1,5 @@
/* /*
* SPDX-FileCopyrightText: 2026 Coffey Labs LLC * SPDX-FileCopyrightText: 2026 Coffey Labs
* *
* SPDX-License-Identifier: AGPL-3.0-only * SPDX-License-Identifier: AGPL-3.0-only
*/ */
+1 -1
View File
@@ -1,5 +1,5 @@
/* /*
* SPDX-FileCopyrightText: 2026 Coffey Labs LLC * SPDX-FileCopyrightText: 2026 Coffey Labs
* *
* SPDX-License-Identifier: AGPL-3.0-only * SPDX-License-Identifier: AGPL-3.0-only
*/ */
+1 -1
View File
@@ -1,5 +1,5 @@
/* /*
* SPDX-FileCopyrightText: 2026 Coffey Labs LLC * SPDX-FileCopyrightText: 2026 Coffey Labs
* *
* SPDX-License-Identifier: AGPL-3.0-only * SPDX-License-Identifier: AGPL-3.0-only
*/ */
+1 -1
View File
@@ -1,5 +1,5 @@
/* /*
* SPDX-FileCopyrightText: 2026 Coffey Labs LLC * SPDX-FileCopyrightText: 2026 Coffey Labs
* *
* SPDX-License-Identifier: AGPL-3.0-only * SPDX-License-Identifier: AGPL-3.0-only
*/ */
+1 -1
View File
@@ -1,5 +1,5 @@
/* /*
* SPDX-FileCopyrightText: 2026 Coffey Labs LLC * SPDX-FileCopyrightText: 2026 Coffey Labs
* *
* SPDX-License-Identifier: AGPL-3.0-only * SPDX-License-Identifier: AGPL-3.0-only
*/ */
-282
View File
@@ -1,282 +0,0 @@
/*
* SPDX-FileCopyrightText: 2026 Coffey Labs LLC
*
* SPDX-License-Identifier: AGPL-3.0-only
*/
//! The blocklists a node asks about itself (deliverability spec, DL-6), and
//! how to read each one's answer.
//!
//! A list answers with an address in 127.0.0.0/8. Each list says which of
//! those mean "listed" and which mean "I won't answer you": Spamhaus, for
//! one, answers `127.255.255.254` to a query that came through a public
//! resolver. A refusal is never read as a listing (DL-4).
use std::net::{IpAddr, Ipv4Addr};
#[derive(Debug, Clone, Copy, PartialEq, Eq)]
pub enum Scope {
/// Looked up by the reversed address: `2.0.0.127.zen.spamhaus.org`.
Ip,
/// Looked up by name: `example.org.dbl.spamhaus.org`.
Domain,
}
#[derive(Debug, Clone, Copy)]
pub struct BlockList {
/// What the page and the settings call it.
pub name: &'static str,
pub zone: &'static str,
pub scope: Scope,
/// Where an administrator looks the address up and asks for removal.
pub lookup: &'static str,
/// Something the page says beside the list.
pub note: Option<&'static str>,
read: fn(Ipv4Addr) -> Answer,
}
/// What a list's answer means.
#[derive(Debug, Clone, PartialEq, Eq)]
pub enum Answer {
Listed(&'static str),
/// The list won't answer this resolver, or not now.
Refused(&'static str),
/// A code the list doesn't define: neither listed nor clean.
Unknown,
}
impl BlockList {
pub fn read(&self, answer: Ipv4Addr) -> Answer {
(self.read)(answer)
}
/// The name to look up for `subject`, or None when the subject doesn't
/// suit the list (a domain on an IP list, or an IPv6 address: none of
/// these lists publish IPv6 zones worth asking).
pub fn query(&self, subject: &Subject<'_>) -> Option<String> {
match (self.scope, subject) {
(Scope::Ip, Subject::Ip(IpAddr::V4(ip))) => {
let [a, b, c, d] = ip.octets();
Some(format!("{d}.{c}.{b}.{a}.{}.", self.zone))
}
(Scope::Domain, Subject::Domain(domain)) => {
Some(format!("{}.{}.", domain.trim_end_matches('.'), self.zone))
}
_ => None,
}
}
}
pub enum Subject<'x> {
Ip(IpAddr),
Domain(&'x str),
}
/// Spamhaus' error codes, the same on every Spamhaus zone.
fn spamhaus_refusal(ip: Ipv4Addr) -> Option<Answer> {
match ip.octets() {
[127, 255, 255, 252] => Some(Answer::Refused("The query was malformed")),
[127, 255, 255, 254] => Some(Answer::Refused(
"Spamhaus doesn't answer public resolvers; use the server's own",
)),
[127, 255, 255, 255] => Some(Answer::Refused("Too many queries from this resolver")),
_ => None,
}
}
fn zen(ip: Ipv4Addr) -> Answer {
if let Some(refused) = spamhaus_refusal(ip) {
return refused;
}
match ip.octets() {
[127, 0, 0, 2] => Answer::Listed("SBL: a known spam source"),
[127, 0, 0, 3] => Answer::Listed("CSS: sent spam recently"),
[127, 0, 0, 4..=7] => Answer::Listed("XBL: a compromised or infected host"),
[127, 0, 0, 9] => Answer::Listed("DROP: a hijacked or criminal network"),
[127, 0, 0, 10 | 11] => {
Answer::Listed("PBL: an address that isn't meant to send mail directly")
}
_ => Answer::Unknown,
}
}
fn dbl(ip: Ipv4Addr) -> Answer {
if let Some(refused) = spamhaus_refusal(ip) {
return refused;
}
match ip.octets() {
[127, 0, 1, 2] => Answer::Listed("A spam domain"),
[127, 0, 1, 4] => Answer::Listed("A phishing domain"),
[127, 0, 1, 5] => Answer::Listed("A malware domain"),
[127, 0, 1, 6] => Answer::Listed("A botnet controller"),
[127, 0, 1, 102..=106] => Answer::Listed("A legitimate domain being abused"),
[127, 0, 1, 255] => Answer::Refused("The query was malformed"),
_ => Answer::Unknown,
}
}
/// Most lists answer 127.0.0.2 for "listed" and define nothing else.
fn just_two(ip: Ipv4Addr) -> Answer {
match ip.octets() {
[127, 0, 0, 2] => Answer::Listed("Listed"),
_ => Answer::Unknown,
}
}
fn surbl(ip: Ipv4Addr) -> Answer {
match ip.octets() {
[127, 0, 0, 1] => Answer::Refused("SURBL doesn't answer this resolver"),
[127, 0, 0, bits] if bits & (8 | 16 | 64 | 128) != 0 => {
Answer::Listed("Seen in phishing, malware, abuse or cracked sites")
}
_ => Answer::Unknown,
}
}
fn uribl(ip: Ipv4Addr) -> Answer {
match ip.octets() {
[127, 0, 0, 1] => Answer::Refused("URIBL doesn't answer public resolvers"),
[127, 0, 0, bits] if bits & (2 | 8) != 0 => Answer::Listed("Seen in spam"),
[127, 0, 0, bits] if bits & 4 != 0 => {
Answer::Listed("Grey: seen in bulk mail some people don't want")
}
_ => Answer::Unknown,
}
}
pub const LISTS: &[BlockList] = &[
BlockList {
name: "Spamhaus ZEN",
zone: "zen.spamhaus.org",
scope: Scope::Ip,
lookup: "https://check.spamhaus.org/",
note: None,
read: zen,
},
BlockList {
name: "SpamCop",
zone: "bl.spamcop.net",
scope: Scope::Ip,
lookup: "https://www.spamcop.net/bl.shtml",
note: None,
read: just_two,
},
BlockList {
name: "Barracuda",
zone: "b.barracudacentral.org",
scope: Scope::Ip,
lookup: "https://www.barracudacentral.org/lookups",
note: Some(
"Barracuda answers only resolvers whose address is registered with it (free, at barracudacentral.org/rbl). Until then its lookups can't be checked.",
),
read: just_two,
},
BlockList {
name: "UCEPROTECT level 1",
zone: "dnsbl-1.uceprotect.net",
scope: Scope::Ip,
lookup: "https://www.uceprotect.net/en/rblcheck.php",
note: None,
read: just_two,
},
BlockList {
name: "Mailspike",
zone: "bl.mailspike.net",
scope: Scope::Ip,
lookup: "https://mailspike.org/iplookup.html",
note: None,
read: just_two,
},
BlockList {
name: "PSBL",
zone: "psbl.surriel.com",
scope: Scope::Ip,
lookup: "https://psbl.org/",
note: None,
read: just_two,
},
BlockList {
name: "Spamhaus DBL",
zone: "dbl.spamhaus.org",
scope: Scope::Domain,
lookup: "https://check.spamhaus.org/",
note: None,
read: dbl,
},
BlockList {
name: "SURBL",
zone: "multi.surbl.org",
scope: Scope::Domain,
lookup: "https://surbl.org/surbl-analysis",
note: None,
read: surbl,
},
BlockList {
name: "URIBL",
zone: "multi.uribl.com",
scope: Scope::Domain,
lookup: "https://admin.uribl.com/",
note: None,
read: uribl,
},
];
pub fn by_name(name: &str) -> Option<&'static BlockList> {
LISTS.iter().find(|list| list.name == name)
}
#[cfg(test)]
mod tests {
use super::*;
fn ip(s: &str) -> Ipv4Addr {
s.parse().unwrap()
}
#[test]
fn a_refusal_is_not_a_listing() {
let zen = by_name("Spamhaus ZEN").unwrap();
assert!(matches!(
zen.read(ip("127.255.255.254")),
Answer::Refused(_)
));
assert!(matches!(zen.read(ip("127.0.0.2")), Answer::Listed(_)));
assert!(matches!(zen.read(ip("127.0.0.10")), Answer::Listed(_)));
assert_eq!(zen.read(ip("127.0.0.200")), Answer::Unknown);
let uribl = by_name("URIBL").unwrap();
assert!(matches!(uribl.read(ip("127.0.0.1")), Answer::Refused(_)));
assert!(matches!(uribl.read(ip("127.0.0.2")), Answer::Listed(_)));
}
#[test]
fn queries_are_built_per_scope() {
let zen = by_name("Spamhaus ZEN").unwrap();
let dbl = by_name("Spamhaus DBL").unwrap();
let v4 = Subject::Ip("192.0.2.10".parse().unwrap());
let v6 = Subject::Ip("2001:db8::1".parse().unwrap());
let domain = Subject::Domain("example.org");
assert_eq!(
zen.query(&v4).as_deref(),
Some("10.2.0.192.zen.spamhaus.org.")
);
assert_eq!(zen.query(&v6), None);
assert_eq!(zen.query(&domain), None);
assert_eq!(
dbl.query(&domain).as_deref(),
Some("example.org.dbl.spamhaus.org.")
);
assert_eq!(dbl.query(&v4), None);
}
#[test]
fn names_are_unique() {
for (i, a) in LISTS.iter().enumerate() {
assert!(
LISTS[i + 1..].iter().all(|b| b.name != a.name),
"{}",
a.name
);
}
}
}
-410
View File
@@ -1,410 +0,0 @@
/*
* SPDX-FileCopyrightText: 2026 Coffey Labs LLC
*
* SPDX-License-Identifier: AGPL-3.0-only
*/
//! The deliverability check (deliverability spec): what other mail servers
//! see when this one sends. Not a rebuild of anything upstream ships.
//!
//! Every node that sends mail checks itself, because only it knows which
//! address it leaves from, and keeps one report. The report holds facts: an
//! address's reverse DNS, what each blocklist answered, what SPF said for
//! each address, whether a DKIM key in DNS matches the one signing. The
//! console grades them, so its wording can change without a server release.
//!
//! Kept in the fork's subspace (`store::SUBSPACE_INBUXA`). Every key starts
//! with `D`, then one byte for the kind:
//!
//! - `r` + node id (u64): that node's last report, as JSON.
//! - `s`: the settings, as JSON.
//!
//! Numbers are big-endian.
pub mod lists;
use serde::{Deserialize as SerdeDeserialize, Serialize as SerdeSerialize};
use store::{
Deserialize, IterateParams, SUBSPACE_INBUXA, Serialize, Store, ValueKey,
write::{AnyClass, BatchBuilder, ValueClass},
};
use trc::AddContext;
const FEATURE: u8 = b'D';
const KIND_REPORT: u8 = b'r';
const KIND_SETTINGS: u8 = b's';
/// DL-15: **Check now** runs a node again only this long after its last run.
pub const MIN_INTERVAL_SECS: u64 = 600;
#[derive(Debug, Clone, Default, PartialEq, SerdeSerialize, SerdeDeserialize)]
#[serde(rename_all = "camelCase", default)]
pub struct Report {
/// The node's cluster id, as metric samples carry it.
pub node_id: u64,
pub hostname: String,
/// Seconds since the epoch.
pub checked_at: u64,
pub addresses: Vec<Address>,
pub domains: Vec<DomainReport>,
pub certificates: Vec<Certificate>,
}
#[derive(Debug, Clone, Default, PartialEq, SerdeSerialize, SerdeDeserialize)]
#[serde(rename_all = "camelCase", default)]
pub struct Address {
pub ip: String,
/// DL-2: how the node came by the address.
pub source: AddressSource,
/// The connection strategy that sends from it.
pub strategy: String,
/// The name the node greets with from this address.
pub ehlo: String,
/// The PTR names, empty when there's none.
pub ptr: Vec<String>,
/// Some PTR name resolves back to the address.
pub forward_confirmed: bool,
/// The forward-confirmed name is the EHLO name.
pub ehlo_matches: bool,
/// Set when the reverse lookup itself failed, rather than found nothing.
pub ptr_error: Option<String>,
pub listings: Vec<Listing>,
}
#[derive(Debug, Clone, Copy, Default, PartialEq, Eq, SerdeSerialize, SerdeDeserialize)]
#[serde(rename_all = "camelCase")]
pub enum AddressSource {
/// Set in the connection strategy's source addresses.
#[default]
Configured,
/// What the EHLO name resolves to.
Ehlo,
}
#[derive(Debug, Clone, Default, PartialEq, SerdeSerialize, SerdeDeserialize)]
#[serde(rename_all = "camelCase", default)]
pub struct Listing {
/// The list's name, as in [`lists::LISTS`].
pub list: String,
pub state: ListingState,
/// The address the list answered, when it answered one.
pub code: Option<String>,
/// What the list says the answer means.
pub meaning: Option<String>,
}
#[derive(Debug, Clone, Copy, Default, PartialEq, Eq, SerdeSerialize, SerdeDeserialize)]
#[serde(rename_all = "camelCase")]
pub enum ListingState {
#[default]
Clean,
Listed,
/// The list wouldn't answer, or the lookup failed: neither listed nor clean.
Refused,
Error,
/// Switched off in the settings, so not asked.
Off,
}
#[derive(Debug, Clone, Default, PartialEq, SerdeSerialize, SerdeDeserialize)]
#[serde(rename_all = "camelCase", default)]
pub struct DomainReport {
pub domain: String,
/// DL-20: a tenant administrator sees only their tenant's domains.
pub tenant_id: Option<u32>,
/// DL-7: what SPF says for each of the node's addresses.
pub spf: Vec<SpfResult>,
/// DL-8: each DKIM key the domain signs with.
pub dkim: Vec<DkimKey>,
/// DL-9: the DMARC record, if there's one.
pub dmarc: Option<Dmarc>,
/// DL-10.
pub mta_sts: MtaSts,
/// DL-11: there's a `_smtp._tls` record.
pub tls_rpt: bool,
/// DL-12.
pub listings: Vec<Listing>,
}
#[derive(Debug, Clone, Default, PartialEq, SerdeSerialize, SerdeDeserialize)]
#[serde(rename_all = "camelCase", default)]
pub struct SpfResult {
pub ip: String,
/// `pass`, `fail`, `softFail`, `neutral`, `none`, `tempError` or `permError`.
pub result: String,
}
#[derive(Debug, Clone, Default, PartialEq, SerdeSerialize, SerdeDeserialize)]
#[serde(rename_all = "camelCase", default)]
pub struct DkimKey {
pub selector: String,
pub state: DkimState,
}
#[derive(Debug, Clone, Copy, Default, PartialEq, Eq, SerdeSerialize, SerdeDeserialize)]
#[serde(rename_all = "camelCase")]
pub enum DkimState {
#[default]
Matches,
/// Nothing published at `<selector>._domainkey.<domain>`.
Missing,
/// Published, but a different key.
Different,
/// The lookup failed.
Error,
}
#[derive(Debug, Clone, Default, PartialEq, SerdeSerialize, SerdeDeserialize)]
#[serde(rename_all = "camelCase", default)]
pub struct Dmarc {
/// `none`, `quarantine` or `reject`.
pub policy: String,
/// DKIM alignment: `relaxed` or `strict`.
pub adkim: String,
/// SPF alignment: `relaxed` or `strict`.
pub aspf: String,
}
#[derive(Debug, Clone, Default, PartialEq, SerdeSerialize, SerdeDeserialize)]
#[serde(rename_all = "camelCase", default)]
pub struct MtaSts {
/// The `_mta-sts` record's id; None when there's no record.
pub record_id: Option<String>,
/// The policy was fetched and parsed. False with a record means the
/// fetch or the parse failed, and `error` says why.
pub fetched: bool,
pub error: Option<String>,
/// `enforce`, `testing` or `none`.
pub mode: Option<String>,
pub max_age: Option<u64>,
/// The domain's MX names no `mx:` line matches.
pub mx_not_covered: Vec<String>,
}
#[derive(Debug, Clone, Default, PartialEq, SerdeSerialize, SerdeDeserialize)]
#[serde(rename_all = "camelCase", default)]
pub struct Certificate {
/// The EHLO name, or an MX name that points at this node.
pub name: String,
/// The node holds a certificate for the name.
pub covered: bool,
}
#[derive(Debug, Clone, Default, PartialEq, SerdeSerialize, SerdeDeserialize)]
#[serde(rename_all = "camelCase", default)]
pub struct Settings {
/// DL-6: lists not to ask, by name.
pub disabled_lists: Vec<String>,
}
impl Settings {
pub fn is_off(&self, list: &str) -> bool {
self.disabled_lists.iter().any(|name| name == list)
}
/// Only the built-in lists' names, once each.
pub fn validate(&self) -> Result<(), String> {
for (i, name) in self.disabled_lists.iter().enumerate() {
if lists::by_name(name).is_none() {
return Err(format!("There's no list called {name:?}."));
}
if self.disabled_lists[..i].contains(name) {
return Err(format!("{name:?} is named twice."));
}
}
Ok(())
}
}
impl Report {
/// DL-20: what a tenant administrator may see: their tenant's domains
/// and nothing about the node's addresses or certificates.
pub fn for_tenant(&self, tenant_id: u32) -> Report {
Report {
node_id: self.node_id,
hostname: self.hostname.clone(),
checked_at: self.checked_at,
addresses: Vec::new(),
domains: self
.domains
.iter()
.filter(|d| d.tenant_id == Some(tenant_id))
.cloned()
.collect(),
certificates: Vec::new(),
}
}
}
// --- Storage --------------------------------------------------------------
struct Json<T>(T);
impl<T: SerdeSerialize> Serialize for Json<T> {
fn serialize(&self) -> trc::Result<Vec<u8>> {
serde_json::to_vec(&self.0).map_err(|err| {
trc::StoreEvent::UnexpectedError
.into_err()
.details("Failed to serialize deliverability data")
.reason(err)
})
}
}
impl<T: for<'de> SerdeDeserialize<'de> + Send + Sync> Deserialize for Json<T> {
fn deserialize(bytes: &[u8]) -> trc::Result<Self> {
serde_json::from_slice(bytes).map(Json).map_err(|err| {
trc::StoreEvent::DataCorruption
.into_err()
.details("Invalid deliverability data")
.reason(err)
})
}
}
fn class(kind: u8, node_id: Option<u64>) -> ValueClass {
let mut key = Vec::with_capacity(10);
key.push(FEATURE);
key.push(kind);
if let Some(node_id) = node_id {
key.extend_from_slice(&node_id.to_be_bytes());
}
ValueClass::Any(AnyClass {
subspace: SUBSPACE_INBUXA,
key,
})
}
pub async fn report(data: &Store, node_id: u64) -> trc::Result<Option<Report>> {
Ok(data
.get_value::<Json<Report>>(ValueKey::from(class(KIND_REPORT, Some(node_id))))
.await
.caused_by(trc::location!())?
.map(|Json(report)| report))
}
/// Every node's report, by node id.
pub async fn reports(data: &Store) -> trc::Result<Vec<Report>> {
let mut out = Vec::new();
data.iterate(
IterateParams::new(
ValueKey::from(class(KIND_REPORT, Some(0))),
ValueKey::from(class(KIND_REPORT, Some(u64::MAX))),
),
|_, value| {
if let Ok(Json(report)) = Json::<Report>::deserialize(value) {
out.push(report);
}
Ok(true)
},
)
.await
.caused_by(trc::location!())?;
out.sort_by_key(|r| r.node_id);
Ok(out)
}
/// Replaces the node's report.
pub async fn put_report(data: &Store, report: &Report) -> trc::Result<()> {
let mut batch = BatchBuilder::new();
batch.set(
class(KIND_REPORT, Some(report.node_id)),
Json(report).serialize()?,
);
data.write(batch.build_all())
.await
.caused_by(trc::location!())?;
Ok(())
}
pub async fn settings(data: &Store) -> trc::Result<Settings> {
Ok(data
.get_value::<Json<Settings>>(ValueKey::from(class(KIND_SETTINGS, None)))
.await
.caused_by(trc::location!())?
.map(|Json(settings)| settings)
.unwrap_or_default())
}
pub async fn put_settings(data: &Store, settings: &Settings) -> trc::Result<()> {
let mut batch = BatchBuilder::new();
batch.set(class(KIND_SETTINGS, None), Json(settings).serialize()?);
data.write(batch.build_all())
.await
.caused_by(trc::location!())?;
Ok(())
}
#[cfg(test)]
mod tests {
use super::*;
#[test]
fn settings_name_only_built_in_lists_once() {
let ok = Settings {
disabled_lists: vec!["Barracuda".into(), "URIBL".into()],
};
assert!(ok.validate().is_ok());
assert!(ok.is_off("Barracuda"));
assert!(!ok.is_off("SpamCop"));
let unknown = Settings {
disabled_lists: vec!["My list".into()],
};
assert!(unknown.validate().is_err());
let twice = Settings {
disabled_lists: vec!["URIBL".into(), "URIBL".into()],
};
assert!(twice.validate().is_err());
}
#[test]
fn a_tenant_sees_only_its_domains() {
let report = Report {
node_id: 2,
hostname: "mx2.example.org".into(),
checked_at: 1,
addresses: vec![Address {
ip: "192.0.2.10".into(),
..Default::default()
}],
domains: vec![
DomainReport {
domain: "a.example".into(),
tenant_id: Some(7),
..Default::default()
},
DomainReport {
domain: "b.example".into(),
tenant_id: Some(8),
..Default::default()
},
DomainReport {
domain: "server.example".into(),
tenant_id: None,
..Default::default()
},
],
certificates: vec![Certificate {
name: "mx2.example.org".into(),
covered: true,
}],
};
let seen = report.for_tenant(7);
assert!(seen.addresses.is_empty());
assert!(seen.certificates.is_empty());
assert_eq!(
seen.domains
.iter()
.map(|d| d.domain.as_str())
.collect::<Vec<_>>(),
["a.example"]
);
}
#[test]
fn a_report_reads_back_with_missing_fields() {
let report: Report = serde_json::from_str(r#"{"nodeId": 3}"#).unwrap();
assert_eq!(report.node_id, 3);
assert!(report.domains.is_empty());
}
}
+1 -1
View File
@@ -1,5 +1,5 @@
/* /*
* SPDX-FileCopyrightText: 2026 Coffey Labs LLC * SPDX-FileCopyrightText: 2026 Coffey Labs
* *
* SPDX-License-Identifier: AGPL-3.0-only * SPDX-License-Identifier: AGPL-3.0-only
*/ */
-120
View File
@@ -1,120 +0,0 @@
/*
* SPDX-FileCopyrightText: 2026 Coffey Labs LLC
*
* SPDX-License-Identifier: AGPL-3.0-only
*/
//! Reports on their way to an outside archive (JR-7). Keys, after `J`:
//!
//! - `o` + the report's queue id: what goes into the built-in journal if
//! the archive never takes the report, as JSON. Cleared once it's
//! delivered or kept.
//! - `w` + journal id (u32): how often that journal's archive didn't take a
//! report, and the last time and reason, for the console's warning.
use super::{FEATURE, Json, entries::Entry};
use serde::{Deserialize as SerdeDeserialize, Serialize as SerdeSerialize};
use store::{
SUBSPACE_INBUXA, Serialize, Store, ValueKey,
write::{AnyClass, BatchBuilder, ValueClass},
};
use trc::AddContext;
const KIND_PENDING: u8 = b'o';
const KIND_FAILURES: u8 = b'w';
/// A report queued to an archive.
#[derive(Debug, Clone, PartialEq, Eq, SerdeSerialize, SerdeDeserialize)]
#[serde(rename_all = "camelCase")]
pub struct Pending {
pub address: String,
/// The entry, should the archive not take it: its own, with the
/// sending journals' retention, whatever else the built-in journal has.
pub entry: Entry,
}
/// How a journal's archive has been taking its reports.
#[derive(Debug, Clone, Default, PartialEq, Eq, SerdeSerialize, SerdeDeserialize)]
#[serde(rename_all = "camelCase")]
pub struct Failures {
pub count: u64,
/// Seconds.
pub last_at: u64,
pub last_reason: String,
}
fn class(kind: u8, id: &[u8]) -> ValueClass {
let mut key = Vec::with_capacity(2 + id.len());
key.push(FEATURE);
key.push(kind);
key.extend_from_slice(id);
ValueClass::Any(AnyClass {
subspace: SUBSPACE_INBUXA,
key,
})
}
pub async fn set_pending(data: &Store, queue_id: u64, pending: &Pending) -> trc::Result<()> {
let mut batch = BatchBuilder::new();
batch.set(
class(KIND_PENDING, &queue_id.to_be_bytes()),
Json(pending).serialize()?,
);
data.write(batch.build_all())
.await
.caused_by(trc::location!())?;
Ok(())
}
pub async fn pending(data: &Store, queue_id: u64) -> trc::Result<Option<Pending>> {
Ok(data
.get_value::<Json<Pending>>(ValueKey::from(class(KIND_PENDING, &queue_id.to_be_bytes())))
.await
.caused_by(trc::location!())?
.map(|Json(pending)| pending))
}
pub async fn clear_pending(data: &Store, queue_id: u64) -> trc::Result<()> {
let mut batch = BatchBuilder::new();
batch.clear(class(KIND_PENDING, &queue_id.to_be_bytes()));
data.write(batch.build_all())
.await
.caused_by(trc::location!())?;
Ok(())
}
pub async fn failures(data: &Store, journal_id: u32) -> trc::Result<Failures> {
Ok(data
.get_value::<Json<Failures>>(ValueKey::from(class(
KIND_FAILURES,
&journal_id.to_be_bytes(),
)))
.await
.caused_by(trc::location!())?
.map(|Json(failures)| failures)
.unwrap_or_default())
}
/// Counts one report an archive didn't take, for each of `journals`.
pub async fn record_failure(
data: &Store,
journals: &[u32],
at: u64,
reason: &str,
) -> trc::Result<()> {
for journal_id in journals {
let mut failures = failures(data, *journal_id).await?;
failures.count += 1;
failures.last_at = at;
failures.last_reason = reason.chars().take(500).collect();
let mut batch = BatchBuilder::new();
batch.set(
class(KIND_FAILURES, &journal_id.to_be_bytes()),
Json(&failures).serialize()?,
);
data.write(batch.build_all())
.await
.caused_by(trc::location!())?;
}
Ok(())
}
-869
View File
@@ -1,869 +0,0 @@
/*
* SPDX-FileCopyrightText: 2026 Coffey Labs LLC
*
* SPDX-License-Identifier: AGPL-3.0-only
*/
//! The built-in journal (JR-5, JR-6, JR-13). Keys, after `J`:
//!
//! - `e` + node + seq: a chain link: its seq, the hash of the link before
//! it, and the SHA-256 of its entry. One chain per node, as the audit log
//! keeps (AU-6), but a link names its entry by hash instead of holding it,
//! so an entry can go at the end of its own retention without breaking
//! the chain: entries don't expire in chain order.
//! - `c` + node + seq: the entry, as JSON; its bytes are what the link's
//! hash names.
//! - `p` + node + seq: when an entry past its retention was purged. A link
//! whose entry is gone without this marker is a broken chain.
//! - `t` + time + node + seq: the time index, for search.
//! - `x` + expiry + node + seq: the expiry index, for purge.
//! - `h` + node: the chain's head: its hash, then its seq as the last eight
//! bytes, which each append asserts.
//! - `f` + node: where the chain starts after purged links at its start
//! were cleared, and the hash the first kept link names.
//!
//! The report itself is a blob, kept by a temporary link that lasts until
//! its entry is purged. Nothing here changes or removes an entry before
//! its time; nothing in JMAP can.
use super::{Direction, FEATURE, Json};
use crate::hold::HELD_UNTIL;
use serde::{Deserialize as SerdeDeserialize, Serialize as SerdeSerialize};
use sha2::{Digest, Sha256};
use std::fmt;
use store::{
BlobStore, Deserialize, IterateParams, SUBSPACE_INBUXA, Serialize, Store, ValueKey,
write::{AnyClass, BatchBuilder, BlobLink, BlobOp, ValueClass, assert::AssertValue},
};
use tokio::sync::Mutex;
use trc::AddContext;
use types::blob_hash::BlobHash;
const KIND_LINK: u8 = b'e';
const KIND_CONTENT: u8 = b'c';
const KIND_PURGED: u8 = b'p';
const KIND_TIME: u8 = b't';
const KIND_EXPIRY: u8 = b'x';
const KIND_HEAD: u8 = b'h';
const KIND_FLOOR: u8 = b'f';
const APPEND_ATTEMPTS: usize = 5;
/// Entries purged per batch.
const PURGE_BATCH: usize = 100;
/// Where one entry sits: its node's chain and its place in it.
#[derive(Debug, Clone, Copy, PartialEq, Eq, Hash, PartialOrd, Ord)]
pub struct EntryId {
pub node: u64,
pub seq: u64,
}
impl EntryId {
/// As one number, for JMAP ids: the node in the top 16 bits.
pub fn to_u64(&self) -> u64 {
(self.node << 48) | (self.seq & ((1 << 48) - 1))
}
pub fn from_u64(id: u64) -> Self {
EntryId {
node: id >> 48,
seq: id & ((1 << 48) - 1),
}
}
}
impl fmt::Display for EntryId {
fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result {
write!(f, "{}-{}", self.node, self.seq)
}
}
/// One journaled message (JR-5).
#[derive(Debug, Clone, PartialEq, Eq, SerdeSerialize, SerdeDeserialize)]
#[serde(rename_all = "camelCase")]
pub struct Entry {
pub queue_id: u64,
/// Seconds.
pub at: u64,
pub direction: Direction,
pub sender: String,
pub authenticated: bool,
pub recipients: Vec<String>,
pub subject: String,
pub message_id: String,
/// The people here on either side, whose holds keep the entry.
pub accounts: Vec<u32>,
pub tenants: Vec<u32>,
/// The journals that took it.
pub journals: Vec<u32>,
pub held: bool,
/// The report's blob, hex.
pub blob: String,
pub size: u64,
/// SHA-256 of the report, hex.
pub sha256: String,
/// Seconds.
pub expires_at: u64,
}
impl Entry {
pub fn blob_hash(&self) -> Option<BlobHash> {
let bytes = unhex(&self.blob)?;
BlobHash::try_from_hash_slice(&bytes).ok()
}
}
#[derive(Debug, Clone, SerdeSerialize, SerdeDeserialize)]
#[serde(rename_all = "camelCase")]
struct Link {
seq: u64,
prev: String,
content: String,
}
#[derive(Debug, Clone, Default, PartialEq, SerdeSerialize, SerdeDeserialize)]
struct Floor {
seq: u64,
prev: String,
}
#[derive(Debug, Clone, Default, PartialEq)]
struct Head {
seq: u64,
hash: String,
}
impl Head {
fn to_bytes(&self) -> Vec<u8> {
let mut bytes = self.hash.as_bytes().to_vec();
bytes.extend_from_slice(&self.seq.to_be_bytes());
bytes
}
}
impl Deserialize for Head {
fn deserialize(bytes: &[u8]) -> trc::Result<Self> {
let split = bytes.len().checked_sub(8).ok_or_else(|| {
trc::StoreEvent::DataCorruption
.into_err()
.details("Invalid journal chain head")
})?;
Ok(Head {
seq: u64::from_be_bytes(bytes[split..].try_into().unwrap()),
hash: String::from_utf8_lossy(&bytes[..split]).into_owned(),
})
}
}
struct Raw(Vec<u8>);
impl Deserialize for Raw {
fn deserialize(bytes: &[u8]) -> trc::Result<Self> {
Ok(Raw(bytes.to_vec()))
}
}
fn class(kind: u8, parts: &[u64]) -> ValueClass {
let mut key = Vec::with_capacity(2 + parts.len() * 8);
key.push(FEATURE);
key.push(kind);
for part in parts {
key.extend_from_slice(&part.to_be_bytes());
}
ValueClass::Any(AnyClass {
subspace: SUBSPACE_INBUXA,
key,
})
}
fn key(kind: u8, parts: &[u64]) -> ValueKey<ValueClass> {
ValueKey::from(class(kind, parts))
}
/// Where an entry's content is kept, for tests that check tampering shows.
pub fn content_key(id: EntryId) -> ValueKey<ValueClass> {
key(KIND_CONTENT, &[id.node, id.seq])
}
/// The numbers after the kind byte, from the key's tail.
fn parse_key(key: &[u8], kind: u8, parts: usize) -> Option<Vec<u64>> {
let len = 2 + parts * 8;
let tail = key.get(key.len().checked_sub(len)?..)?;
(tail[0] == FEATURE && tail[1] == kind).then_some(())?;
Some(
tail[2..]
.chunks_exact(8)
.map(|chunk| u64::from_be_bytes(chunk.try_into().unwrap()))
.collect(),
)
}
pub fn hex(bytes: &[u8]) -> String {
bytes.iter().map(|b| format!("{b:02x}")).collect()
}
fn unhex(value: &str) -> Option<Vec<u8>> {
(value.len() % 2 == 0).then_some(())?;
(0..value.len())
.step_by(2)
.map(|i| u8::from_str_radix(value.get(i..i + 2)?, 16).ok())
.collect()
}
pub fn sha256(bytes: &[u8]) -> String {
hex(&Sha256::digest(bytes))
}
async fn head(data: &Store, node: u64) -> trc::Result<Option<Head>> {
data.get_value::<Head>(key(KIND_HEAD, &[node]))
.await
.caused_by(trc::location!())
}
async fn floor(data: &Store, node: u64) -> trc::Result<Floor> {
Ok(data
.get_value::<Json<Floor>>(key(KIND_FLOOR, &[node]))
.await
.caused_by(trc::location!())?
.map(|Json(floor)| floor)
.unwrap_or(Floor {
seq: 1,
prev: String::new(),
}))
}
async fn nodes(data: &Store) -> trc::Result<Vec<u64>> {
let mut nodes = Vec::new();
data.iterate(
IterateParams::new(key(KIND_HEAD, &[0]), key(KIND_HEAD, &[u64::MAX])).no_values(),
|key, _| {
if let Some(parts) = parse_key(key, KIND_HEAD, 1) {
nodes.push(parts[0]);
}
Ok(true)
},
)
.await
.caused_by(trc::location!())?;
Ok(nodes)
}
/// Lines up this process's appends; the store's assert settles the rest.
static APPENDING: Mutex<()> = Mutex::const_new(());
/// Adds an entry to this node's chain, and links its report's blob (already
/// written) until the entry is purged. An error means nothing was written.
pub async fn append(data: &Store, node: u64, entry: &Entry) -> trc::Result<EntryId> {
let blob = entry.blob_hash().ok_or_else(|| {
trc::StoreEvent::UnexpectedError
.into_err()
.details("Journal entry without a blob")
})?;
let content = Json(entry).serialize()?;
let content_hash = sha256(&content);
let _appending = APPENDING.lock().await;
let mut attempt = 0;
loop {
attempt += 1;
let current = head(data, node).await?;
let (seq, prev) = current
.as_ref()
.map_or((1, String::new()), |head| (head.seq + 1, head.hash.clone()));
let link = Json(&Link {
seq,
prev,
content: content_hash.clone(),
})
.serialize()?;
let new_head = Head {
seq,
hash: sha256(&link),
};
let mut batch = BatchBuilder::new();
batch.assert_value(
class(KIND_HEAD, &[node]),
current.map_or(AssertValue::None, |head| AssertValue::U64(head.seq)),
);
batch
.set(class(KIND_LINK, &[node, seq]), link)
.set(class(KIND_CONTENT, &[node, seq]), content.clone())
.set(class(KIND_TIME, &[entry.at, node, seq]), vec![])
.set(class(KIND_EXPIRY, &[entry.expires_at, node, seq]), vec![])
.set(class(KIND_HEAD, &[node]), new_head.to_bytes())
.set(
BlobOp::Link {
hash: blob.clone(),
to: BlobLink::Temporary { until: HELD_UNTIL },
},
vec![],
)
.set(BlobOp::Commit { hash: blob.clone() }, vec![]);
match data.write(batch.build_all()).await {
Ok(_) => return Ok(EntryId { node, seq }),
Err(err)
if attempt < APPEND_ATTEMPTS
&& matches!(
err.as_ref(),
trc::EventType::Store(trc::StoreEvent::AssertValueFailed)
) => {}
Err(err) => return Err(err.caused_by(trc::location!())),
}
}
}
/// One entry, unless it was purged.
pub async fn get(data: &Store, id: EntryId) -> trc::Result<Option<Entry>> {
Ok(data
.get_value::<Json<Entry>>(key(KIND_CONTENT, &[id.node, id.seq]))
.await
.caused_by(trc::location!())?
.map(|Json(entry)| entry))
}
/// Entries written in `[after, before)` (seconds), newest first, up to
/// `limit`.
pub async fn list(
data: &Store,
after: u64,
before: u64,
limit: usize,
) -> trc::Result<Vec<(EntryId, Entry)>> {
let mut ids = Vec::new();
data.iterate(
IterateParams::new(
key(KIND_TIME, &[after, 0, 0]),
key(KIND_TIME, &[before.saturating_sub(1), u64::MAX, u64::MAX]),
)
.descending()
.no_values(),
|key, _| {
if let Some(parts) = parse_key(key, KIND_TIME, 3) {
ids.push(EntryId {
node: parts[1],
seq: parts[2],
});
}
Ok(ids.len() < limit)
},
)
.await
.caused_by(trc::location!())?;
let mut out = Vec::with_capacity(ids.len());
for id in ids {
if let Some(entry) = get(data, id).await? {
out.push((id, entry));
}
}
Ok(out)
}
/// Most results one search page returns.
pub const MAX_QUERY_LIMIT: usize = 500;
/// A search of the journal (JR-15): conditions that must all hold.
#[derive(Debug, Clone, Default, PartialEq, Eq, SerdeSerialize)]
#[serde(rename_all = "camelCase")]
pub struct Filter {
/// From this time on, in seconds.
#[serde(skip_serializing_if = "Option::is_none")]
pub after: Option<u64>,
/// Before this time, in seconds.
#[serde(skip_serializing_if = "Option::is_none")]
pub before: Option<u64>,
/// Part of the sender's address, ignoring case.
#[serde(skip_serializing_if = "Option::is_none")]
pub sender: Option<String>,
/// Part of any recipient's address, ignoring case.
#[serde(skip_serializing_if = "Option::is_none")]
pub recipient: Option<String>,
/// Part of the sender's or any recipient's address.
#[serde(skip_serializing_if = "Option::is_none")]
pub address: Option<String>,
#[serde(skip_serializing_if = "Option::is_none")]
pub direction: Option<Direction>,
/// Words that must all appear in the subject, ignoring case.
#[serde(skip_serializing_if = "Option::is_none")]
pub text: Option<String>,
#[serde(skip_serializing_if = "Option::is_none")]
pub message_id: Option<String>,
#[serde(skip_serializing_if = "Option::is_none")]
pub journal_id: Option<u32>,
}
impl Filter {
pub fn matches(&self, entry: &Entry) -> bool {
let has = |value: &str, part: &str| value.to_lowercase().contains(&part.to_lowercase());
self.after.is_none_or(|after| entry.at >= after)
&& self.before.is_none_or(|before| entry.at < before)
&& self.sender.as_deref().is_none_or(|s| has(&entry.sender, s))
&& self
.recipient
.as_deref()
.is_none_or(|r| entry.recipients.iter().any(|a| has(a, r)))
&& self
.address
.as_deref()
.is_none_or(|a| has(&entry.sender, a) || entry.recipients.iter().any(|r| has(r, a)))
&& self
.direction
.is_none_or(|d| d == Direction::Any || d == entry.direction)
&& self.text.as_deref().is_none_or(|text| {
let subject = entry.subject.to_lowercase();
text.to_lowercase()
.split_whitespace()
.all(|word| subject.contains(word))
})
&& self.message_id.as_deref().is_none_or(|id| {
entry.message_id.trim_matches(['<', '>']) == id.trim_matches(['<', '>'])
})
&& self.journal_id.is_none_or(|j| entry.journals.contains(&j))
}
}
/// Entries matching `filter`, newest first: a page from `position`, up to
/// `limit`, and, when asked, how many match in all.
pub async fn query(
data: &Store,
filter: &Filter,
position: usize,
limit: usize,
count_all: bool,
) -> trc::Result<(Vec<EntryId>, usize)> {
let after = filter.after.unwrap_or(0);
let before = filter.before.unwrap_or(u64::MAX);
let mut ids = Vec::new();
data.iterate(
IterateParams::new(
key(KIND_TIME, &[after, 0, 0]),
key(KIND_TIME, &[before.saturating_sub(1), u64::MAX, u64::MAX]),
)
.descending()
.no_values(),
|key, _| {
if let Some(parts) = parse_key(key, KIND_TIME, 3) {
ids.push(EntryId {
node: parts[1],
seq: parts[2],
});
}
Ok(true)
},
)
.await
.caused_by(trc::location!())?;
let mut page = Vec::new();
let mut total = 0;
for id in ids {
let Some(entry) = get(data, id).await? else {
continue;
};
if !filter.matches(&entry) {
continue;
}
if total >= position && page.len() < limit {
page.push(id);
}
total += 1;
if !count_all && page.len() >= limit {
break;
}
}
Ok((page, total))
}
/// What a purge did.
#[derive(Debug, Clone, Default, PartialEq, Eq)]
pub struct Purged {
pub removed: usize,
/// Past their time, kept for a legal hold.
pub kept_for_hold: usize,
}
/// Removes entries past their retention (JR-13), except those `held` keeps:
/// the entry, its indexes and its blob's link go; the chain link stays,
/// with a purge marker. Then each chain's start moves past purged links.
pub async fn purge(
data: &Store,
now: u64,
held: impl Fn(&Entry) -> bool + Sync + Send,
) -> trc::Result<Purged> {
let mut due = Vec::new();
data.iterate(
IterateParams::new(
key(KIND_EXPIRY, &[0, 0, 0]),
key(KIND_EXPIRY, &[now, u64::MAX, u64::MAX]),
)
.ascending()
.no_values(),
|key, _| {
if let Some(parts) = parse_key(key, KIND_EXPIRY, 3) {
due.push((
parts[0],
EntryId {
node: parts[1],
seq: parts[2],
},
));
}
Ok(due.len() < 100_000)
},
)
.await
.caused_by(trc::location!())?;
let mut purged = Purged::default();
for chunk in due.chunks(PURGE_BATCH) {
let mut batch = BatchBuilder::new();
for (expires_at, id) in chunk {
let parts = [id.node, id.seq];
let Some(entry) = get(data, *id).await? else {
// Its entry is already gone: only the index is left
batch.clear(class(KIND_EXPIRY, &[*expires_at, id.node, id.seq]));
continue;
};
if held(&entry) {
purged.kept_for_hold += 1;
continue;
}
batch
.clear(class(KIND_CONTENT, &parts))
.clear(class(KIND_TIME, &[entry.at, id.node, id.seq]))
.clear(class(KIND_EXPIRY, &[*expires_at, id.node, id.seq]))
.set(class(KIND_PURGED, &parts), now.to_be_bytes().to_vec());
if let Some(blob) = entry.blob_hash() {
batch.clear(BlobOp::Link {
hash: blob,
to: BlobLink::Temporary { until: HELD_UNTIL },
});
}
purged.removed += 1;
}
if !batch.is_empty() {
data.write(batch.build_all())
.await
.caused_by(trc::location!())?;
}
}
for node in nodes(data).await? {
advance_floor(data, node).await?;
}
Ok(purged)
}
/// Clears the purged links at the start of a node's chain, recording where
/// it now starts and the hash that start names.
async fn advance_floor(data: &Store, node: u64) -> trc::Result<()> {
let start = floor(data, node).await?;
let mut cleared: Vec<u64> = Vec::new();
let mut next = start.clone();
let mut purged_seqs = Vec::new();
data.iterate(
IterateParams::new(
key(KIND_PURGED, &[node, start.seq]),
key(KIND_PURGED, &[node, u64::MAX]),
)
.ascending()
.no_values(),
|key, _| {
if let Some(parts) = parse_key(key, KIND_PURGED, 2) {
purged_seqs.push(parts[1]);
}
Ok(purged_seqs.len() < 100_000)
},
)
.await
.caused_by(trc::location!())?;
for seq in purged_seqs {
if seq != next.seq {
break;
}
let Some(Raw(link)) = data
.get_value::<Raw>(key(KIND_LINK, &[node, seq]))
.await
.caused_by(trc::location!())?
else {
break;
};
next = Floor {
seq: seq + 1,
prev: sha256(&link),
};
cleared.push(seq);
}
if cleared.is_empty() {
return Ok(());
}
// The floor moves first: a run cut short leaves links before it, which
// the next run clears, never a chain that looks broken
let mut batch = BatchBuilder::new();
batch.set(class(KIND_FLOOR, &[node]), Json(&next).serialize()?);
data.write(batch.build_all())
.await
.caused_by(trc::location!())?;
for chunk in cleared.chunks(PURGE_BATCH) {
let mut batch = BatchBuilder::new();
for seq in chunk {
batch
.clear(class(KIND_LINK, &[node, *seq]))
.clear(class(KIND_PURGED, &[node, *seq]));
}
data.write(batch.build_all())
.await
.caused_by(trc::location!())?;
}
Ok(())
}
/// One node's chain, as [`verify`] found it.
#[derive(Debug, Clone, PartialEq, Eq, SerdeSerialize)]
#[serde(rename_all = "camelCase")]
pub struct ChainReport {
pub node: u64,
pub entries: u64,
pub purged: u64,
pub first_seq: u64,
pub last_seq: u64,
#[serde(skip_serializing_if = "Option::is_none")]
pub broken_at: Option<String>,
#[serde(skip_serializing_if = "Option::is_none")]
pub reason: Option<String>,
}
/// Rechecks every node's chain (JR-6): each link names the hash of the one
/// before it, seqs run without gaps, the head matches the last link, each
/// entry hashes to what its link names or was purged, and, with `blobs`,
/// each report is there and hashes to what its entry names.
pub async fn verify(data: &Store, blobs: Option<&BlobStore>) -> trc::Result<Vec<ChainReport>> {
let mut reports = Vec::new();
for node in nodes(data).await? {
let start = floor(data, node).await?;
let head = head(data, node).await?.unwrap_or_default();
let mut report = ChainReport {
node,
entries: 0,
purged: 0,
first_seq: start.seq,
last_seq: start.seq.saturating_sub(1),
broken_at: None,
reason: None,
};
let mut links = Vec::new();
data.iterate(
IterateParams::new(
key(KIND_LINK, &[node, start.seq]),
key(KIND_LINK, &[node, u64::MAX]),
)
.ascending(),
|key, value| {
if let Some(parts) = parse_key(key, KIND_LINK, 2) {
links.push((parts[1], value.to_vec()));
}
Ok(true)
},
)
.await
.caused_by(trc::location!())?;
let mut expected_seq = start.seq;
let mut expected_prev = start.prev.clone();
for (seq, bytes) in links {
let broken = |report: &mut ChainReport, reason: &str| {
report.broken_at = Some(EntryId { node, seq }.to_string());
report.reason = Some(reason.to_string());
};
let Ok(Json(link)) = Json::<Link>::deserialize(&bytes) else {
broken(&mut report, "The link can't be read.");
break;
};
if seq != expected_seq || link.seq != seq {
report.broken_at = Some(EntryId { node, seq }.to_string());
report.reason = Some(format!(
"Entry {expected_seq} is missing; the next one found is {seq}."
));
break;
}
if link.prev != expected_prev {
broken(
&mut report,
"The link doesn't follow from the one before it: one of them was changed.",
);
break;
}
match data
.get_value::<Raw>(key(KIND_CONTENT, &[node, seq]))
.await
.caused_by(trc::location!())?
{
Some(Raw(content)) => {
if sha256(&content) != link.content {
broken(&mut report, "The entry was changed after it was written.");
break;
}
if let Some(blobs) = blobs {
let Ok(Json(entry)) = Json::<Entry>::deserialize(&content) else {
broken(&mut report, "The entry can't be read.");
break;
};
let report_bytes = match entry.blob_hash() {
Some(hash) => blobs
.get_blob(hash.as_slice(), 0..usize::MAX)
.await
.caused_by(trc::location!())?,
None => None,
};
match report_bytes {
Some(bytes) if sha256(&bytes) == entry.sha256 => {}
Some(_) => {
broken(&mut report, "The report doesn't match its entry.");
break;
}
None => {
broken(&mut report, "The report is missing.");
break;
}
}
}
report.entries += 1;
}
None => {
if data
.get_value::<Raw>(key(KIND_PURGED, &[node, seq]))
.await
.caused_by(trc::location!())?
.is_none()
{
broken(&mut report, "The entry was removed before its time.");
break;
}
report.purged += 1;
}
}
expected_prev = sha256(&bytes);
expected_seq = seq + 1;
report.last_seq = seq;
}
if report.broken_at.is_none()
&& (head.seq != report.last_seq
|| (report.last_seq >= report.first_seq && head.hash != expected_prev))
{
report.broken_at = Some(
EntryId {
node,
seq: report.last_seq,
}
.to_string(),
);
report.reason = Some(
"The chain's recorded end doesn't match its last link: entries were removed \
or changed at the end."
.into(),
);
}
reports.push(report);
}
Ok(reports)
}
#[cfg(test)]
mod tests {
use super::*;
#[test]
fn keys_read_back() {
let ValueClass::Any(any) = class(KIND_EXPIRY, &[5, 3, 9]) else {
panic!()
};
assert_eq!(parse_key(&any.key, KIND_EXPIRY, 3), Some(vec![5, 3, 9]));
let mut with_subspace = vec![SUBSPACE_INBUXA];
with_subspace.extend_from_slice(&any.key);
assert_eq!(
parse_key(&with_subspace, KIND_EXPIRY, 3),
Some(vec![5, 3, 9])
);
assert_eq!(parse_key(&any.key, KIND_TIME, 3), None);
}
#[test]
fn filters_match() {
let entry = Entry {
queue_id: 1,
at: 100,
direction: Direction::Outgoing,
sender: "[email protected]".into(),
authenticated: true,
recipients: vec!["[email protected]".into()],
subject: "Q3 figures, final".into(),
message_id: "<[email protected]>".into(),
accounts: vec![3],
tenants: vec![],
journals: vec![2],
held: false,
blob: String::new(),
size: 0,
sha256: String::new(),
expires_at: 0,
};
let yes = |f: Filter| assert!(f.matches(&entry), "{f:?}");
let no = |f: Filter| assert!(!f.matches(&entry), "{f:?}");
yes(Filter::default());
yes(Filter {
sender: Some("alice@".into()),
..Default::default()
});
yes(Filter {
address: Some("BANK".into()),
..Default::default()
});
yes(Filter {
text: Some("final q3".into()),
..Default::default()
});
yes(Filter {
message_id: Some("[email protected]".into()),
..Default::default()
});
yes(Filter {
direction: Some(Direction::Any),
..Default::default()
});
no(Filter {
direction: Some(Direction::Incoming),
..Default::default()
});
no(Filter {
recipient: Some("alice".into()),
..Default::default()
});
no(Filter {
before: Some(100),
..Default::default()
});
yes(Filter {
after: Some(100),
journal_id: Some(2),
..Default::default()
});
no(Filter {
journal_id: Some(5),
..Default::default()
});
}
#[test]
fn hex_round_trips() {
let bytes = [0u8, 1, 0xab, 0xff];
assert_eq!(unhex(&hex(&bytes)), Some(bytes.to_vec()));
assert_eq!(unhex("abc"), None);
assert_eq!(unhex("zz"), None);
}
#[test]
fn ids_read_back() {
let id = EntryId { node: 3, seq: 77 };
assert_eq!(EntryId::from_u64(id.to_u64()), id);
assert_eq!(id.to_string(), "3-77");
}
}
-512
View File
@@ -1,512 +0,0 @@
/*
* SPDX-FileCopyrightText: 2026 Coffey Labs LLC
*
* SPDX-License-Identifier: AGPL-3.0-only
*/
//! Journaling (journaling spec, JR-1 to JR-18): a copy of each message the
//! server queues, with its envelope, kept where nothing in the product
//! changes or removes it before its retention ends.
//!
//! - this module: journals, what makes one valid, and where they're kept;
//! - [`report`]: the journal report around the untouched message (JR-3);
//! - [`entries`]: the built-in journal and its chain (JR-5, JR-6, JR-13).
//!
//! Kept in the fork's subspace (`store::SUBSPACE_INBUXA`). Every key starts
//! with `J`; journals are `j` + id (u32), as JSON. There are few, so they're
//! read whole.
pub mod archive;
pub mod entries;
pub mod report;
use crate::{hold::Member, mailflow::rules::jmap_ids};
use serde::{Deserialize as SerdeDeserialize, Serialize as SerdeSerialize, de::DeserializeOwned};
use std::{
sync::{Arc, RwLock},
time::{Duration, Instant},
};
use store::{
Deserialize, IterateParams, SUBSPACE_INBUXA, Serialize, Store, ValueKey,
write::{AnyClass, BatchBuilder, ValueClass, assert::AssertValue},
};
use trc::AddContext;
pub(crate) const FEATURE: u8 = b'J';
const KIND_JOURNAL: u8 = b'j';
const CREATE_ATTEMPTS: usize = 5;
/// Retention a journal may be given, in days (settled answer 3).
pub const MIN_RETENTION_DAYS: u32 = 30;
pub const MAX_RETENTION_DAYS: u32 = 3650;
/// Most entries in one scope list.
const MAX_LIST: usize = 5_000;
/// Which way a message goes, from this server's side (JR-9).
#[derive(Debug, Clone, Copy, PartialEq, Eq, SerdeSerialize, SerdeDeserialize)]
#[serde(rename_all = "camelCase")]
pub enum Direction {
/// From someone here to at least one recipient elsewhere.
Outgoing,
/// From elsewhere to someone here.
Incoming,
/// From someone here, to people here only.
Internal,
Any,
}
impl Direction {
pub fn as_str(&self) -> &'static str {
match self {
Direction::Outgoing => "outgoing",
Direction::Incoming => "incoming",
Direction::Internal => "internal",
Direction::Any => "any",
}
}
/// A message's direction: `Any` is never one.
pub fn of(sender_local: bool, any_remote: bool, any_local: bool) -> Direction {
match (sender_local, any_remote) {
(true, true) => Direction::Outgoing,
(true, false) => Direction::Internal,
(false, _) if any_local => Direction::Incoming,
// Nobody here on either side: relayed mail counts as outgoing
(false, _) => Direction::Outgoing,
}
}
fn includes(&self, direction: Direction) -> bool {
*self == Direction::Any || *self == direction
}
}
/// Whose mail a journal takes (JR-9): everyone, or people reached through
/// their account, domain, group or tenant. Ids are in the JMAP form.
#[derive(Debug, Clone, Default, PartialEq, Eq, SerdeSerialize, SerdeDeserialize)]
#[serde(rename_all = "camelCase")]
pub struct Scope {
#[serde(default)]
pub everyone: bool,
#[serde(default, with = "jmap_ids")]
pub accounts: Vec<u32>,
#[serde(default, with = "jmap_ids")]
pub groups: Vec<u32>,
#[serde(default, with = "jmap_ids")]
pub domains: Vec<u32>,
#[serde(default, with = "jmap_ids")]
pub tenants: Vec<u32>,
}
impl Scope {
fn lists(&self) -> [&Vec<u32>; 4] {
[&self.accounts, &self.groups, &self.domains, &self.tenants]
}
/// Whether this scope reaches one person here.
pub fn covers(&self, member: &Member) -> bool {
self.everyone
|| self.accounts.contains(&member.account)
|| member.domains.iter().any(|d| self.domains.contains(d))
|| member.groups.iter().any(|g| self.groups.contains(g))
|| member.tenant.is_some_and(|t| self.tenants.contains(&t))
}
}
/// A journal (JR-9): what it takes, and how long its entries are kept.
#[derive(Debug, Clone, PartialEq, Eq, SerdeSerialize, SerdeDeserialize)]
#[serde(rename_all = "camelCase")]
pub struct Journal {
#[serde(default)]
pub id: u32,
pub name: String,
#[serde(default)]
pub description: String,
#[serde(default)]
pub enabled: bool,
pub direction: Direction,
pub scope: Scope,
/// How long an entry this journal writes is kept. An entry keeps the
/// retention it was written with (JR-12).
pub retention_days: u32,
/// Whether entries go into the built-in journal (JR-5).
#[serde(default = "yes")]
pub built_in: bool,
/// An outside archive's journal address, sent each report (JR-7).
#[serde(default, skip_serializing_if = "Option::is_none")]
pub archive_address: Option<String>,
#[serde(default)]
pub created_by: String,
#[serde(default)]
pub created_at: u64,
#[serde(default)]
pub updated_at: u64,
}
/// Why a journal was refused: the property, and what to do.
#[derive(Debug, Clone, PartialEq, Eq)]
pub struct Invalid {
pub property: &'static str,
pub reason: String,
}
fn invalid(property: &'static str, reason: impl Into<String>) -> Result<(), Invalid> {
Err(Invalid {
property,
reason: reason.into(),
})
}
impl Journal {
pub fn validate(&self) -> Result<(), Invalid> {
if self.name.trim().is_empty() {
return invalid("name", "Give the journal a name.");
}
if self.name.len() > 200 || self.description.len() > 2_000 {
return invalid("name", "The name or description is too long.");
}
if !(MIN_RETENTION_DAYS..=MAX_RETENTION_DAYS).contains(&self.retention_days) {
return invalid(
"retentionDays",
format!("Keep entries between {MIN_RETENTION_DAYS} and {MAX_RETENTION_DAYS} days."),
);
}
// Neither is a journal only rules send mail to (JR-10)
let chosen = self.scope.lists().iter().any(|list| !list.is_empty());
if self.scope.everyone && chosen {
return invalid(
"scope",
"Journal everyone, or choose accounts, groups, domains or tenants; not both.",
);
}
if !self.built_in && self.archive_address.is_none() {
return invalid(
"builtIn",
"Keep entries in the built-in journal, send them to an archive, or both.",
);
}
if let Some(address) = &self.archive_address
&& !is_address(address)
{
return invalid(
"archiveAddress",
format!("\"{address}\" isn't an email address."),
);
}
if self.scope.lists().iter().any(|list| list.len() > MAX_LIST) {
return invalid("scope", format!("Choose at most {MAX_LIST} of each."));
}
Ok(())
}
/// Whether this journal takes a message going `direction` with these
/// people here on either side.
/// Whether only rules send this journal mail (JR-10).
pub fn rules_only(&self) -> bool {
!self.scope.everyone && self.scope.lists().iter().all(|list| list.is_empty())
}
pub fn takes(&self, direction: Direction, members: &[Member]) -> bool {
self.enabled
&& self.direction.includes(direction)
&& (self.scope.everyone || members.iter().any(|m| self.scope.covers(m)))
}
}
fn yes() -> bool {
true
}
/// An address an archive can be sent to: one `@`, something either side,
/// nothing that would break an envelope.
fn is_address(address: &str) -> bool {
address.len() <= 320
&& address.split_once('@').is_some_and(|(local, domain)| {
!local.is_empty() && domain.contains('.') && !domain.contains('@')
})
&& !address
.chars()
.any(|c| c.is_whitespace() || c.is_control() || matches!(c, '<' | '>' | ',' | ';'))
}
/// A value stored as JSON.
pub(crate) struct Json<T>(pub T);
impl<T: SerdeSerialize> Serialize for Json<T> {
fn serialize(&self) -> trc::Result<Vec<u8>> {
serde_json::to_vec(&self.0).map_err(|err| {
trc::StoreEvent::UnexpectedError
.into_err()
.details("Failed to serialize a journal record")
.reason(err)
})
}
}
impl<T: DeserializeOwned + Sync + Send> Deserialize for Json<T> {
fn deserialize(bytes: &[u8]) -> trc::Result<Self> {
serde_json::from_slice(bytes).map(Json).map_err(|err| {
trc::StoreEvent::DataCorruption
.into_err()
.details("Invalid journal record")
.reason(err)
})
}
}
fn class(id: u32) -> ValueClass {
let mut key = Vec::with_capacity(6);
key.push(FEATURE);
key.push(KIND_JOURNAL);
key.extend_from_slice(&id.to_be_bytes());
ValueClass::Any(AnyClass {
subspace: SUBSPACE_INBUXA,
key,
})
}
fn key(id: u32) -> ValueKey<ValueClass> {
ValueKey::from(class(id))
}
pub async fn get(data: &Store, id: u32) -> trc::Result<Option<Journal>> {
Ok(data
.get_value::<Json<Journal>>(key(id))
.await
.caused_by(trc::location!())?
.map(|Json(journal)| journal))
}
/// Every journal, oldest first.
pub async fn all(data: &Store) -> trc::Result<Vec<Journal>> {
let mut journals = Vec::new();
data.iterate(IterateParams::new(key(0), key(u32::MAX)), |_, value| {
if let Ok(Json(journal)) = Json::<Journal>::deserialize(value) {
journals.push(journal);
}
Ok(true)
})
.await
.caused_by(trc::location!())?;
journals.sort_by_key(|journal| journal.id);
Ok(journals)
}
/// Writes a new journal under the next free id, which it returns.
pub async fn create(data: &Store, journal: &Journal) -> trc::Result<u32> {
let mut attempt = 0;
loop {
attempt += 1;
let id = all(data).await?.iter().map(|j| j.id).max().unwrap_or(0) + 1;
let stored = Journal {
id,
..journal.clone()
};
let mut batch = BatchBuilder::new();
batch.assert_value(class(id), AssertValue::None);
batch.set(class(id), Json(&stored).serialize()?);
match data.write(batch.build_all()).await {
Ok(_) => {
invalidate();
return Ok(id);
}
Err(err)
if attempt < CREATE_ATTEMPTS
&& matches!(
err.as_ref(),
trc::EventType::Store(trc::StoreEvent::AssertValueFailed)
) => {}
Err(err) => return Err(err.caused_by(trc::location!())),
}
}
}
/// Replaces a stored journal (same id).
pub async fn update(data: &Store, journal: &Journal) -> trc::Result<()> {
let mut batch = BatchBuilder::new();
batch.set(class(journal.id), Json(journal).serialize()?);
data.write(batch.build_all())
.await
.caused_by(trc::location!())?;
invalidate();
Ok(())
}
/// Removes a journal. Its entries stay, each until its own time.
pub async fn delete(data: &Store, id: u32) -> trc::Result<()> {
let mut batch = BatchBuilder::new();
batch.clear(class(id));
data.write(batch.build_all())
.await
.caused_by(trc::location!())?;
invalidate();
Ok(())
}
/// How long a node keeps its copy of the journals before reading them again.
pub const TTL: Duration = Duration::from_secs(30);
type Cached = Option<(Instant, Arc<Vec<Journal>>)>;
static CACHE: RwLock<Cached> = RwLock::new(None);
/// Forgets this node's copy, so the next message reads the journals again.
pub fn invalidate() {
if let Ok(mut cache) = CACHE.write() {
*cache = None;
}
}
/// The enabled journals, from this node's copy (refreshed every [`TTL`]).
pub async fn enabled(data: &Store) -> trc::Result<Arc<Vec<Journal>>> {
if let Ok(cache) = CACHE.read()
&& let Some((at, journals)) = cache.as_ref()
&& at.elapsed() < TTL
{
return Ok(journals.clone());
}
let journals = Arc::new(
all(data)
.await?
.into_iter()
.filter(|journal| journal.enabled)
.collect::<Vec<_>>(),
);
if let Ok(mut cache) = CACHE.write() {
*cache = Some((Instant::now(), journals.clone()));
}
Ok(journals)
}
#[cfg(test)]
mod tests {
use super::*;
fn journal(scope: Scope) -> Journal {
Journal {
id: 1,
name: "Finance".into(),
description: String::new(),
enabled: true,
direction: Direction::Any,
scope,
retention_days: 365,
built_in: true,
archive_address: None,
created_by: String::new(),
created_at: 0,
updated_at: 0,
}
}
fn member(account: u32, groups: Vec<u32>) -> Member {
Member {
account,
domains: vec![1],
groups,
tenant: None,
}
}
#[test]
fn scope_is_everyone_or_chosen() {
assert!(
journal(Scope {
everyone: true,
..Default::default()
})
.validate()
.is_ok()
);
// Nobody chosen: only rules send it mail
let rules_only = journal(Scope::default());
assert!(rules_only.validate().is_ok());
assert!(rules_only.rules_only());
assert!(!rules_only.takes(Direction::Any, &[member(3, vec![7])]));
let both = Scope {
everyone: true,
groups: vec![4],
..Default::default()
};
assert_eq!(journal(both).validate().unwrap_err().property, "scope");
}
#[test]
fn destinations() {
let mut j = journal(Scope {
everyone: true,
..Default::default()
});
j.built_in = false;
assert_eq!(j.validate().unwrap_err().property, "builtIn");
j.archive_address = Some("[email protected]".into());
assert!(j.validate().is_ok());
for bad in [
"archive",
"a@b",
"a [email protected]",
"<[email protected]>",
"a@[email protected]",
] {
j.archive_address = Some(bad.into());
assert_eq!(
j.validate().unwrap_err().property,
"archiveAddress",
"{bad}"
);
}
// Stored before destinations existed: the built-in journal
let old: Journal = serde_json::from_str(
r#"{"name":"Old","direction":"any","scope":{"everyone":true},"retentionDays":30}"#,
)
.unwrap();
assert!(old.built_in && old.archive_address.is_none());
}
#[test]
fn retention_has_bounds() {
let mut j = journal(Scope {
everyone: true,
..Default::default()
});
j.retention_days = 29;
assert_eq!(j.validate().unwrap_err().property, "retentionDays");
j.retention_days = 3651;
assert!(j.validate().is_err());
j.retention_days = 3650;
assert!(j.validate().is_ok());
}
#[test]
fn takes_by_direction_and_member() {
let mut j = journal(Scope {
groups: vec![7],
..Default::default()
});
assert!(j.takes(Direction::Outgoing, &[member(3, vec![7])]));
assert!(!j.takes(Direction::Outgoing, &[member(3, vec![8])]));
assert!(!j.takes(Direction::Outgoing, &[]));
j.direction = Direction::Incoming;
assert!(!j.takes(Direction::Outgoing, &[member(3, vec![7])]));
j.enabled = false;
assert!(!j.takes(Direction::Incoming, &[member(3, vec![7])]));
}
#[test]
fn directions() {
assert_eq!(Direction::of(true, true, true), Direction::Outgoing);
assert_eq!(Direction::of(true, false, true), Direction::Internal);
assert_eq!(Direction::of(false, false, true), Direction::Incoming);
assert_eq!(Direction::of(false, true, true), Direction::Incoming);
}
#[test]
fn scope_ids_are_jmap_ids() {
let scope: Scope = serde_json::from_str(r#"{"groups":["b"],"tenants":[7]}"#).unwrap();
assert_eq!(scope.groups, vec![1]);
assert_eq!(scope.tenants, vec![7]);
assert_eq!(
serde_json::to_value(&scope).unwrap()["tenants"],
serde_json::json!(["h"])
);
}
}
-385
View File
@@ -1,385 +0,0 @@
/*
* SPDX-FileCopyrightText: 2026 Coffey Labs LLC
*
* SPDX-License-Identifier: AGPL-3.0-only
*/
//! The journal report (JR-3, JR-4): a message whose first part lists the
//! envelope, one field a line, and whose second part is the message as it
//! was queued, byte for byte, as `message/rfc822`. Field names are fixed
//! English: a report is a record, and scripts read it.
use super::Direction;
use mail_builder::headers::{Header, date::Date, text::Text};
use mail_parser::MessageParser;
use sha2::{Digest, Sha256};
/// One envelope recipient, with the address it was given as (a list's, for
/// the list's members).
#[derive(Debug, Clone, PartialEq, Eq)]
pub struct Recipient {
pub address: String,
pub orcpt: Option<String>,
/// The mail flow rule that added or redirected to it.
pub added_by: Option<String>,
}
/// What the queue knows about a message.
#[derive(Debug, Clone)]
pub struct Envelope<'x> {
pub sender: &'x str,
pub authenticated: bool,
pub recipients: &'x [Recipient],
pub queue_id: u64,
/// Seconds.
pub received: u64,
pub direction: Direction,
pub held: bool,
}
/// What a report says, besides the envelope's own fields.
#[derive(Debug, Clone, Default, PartialEq, Eq)]
pub struct Fields {
pub subject: String,
pub message_id: String,
pub to: Vec<String>,
pub cc: Vec<String>,
/// Envelope recipients in neither To nor Cc, nor reached through a list.
pub bcc: Vec<String>,
/// A list's address, and its members among the recipients.
pub expanded: Vec<(String, Vec<String>)>,
/// A rule's name, and the recipients it added.
pub added: Vec<(String, Vec<String>)>,
}
/// One line's worth of a value: no line breaks, no control characters.
fn line(value: &str) -> String {
value
.chars()
.map(|c| if c.is_control() { ' ' } else { c })
.collect::<String>()
.trim()
.to_string()
}
/// The address an ORCPT names, without its `rfc822;` type.
fn orcpt_address(orcpt: &str) -> String {
let orcpt = orcpt.trim();
let bare = match orcpt.split_once(';') {
Some((kind, address)) if kind.eq_ignore_ascii_case("rfc822") => address,
_ => orcpt,
};
bare.trim().to_lowercase()
}
/// Sorts the envelope's recipients by how they were addressed.
pub fn fields(envelope: &Envelope<'_>, original: &[u8]) -> Fields {
let parsed = MessageParser::default().parse_headers(original);
let headed = |which: Option<&mail_parser::Address<'_>>| -> Vec<String> {
which
.map(|list| {
list.iter()
.filter_map(|addr| addr.address())
.map(|address| address.to_lowercase())
.collect()
})
.unwrap_or_default()
};
let (subject, message_id, header_to, header_cc) = match &parsed {
Some(message) => (
message.subject().map(line).unwrap_or_default(),
message
.message_id()
.map(|id| format!("<{}>", line(id)))
.unwrap_or_default(),
headed(message.to()),
headed(message.cc()),
),
None => Default::default(),
};
let mut fields = Fields {
subject,
message_id,
..Default::default()
};
for rcpt in envelope.recipients {
let address = rcpt.address.to_lowercase();
let via = rcpt
.orcpt
.as_deref()
.map(orcpt_address)
.filter(|via| !via.is_empty() && *via != address);
if let Some(rule) = &rcpt.added_by {
match fields.added.iter_mut().find(|(name, _)| name == rule) {
Some((_, added)) => added.push(line(&rcpt.address)),
None => fields.added.push((line(rule), vec![line(&rcpt.address)])),
}
} else if header_to.contains(&address) {
fields.to.push(line(&rcpt.address));
} else if header_cc.contains(&address) {
fields.cc.push(line(&rcpt.address));
} else if let Some(via) = via {
match fields.expanded.iter_mut().find(|(list, _)| *list == via) {
Some((_, members)) => members.push(line(&rcpt.address)),
None => fields
.expanded
.push((line(&via), vec![line(&rcpt.address)])),
}
} else {
fields.bcc.push(line(&rcpt.address));
}
}
fields
}
/// The report's first part.
pub fn text(envelope: &Envelope<'_>, fields: &Fields) -> String {
let mut out = String::new();
let mut field = |name: &str, value: &str| {
if !value.is_empty() {
out.push_str(name);
out.push_str(": ");
out.push_str(value);
out.push_str("\r\n");
}
};
let sender = if envelope.sender.is_empty() {
"<>".to_string()
} else {
line(envelope.sender)
};
field("Sender", &sender);
field(
"Authenticated",
if envelope.authenticated { "yes" } else { "no" },
);
field("Subject", &fields.subject);
field("Message-ID", &fields.message_id);
field("Queue ID", &format!("{:x}", envelope.queue_id));
field(
"Received",
&mail_parser::DateTime::from_timestamp(envelope.received as i64).to_rfc3339(),
);
field("Direction", envelope.direction.as_str());
field("To", &fields.to.join(", "));
field("Cc", &fields.cc.join(", "));
field("Bcc", &fields.bcc.join(", "));
for (list, members) in &fields.expanded {
field("Expanded", &format!("{list} -> {}", members.join(", ")));
}
for (rule, added) in &fields.added {
field("Added by rule", &format!("{rule} -> {}", added.join(", ")));
}
if envelope.held {
field("Held for review", "yes");
}
out
}
fn hex(bytes: &[u8]) -> String {
bytes.iter().map(|b| format!("{b:02x}")).collect()
}
/// Whether a message can travel as 8bit: no NULs, no line past 998 bytes.
fn fits_8bit(message: &[u8]) -> bool {
!message.contains(&0) && message.split(|b| *b == b'\n').all(|l| l.len() <= 998)
}
/// The whole report: headers, the fields, then the original untouched.
/// `from` is the address the report is from; `host` names the server in its
/// Message-ID.
pub fn build(
envelope: &Envelope<'_>,
original: &[u8],
from: &str,
host: &str,
) -> (Vec<u8>, Fields) {
let fields = fields(envelope, original);
let body = text(envelope, &fields);
// A boundary that can't occur in the original
let mut boundary = format!("journal-{}", &hex(&Sha256::digest(original))[..32]);
while original
.windows(boundary.len())
.any(|window| window == boundary.as_bytes())
{
boundary.push('x');
}
let mut out: Vec<u8> = Vec::with_capacity(original.len() + body.len() + 1024);
out.extend_from_slice(format!("From: Journal <{}>\r\n", line(from)).as_bytes());
out.extend_from_slice(b"Date: ");
out.extend_from_slice(Date::new(envelope.received as i64).to_rfc822().as_bytes());
out.extend_from_slice(b"\r\n");
out.extend_from_slice(b"Subject: ");
let subject = if fields.subject.is_empty() {
"Journal report".to_string()
} else {
format!("Journal report: {}", fields.subject)
};
Text::new(subject).write_header(&mut out, "Subject: ".len());
out.extend_from_slice(
format!(
"Message-ID: <journal.{:x}.{}@{}>\r\n",
envelope.queue_id,
envelope.received,
line(host)
)
.as_bytes(),
);
out.extend_from_slice(format!("X-Inbuxa-Journal: {:x}\r\n", envelope.queue_id).as_bytes());
out.extend_from_slice(b"MIME-Version: 1.0\r\n");
out.extend_from_slice(
format!("Content-Type: multipart/mixed; boundary=\"{boundary}\"\r\n\r\n").as_bytes(),
);
out.extend_from_slice(format!("--{boundary}\r\n").as_bytes());
out.extend_from_slice(
b"Content-Type: text/plain; charset=utf-8\r\nContent-Transfer-Encoding: 8bit\r\n\r\n",
);
out.extend_from_slice(body.as_bytes());
out.extend_from_slice(format!("\r\n--{boundary}\r\n").as_bytes());
out.extend_from_slice(b"Content-Type: message/rfc822\r\n");
out.extend_from_slice(b"Content-Disposition: attachment; filename=\"original.eml\"\r\n");
out.extend_from_slice(if fits_8bit(original) {
b"Content-Transfer-Encoding: 8bit\r\n\r\n".as_slice()
} else {
b"Content-Transfer-Encoding: binary\r\n\r\n".as_slice()
});
out.extend_from_slice(original);
// The line break before a boundary belongs to the boundary: the
// original keeps its own last one
out.extend_from_slice(format!("\r\n--{boundary}--\r\n").as_bytes());
(out, fields)
}
/// Where the original starts and ends inside a report [`build`] made.
pub fn original(report: &[u8]) -> Option<&[u8]> {
let parsed = MessageParser::default().parse(report)?;
let part = parsed.attachment(0)?;
let start = part.raw_body_offset() as usize;
let end = part.raw_end_offset() as usize;
report.get(start..end)
}
#[cfg(test)]
mod tests {
use super::*;
const ORIGINAL: &[u8] = b"From: [email protected]\r\n\
To: Bank <pay@bank.example>\r\n\
Cc: bob@example.com\r\n\
Subject: Q3 figures\r\n\
Message-ID: <abc@example.com>\r\n\
\r\n\
The figures.\r\n";
fn rcpt(address: &str, orcpt: Option<&str>) -> Recipient {
Recipient {
address: address.into(),
orcpt: orcpt.map(Into::into),
added_by: None,
}
}
fn envelope(recipients: &[Recipient]) -> Envelope<'_> {
Envelope {
sender: "[email protected]",
authenticated: true,
recipients,
queue_id: 0x1a2b,
received: 1_790_000_000,
direction: Direction::Outgoing,
held: false,
}
}
#[test]
fn recipients_sorted_by_how_they_were_addressed() {
let recipients = [
rcpt("[email protected]", None),
rcpt("[email protected]", Some("rfc822;[email protected]")),
rcpt("[email protected]", None),
rcpt("[email protected]", Some("[email protected]")),
rcpt("[email protected]", Some("rfc822;[email protected]")),
];
let fields = fields(&envelope(&recipients), ORIGINAL);
assert_eq!(fields.subject, "Q3 figures");
assert_eq!(fields.message_id, "<[email protected]>");
assert_eq!(fields.to, vec!["[email protected]"]);
assert_eq!(fields.cc, vec!["[email protected]"]);
assert_eq!(fields.bcc, vec!["[email protected]"]);
assert_eq!(
fields.expanded,
vec![(
"[email protected]".to_string(),
vec![
"[email protected]".to_string(),
"[email protected]".to_string()
]
)]
);
}
#[test]
fn report_carries_the_original_untouched() {
let recipients = [
rcpt("[email protected]", None),
rcpt("[email protected]", None),
];
let (report, _) = build(
&envelope(&recipients),
ORIGINAL,
"[email protected]",
"mx.example.com",
);
let text = String::from_utf8_lossy(&report);
assert!(text.contains("Sender: [email protected]\r\n"));
assert!(text.contains("Bcc: [email protected]\r\n"));
assert!(text.contains("Queue ID: 1a2b\r\n"));
assert!(text.contains("Direction: outgoing\r\n"));
assert!(text.contains("Subject: Journal report: Q3 figures\r\n"));
assert!(!text.contains("Held for review"));
assert_eq!(original(&report), Some(ORIGINAL));
let unterminated = &ORIGINAL[..ORIGINAL.len() - 2];
let (report, _) = build(
&envelope(&recipients),
unterminated,
"[email protected]",
"mx.example.com",
);
assert_eq!(original(&report), Some(unterminated));
}
#[test]
fn rule_added_recipients_say_so() {
let mut copied = rcpt("[email protected]", None);
copied.added_by = Some("Copy finance".into());
let recipients = [rcpt("[email protected]", None), copied];
let env = envelope(&recipients);
let fields = fields(&env, ORIGINAL);
assert!(fields.bcc.is_empty(), "{fields:?}");
assert!(
text(&env, &fields).contains("Added by rule: Copy finance -> [email protected]\r\n")
);
}
#[test]
fn values_stay_on_one_line() {
let recipients = [rcpt("[email protected]", None)];
let mut env = envelope(&recipients);
env.sender = "[email protected]\r\nBcc: [email protected]";
env.held = true;
let body = text(&env, &Fields::default());
assert_eq!(body.matches("\r\n").count(), body.lines().count());
assert!(body.contains("Sender: [email protected] Bcc: [email protected]\r\n"));
assert!(body.contains("Held for review: yes\r\n"));
}
#[test]
fn an_empty_sender_is_shown_as_such() {
let recipients = [rcpt("[email protected]", None)];
let mut env = envelope(&recipients);
env.sender = "";
assert!(text(&env, &Fields::default()).starts_with("Sender: <>\r\n"));
}
}
+1 -4
View File
@@ -1,5 +1,5 @@
/* /*
* SPDX-FileCopyrightText: 2026 Coffey Labs LLC * SPDX-FileCopyrightText: 2026 Coffey Labs
* *
* SPDX-License-Identifier: AGPL-3.0-only * SPDX-License-Identifier: AGPL-3.0-only
*/ */
@@ -21,11 +21,8 @@
pub mod ai; pub mod ai;
pub mod audit; pub mod audit;
pub mod branding; pub mod branding;
pub mod deliverability; // inbuxa: the deliverability check (not a rebuild)
pub mod hold; pub mod hold;
pub mod journal;
pub mod lock; pub mod lock;
pub mod mailflow;
pub mod masked_email; pub mod masked_email;
pub mod privacy; pub mod privacy;
pub mod security; pub mod security;
+4 -73
View File
@@ -1,5 +1,5 @@
/* /*
* SPDX-FileCopyrightText: 2026 Coffey Labs LLC * SPDX-FileCopyrightText: 2026 Coffey Labs
* *
* SPDX-License-Identifier: AGPL-3.0-only * SPDX-License-Identifier: AGPL-3.0-only
*/ */
@@ -66,53 +66,6 @@ const KIND_DELEGATE: u8 = b'd';
/// Most delegates one lock may have (AL-5). /// Most delegates one lock may have (AL-5).
pub const MAX_DELEGATES: usize = 10; pub const MAX_DELEGATES: usize = 10;
/// Most people one shared mailbox may have (MA-S): a help desk is bigger
/// than the handful a departed colleague's mail is handed to.
pub const MAX_SHARED_MAILBOX_DELEGATES: usize = 100;
/// What a lock is for (multi-account spec, MA-S).
///
/// Both kinds keep receiving mail, can't be signed in to, and are opened by
/// delegates through real grants. A shared mailbox is a role address such
/// as support@: it needs no reason, holds more people, runs its own Sieve
/// replies (an automatic acknowledgement), records only what is sent as it,
/// and may only send as its own addresses.
#[derive(Debug, Clone, Copy, Default, PartialEq, Eq, Hash, SerdeSerialize, SerdeDeserialize)]
#[serde(rename_all = "camelCase")]
pub enum Kind {
#[default]
Lock,
SharedMailbox,
}
impl Kind {
pub fn as_str(&self) -> &'static str {
match self {
Kind::Lock => "lock",
Kind::SharedMailbox => "sharedMailbox",
}
}
pub fn parse(value: &str) -> Option<Self> {
match value {
"lock" => Some(Kind::Lock),
"sharedMailbox" => Some(Kind::SharedMailbox),
_ => None,
}
}
pub fn is_lock(&self) -> bool {
matches!(self, Kind::Lock)
}
pub fn max_delegates(&self) -> usize {
match self {
Kind::Lock => MAX_DELEGATES,
Kind::SharedMailbox => MAX_SHARED_MAILBOX_DELEGATES,
}
}
}
/// What a delegate may do in the locked account (AL-6). /// What a delegate may do in the locked account (AL-6).
#[derive(Debug, Clone, Copy, PartialEq, Eq, Hash, SerdeSerialize, SerdeDeserialize)] #[derive(Debug, Clone, Copy, PartialEq, Eq, Hash, SerdeSerialize, SerdeDeserialize)]
#[serde(rename_all = "camelCase")] #[serde(rename_all = "camelCase")]
@@ -236,9 +189,6 @@ pub struct Replaced {
#[serde(rename_all = "camelCase")] #[serde(rename_all = "camelCase")]
pub struct Lock { pub struct Lock {
pub account_id: u32, pub account_id: u32,
/// Absent on locks written before shared mailboxes existed: a lock.
#[serde(default, skip_serializing_if = "Kind::is_lock")]
pub kind: Kind,
pub reason: String, pub reason: String,
/// Seconds since the epoch. /// Seconds since the epoch.
pub locked_at: u64, pub locked_at: u64,
@@ -451,9 +401,8 @@ pub async fn all(data: &Store) -> trc::Result<Vec<Lock>> {
Ok(locks) Ok(locks)
} }
/// The accounts delegated to `delegate`, with its delegation in each and /// The accounts delegated to `delegate`, with its delegation in each.
/// the kind of lock it is in. pub async fn delegated_to(data: &Store, delegate: u32) -> trc::Result<Vec<(u32, Delegate)>> {
pub async fn delegated_to(data: &Store, delegate: u32) -> trc::Result<Vec<(u32, Delegate, Kind)>> {
let mut locked = Vec::new(); let mut locked = Vec::new();
data.iterate( data.iterate(
IterateParams::new( IterateParams::new(
@@ -476,7 +425,7 @@ pub async fn delegated_to(data: &Store, delegate: u32) -> trc::Result<Vec<(u32,
if let Some(lock) = get(data, account_id).await? if let Some(lock) = get(data, account_id).await?
&& let Some(delegation) = lock.delegate(delegate) && let Some(delegation) = lock.delegate(delegate)
{ {
delegations.push((account_id, delegation.clone(), lock.kind)); delegations.push((account_id, delegation.clone()));
} }
} }
Ok(delegations) Ok(delegations)
@@ -523,22 +472,6 @@ pub async fn remove(data: &Store, lock: &Lock) -> trc::Result<()> {
mod tests { mod tests {
use super::*; use super::*;
#[test]
fn kind_reads_back_and_defaults_to_lock() {
// MA-S: a lock stored before shared mailboxes existed has no kind
let stored = r#"{"accountId":1,"reason":"r","lockedAt":0,"lockedBy":"admin","delegates":[]}"#;
let lock: Lock = serde_json::from_str(stored).unwrap();
assert_eq!(lock.kind, Kind::Lock);
assert!(!serde_json::to_string(&lock).unwrap().contains("kind"), "a lock is written as before");
let shared = Lock { kind: Kind::SharedMailbox, ..lock };
let written = serde_json::to_string(&shared).unwrap();
assert!(written.contains(r#""kind":"sharedMailbox""#), "{written}");
assert_eq!(serde_json::from_str::<Lock>(&written).unwrap().kind, Kind::SharedMailbox);
assert_eq!(Kind::parse("sharedMailbox"), Some(Kind::SharedMailbox));
assert_eq!(Kind::SharedMailbox.max_delegates(), MAX_SHARED_MAILBOX_DELEGATES);
}
#[test] #[test]
fn keys_read_back() { fn keys_read_back() {
let ValueClass::Any(any) = class(KIND_DELEGATE, &[7, 9]) else { let ValueClass::Any(any) = class(KIND_DELEGATE, &[7, 9]) else {
@@ -576,7 +509,6 @@ mod tests {
fn lock_with(delegates: Vec<Delegate>, replaced: Vec<Replaced>) -> Lock { fn lock_with(delegates: Vec<Delegate>, replaced: Vec<Replaced>) -> Lock {
Lock { Lock {
account_id: 1, account_id: 1,
kind: Kind::Lock,
reason: "r".into(), reason: "r".into(),
locked_at: 0, locked_at: 0,
locked_by: "admin".into(), locked_by: "admin".into(),
@@ -691,7 +623,6 @@ mod tests {
fn expired_delegations_grant_nothing() { fn expired_delegations_grant_nothing() {
let lock = Lock { let lock = Lock {
account_id: 1, account_id: 1,
kind: Kind::Lock,
reason: "Left the company".into(), reason: "Left the company".into(),
locked_at: 100, locked_at: 100,
locked_by: "admin".into(), locked_by: "admin".into(),
-53
View File
@@ -1,53 +0,0 @@
/*
* SPDX-FileCopyrightText: 2026 Coffey Labs LLC
*
* SPDX-License-Identifier: AGPL-3.0-only
*/
//! The compiled rules, kept per node so a message doesn't read the store.
//! A change made on this node applies at once; one made on another node
//! within [`TTL`], when the copy here is next refreshed.
use super::{engine::Compiled, rules};
use std::{
sync::{Arc, RwLock},
time::{Duration, Instant},
};
use store::Store;
/// How long a node keeps its copy before reading the rules again.
pub const TTL: Duration = Duration::from_secs(30);
static CACHE: RwLock<Option<(Instant, Arc<Compiled>)>> = RwLock::new(None);
/// Forgets the copy, so the next message reads the rules again.
pub fn invalidate() {
if let Ok(mut cache) = CACHE.write() {
*cache = None;
}
}
/// The enabled rules, compiled. A rule that no longer compiles is left out
/// and reported, once per refresh.
pub async fn compiled(data: &Store) -> trc::Result<Arc<Compiled>> {
if let Ok(cache) = CACHE.read()
&& let Some((at, compiled)) = cache.as_ref()
&& at.elapsed() < TTL
{
return Ok(compiled.clone());
}
let (compiled, skipped) = Compiled::new(&rules::all(data).await?);
for (id, reason) in skipped {
trc::event!(
Store(trc::StoreEvent::DataCorruption),
Id = u64::from(id),
Reason = reason,
Details = "Mail rule skipped: it no longer compiles"
);
}
let compiled = Arc::new(compiled);
if let Ok(mut cache) = CACHE.write() {
*cache = Some((Instant::now(), compiled.clone()));
}
Ok(compiled)
}
@@ -1,49 +0,0 @@
/*
* SPDX-FileCopyrightText: 2026 Coffey Labs LLC
*
* SPDX-License-Identifier: AGPL-3.0-only
*/
//! African identifiers (§2.3): South Africa's ID number.
use super::{Detector, Findings, Region, Strength, checks, valid_short_date};
use regex::Regex;
use std::sync::LazyLock;
pub static DETECTORS: &[Detector] = &[Detector::new(
"za-id",
"South Africa: ID number",
Region::Africa,
Strength::Checked,
za_id,
)];
/// Birth date `YYMMDD`, four digits, citizenship (0, 1 or 2), 8 or 9, a Luhn
/// check digit. The date and the two fixed digits make it strong enough to
/// count alone.
static ZA_ID: LazyLock<Regex> = LazyLock::new(|| {
Regex::new(r"\b(\d{2})(\d{2})(\d{2})\d{4}[012][89]\d\b").expect("detector pattern")
});
fn za_id(text: &str, findings: &mut Findings) {
for c in ZA_ID.captures_iter(text) {
let n = &c[0];
let num = |s: &str| s.parse::<u32>().unwrap_or(0);
if valid_short_date(num(&c[1]), num(&c[2]), num(&c[3])) && checks::luhn(n) {
findings.insert(n);
}
}
}
#[cfg(test)]
mod tests {
use crate::mailflow::detectors::by_id;
#[test]
fn south_africa() {
let detector = by_id("za-id").unwrap();
assert_eq!(detector.count("ID 8001015009087"), 1);
assert_eq!(detector.count("8001015009088"), 0);
assert_eq!(detector.count("8013015009087"), 0);
}
}
@@ -1,172 +0,0 @@
/*
* SPDX-FileCopyrightText: 2026 Coffey Labs LLC
*
* SPDX-License-Identifier: AGPL-3.0-only
*/
//! Identifiers from the Americas outside the US and Canada (§2.3): Brazil's
//! CPF and CNPJ, and Mexico's CURP.
use super::{Detector, Findings, Region, Strength, digit_values, valid_short_date, word_near};
use regex::Regex;
use std::sync::LazyLock;
pub static DETECTORS: &[Detector] = &[
Detector::new(
"br-cpf",
"Brazil: CPF",
Region::Americas,
Strength::Checked,
br_cpf,
),
Detector::new(
"br-cnpj",
"Brazil: CNPJ",
Region::Americas,
Strength::Checked,
br_cnpj,
),
Detector::new(
"mx-curp",
"Mexico: CURP",
Region::Americas,
Strength::Checked,
mx_curp,
),
];
fn re(pattern: &str) -> Regex {
Regex::new(pattern).expect("detector pattern")
}
/// Brazil's mod 11 check digit over `digits` with `weights`.
fn br_check(digits: &[u32], weights: &[u32]) -> u32 {
match digits.iter().zip(weights).map(|(a, w)| a * w).sum::<u32>() % 11 {
0 | 1 => 0,
r => 11 - r,
}
}
/// `111.444.777-35`, or eleven bare digits.
static CPF: LazyLock<Regex> = LazyLock::new(|| re(r"\b\d{3}(\.?)\d{3}(\.?)\d{3}(-?)\d{2}\b"));
pub fn cpf_valid(n: &str) -> bool {
let d = digit_values(n);
// A run of one digit passes the arithmetic but is never issued
d.len() == 11
&& d.iter().any(|x| *x != d[0])
&& br_check(&d[..9], &[10, 9, 8, 7, 6, 5, 4, 3, 2]) == d[9]
&& br_check(&d[..10], &[11, 10, 9, 8, 7, 6, 5, 4, 3, 2]) == d[10]
}
const CPF_WORDS: &[&str] = &[
"cpf",
"cadastro de pessoas físicas",
"cadastro de pessoa física",
];
fn br_cpf(text: &str, findings: &mut Findings) {
for c in CPF.captures_iter(text) {
let whole = c.get(0).unwrap();
let written = &c[1] == "." && &c[2] == "." && &c[3] == "-";
let n: String = whole
.as_str()
.chars()
.filter(char::is_ascii_digit)
.collect();
if cpf_valid(&n) && (written || word_near(text, whole.start(), whole.end(), CPF_WORDS)) {
findings.insert(n);
}
}
}
/// `11.222.333/0001-81`, or fourteen bare digits.
static CNPJ: LazyLock<Regex> =
LazyLock::new(|| re(r"\b\d{2}(\.?)\d{3}(\.?)\d{3}(/?)\d{4}(-?)\d{2}\b"));
pub fn cnpj_valid(n: &str) -> bool {
let d = digit_values(n);
d.len() == 14
&& d.iter().any(|x| *x != d[0])
&& br_check(&d[..12], &[5, 4, 3, 2, 9, 8, 7, 6, 5, 4, 3, 2]) == d[12]
&& br_check(&d[..13], &[6, 5, 4, 3, 2, 9, 8, 7, 6, 5, 4, 3, 2]) == d[13]
}
const CNPJ_WORDS: &[&str] = &["cnpj", "cadastro nacional da pessoa jurídica"];
fn br_cnpj(text: &str, findings: &mut Findings) {
for c in CNPJ.captures_iter(text) {
let whole = c.get(0).unwrap();
let written = &c[1] == "." && &c[2] == "." && &c[3] == "/" && &c[4] == "-";
let n: String = whole
.as_str()
.chars()
.filter(char::is_ascii_digit)
.collect();
if cnpj_valid(&n) && (written || word_near(text, whole.start(), whole.end(), CNPJ_WORDS)) {
findings.insert(n);
}
}
}
/// Four letters, the birth date, sex (H, M or X), the state, three
/// consonants, a character that tells the century apart, the check digit.
static CURP: LazyLock<Regex> = LazyLock::new(|| {
re(r"(?i)\b[A-Z]{4}(\d{2})(\d{2})(\d{2})[HMX][A-Z]{2}[B-DF-HJ-NP-TV-Z]{3}[A-Z0-9]\d\b")
});
/// RENAPO's check: each character's place in `0-9 A-N Ñ O-Z`, weighted 18
/// down to 2; the digit is 10 minus the sum mod 10 (10 becomes 0).
pub fn curp_valid(curp: &str) -> bool {
const ALPHABET: &str = "0123456789ABCDEFGHIJKLMNÑOPQRSTUVWXYZ";
let mut sum = 0u32;
for (i, c) in curp.chars().take(17).enumerate() {
let Some(value) = ALPHABET.chars().position(|a| a == c) else {
return false;
};
sum += value as u32 * (18 - i as u32);
}
curp.chars().nth(17).and_then(|c| c.to_digit(10)) == Some((10 - sum % 10) % 10)
}
fn mx_curp(text: &str, findings: &mut Findings) {
for c in CURP.captures_iter(text) {
let curp = c[0].to_ascii_uppercase();
if valid_short_date(num(&c[1]), num(&c[2]), num(&c[3])) && curp_valid(&curp) {
findings.insert(curp);
}
}
}
fn num(s: &str) -> u32 {
s.parse().unwrap_or(0)
}
#[cfg(test)]
mod tests {
use crate::mailflow::detectors::by_id;
fn count(id: &str, text: &str) -> usize {
by_id(id).unwrap().count(text)
}
#[test]
fn brazil() {
assert_eq!(count("br-cpf", "CPF 111.444.777-35"), 1);
assert_eq!(count("br-cpf", "111.444.777-36"), 0);
assert_eq!(count("br-cpf", "pedido 11144477735"), 0);
assert_eq!(count("br-cpf", "cpf: 11144477735"), 1);
assert_eq!(count("br-cpf", "CPF 111.111.111-11"), 0);
assert_eq!(count("br-cnpj", "11.222.333/0001-81"), 1);
assert_eq!(count("br-cnpj", "11.222.333/0001-82"), 0);
assert_eq!(count("br-cnpj", "CNPJ 11222333000181"), 1);
}
#[test]
fn mexico() {
// python-stdnum's documented example
assert_eq!(count("mx-curp", "CURP BOXW310820HNERXN09"), 1);
assert_eq!(count("mx-curp", "BOXW310820HNERXN08"), 0);
assert_eq!(count("mx-curp", "BOXW311320HNERXN09"), 0);
}
}
@@ -1,511 +0,0 @@
/*
* SPDX-FileCopyrightText: 2026 Coffey Labs LLC
*
* SPDX-License-Identifier: AGPL-3.0-only
*/
//! Detectors that aren't tied to one country (§2.3, region "Any").
use super::{
Detector, Findings, Region, Strength, checks, digits, stands_alone, valid_date, word_near,
};
use regex::Regex;
use std::sync::LazyLock;
pub static DETECTORS: &[Detector] = &[
Detector::new(
"payment-card",
"Payment card number",
Region::Any,
Strength::Checked,
payment_card,
),
Detector::new("iban", "IBAN", Region::Any, Strength::Checked, iban),
Detector::new(
"swift-bic",
"SWIFT/BIC code",
Region::Any,
Strength::NeedsWord,
swift_bic,
),
Detector::new(
"email-addresses",
"Email addresses",
Region::Any,
Strength::Checked,
email_addresses,
),
Detector::new(
"phone-numbers",
"Phone numbers",
Region::Any,
Strength::NeedsWord,
phone_numbers,
),
Detector::new(
"date-of-birth",
"Date of birth",
Region::Any,
Strength::NeedsWord,
date_of_birth,
),
Detector::new(
"passport",
"Passport number",
Region::Any,
Strength::NeedsWord,
passport,
),
Detector::new(
"private-key",
"Private key",
Region::Any,
Strength::Checked,
private_key,
),
Detector::new(
"credentials",
"Cloud and service credentials",
Region::Any,
Strength::Checked,
credentials,
),
];
fn re(pattern: &str) -> Regex {
Regex::new(pattern).expect("detector pattern")
}
// --- Payment cards --------------------------------------------------------
/// Issuer prefixes (ISO/IEC 7812 IINs) and the lengths each network issues.
fn card_network(number: &str) -> bool {
let len = number.len();
let prefix = |n: usize| number[..n].parse::<u32>().unwrap_or(0);
match number.as_bytes()[0] {
// Visa
b'4' => matches!(len, 13 | 16 | 19),
b'5' => {
// Mastercard 51–55; Maestro 50, 56–58
(51..=55).contains(&prefix(2)) && len == 16
|| matches!(prefix(2), 50 | 56..=58) && (12..=19).contains(&len)
}
// Mastercard 2221–2720
b'2' => (2221..=2720).contains(&prefix(4)) && len == 16,
b'3' => {
// American Express 34, 37; JCB 3528–3589; Diners 300–305, 36, 38, 39
matches!(prefix(2), 34 | 37) && len == 15
|| (3528..=3589).contains(&prefix(4)) && (16..=19).contains(&len)
|| ((300..=305).contains(&prefix(3)) || matches!(prefix(2), 36 | 38 | 39))
&& (14..=19).contains(&len)
}
// Discover 6011, 644–649, 65; UnionPay 62; Maestro 6x
b'6' => (12..=19).contains(&len),
_ => false,
}
}
fn is_card(number: &str) -> bool {
(12..=19).contains(&number.len()) && card_network(number) && checks::luhn(number)
}
static CARD: LazyLock<Regex> = LazyLock::new(|| re(r"\b\d(?:[ -]?\d){11,18}\b"));
fn payment_card(text: &str, findings: &mut Findings) {
for m in CARD.find_iter(text) {
if !stands_alone(text, m.start(), m.end()) {
continue;
}
let whole = digits(m.as_str());
if is_card(&whole) {
findings.insert(whole);
continue;
}
// Two numbers side by side ("4242 4242 4242 4242 2031"): try each
// run of whole groups
let groups: Vec<String> = m.as_str().split([' ', '-']).map(digits).collect();
'runs: for from in 0..groups.len() {
let mut number = String::new();
for group in &groups[from..] {
number.push_str(group);
if is_card(&number) {
findings.insert(number);
break 'runs;
}
}
}
}
}
// --- IBAN -----------------------------------------------------------------
static IBAN: LazyLock<Regex> =
LazyLock::new(|| re(r"\b[A-Za-z]{2}\d{2}(?:[ ]?[A-Za-z0-9]){11,30}"));
fn iban(text: &str, findings: &mut Findings) {
// The pattern can run on into the next words, even the next IBAN: after
// each hit, look again from where that IBAN ended
let mut from = 0;
while let Some(m) = IBAN.find_at(text, from) {
from = m.start() + 1;
let compact = m.as_str().replace(' ', "").to_ascii_uppercase();
let Some(len) = checks::iban_length(&compact[..2]) else {
continue;
};
if compact.len() < len {
continue;
}
// Where the country's length ends in the text, spaces counted
let mut seen = 0;
let Some(end) = m
.as_str()
.char_indices()
.find(|(_, c)| {
if *c != ' ' {
seen += 1;
}
seen == len
})
.map(|(i, c)| m.start() + i + c.len_utf8())
else {
continue;
};
let candidate = &compact[..len];
if stands_alone(text, m.start(), end) && checks::iban(candidate) {
findings.insert(candidate);
from = end;
}
}
}
// --- SWIFT/BIC ------------------------------------------------------------
static BIC: LazyLock<Regex> =
LazyLock::new(|| re(r"\b[A-Z]{4}[A-Z]{2}[A-Z0-9]{2}(?:[A-Z0-9]{3})?\b"));
const BIC_WORDS: &[&str] = &[
"swift",
"bic",
"swift/bic",
"bank",
"banque",
"bankverbindung",
];
fn swift_bic(text: &str, findings: &mut Findings) {
for m in BIC.find_iter(text) {
let code = m.as_str();
if checks::is_country(&code[4..6]) && word_near(text, m.start(), m.end(), BIC_WORDS) {
findings.insert(code);
}
}
}
// --- Contact lists --------------------------------------------------------
static EMAIL: LazyLock<Regex> =
LazyLock::new(|| re(r"(?i)\b[a-z0-9._%+-]+@[a-z0-9-]+(?:\.[a-z0-9-]+)*\.[a-z]{2,}\b"));
fn email_addresses(text: &str, findings: &mut Findings) {
for m in EMAIL.find_iter(text) {
findings.insert(m.as_str().to_lowercase());
}
}
/// International form: found alone. National form: only with a word.
static PHONE_INTL: LazyLock<Regex> = LazyLock::new(|| re(r"\+\d{1,3}(?:[ .-]?\(?\d{1,4}\)?){2,5}"));
static PHONE_NATIONAL: LazyLock<Regex> =
LazyLock::new(|| re(r"\(?\d{2,4}\)?[ .-]\d{3,4}[ .-]\d{3,4}"));
const PHONE_WORDS: &[&str] = &[
"phone",
"tel",
"telephone",
"mobile",
"cell",
"fax",
"telefon",
"téléphone",
"teléfono",
"telefono",
"handy",
"portable",
"móvil",
"cellulare",
"mobiel",
];
fn phone_numbers(text: &str, findings: &mut Findings) {
let mut international = Vec::new();
for m in PHONE_INTL.find_iter(text) {
let number = digits(m.as_str());
if (8..=15).contains(&number.len()) && stands_alone(text, m.start() + 1, m.end()) {
findings.insert(number);
international.push(m.range());
}
}
for m in PHONE_NATIONAL.find_iter(text) {
let number = digits(m.as_str());
// Not the tail of an international number already counted
if international.iter().any(|r| r.contains(&m.start())) {
continue;
}
if (9..=11).contains(&number.len())
&& stands_alone(text, m.start(), m.end())
&& !text[..m.start()].ends_with('+')
&& word_near(text, m.start(), m.end(), PHONE_WORDS)
{
findings.insert(number);
}
}
}
// --- Date of birth --------------------------------------------------------
static DATE_ISO: LazyLock<Regex> = LazyLock::new(|| re(r"\b(\d{4})-(\d{2})-(\d{2})\b"));
static DATE_NUMERIC: LazyLock<Regex> =
LazyLock::new(|| re(r"\b(\d{1,2})[./-](\d{1,2})[./-](\d{4})\b"));
static DATE_WORDS: LazyLock<Regex> = LazyLock::new(|| {
re(
r"(?i)\b(?:(\d{1,2})\s+(jan|feb|mar|apr|may|jun|jul|aug|sep|oct|nov|dec)[a-z]*\.?,?\s+(\d{4})|(jan|feb|mar|apr|may|jun|jul|aug|sep|oct|nov|dec)[a-z]*\.?\s+(\d{1,2}),?\s+(\d{4}))\b",
)
});
const BIRTH_WORDS: &[&str] = &[
"born",
"birth",
"dob",
"d.o.b",
"birthday",
"birthdate",
"geburtsdatum",
"geboren",
"naissance",
"né le",
"née le",
"nacimiento",
"nacido",
"nacida",
"nascita",
"nato il",
"nata il",
"geboortedatum",
"födelsedatum",
"fødselsdato",
"syntymäaika",
"urodzenia",
"nascimento",
];
fn month_number(name: &str) -> u32 {
const MONTHS: [&str; 12] = [
"jan", "feb", "mar", "apr", "may", "jun", "jul", "aug", "sep", "oct", "nov", "dec",
];
let name = name.to_lowercase();
MONTHS
.iter()
.position(|m| *m == name)
.map_or(0, |i| i as u32 + 1)
}
fn date_of_birth(text: &str, findings: &mut Findings) {
let mut add = |start: usize, end: usize, key: String| {
if word_near(text, start, end, BIRTH_WORDS) {
findings.insert(key);
}
};
let num = |s: &str| s.parse::<u32>().unwrap_or(0);
for c in DATE_ISO.captures_iter(text) {
let (y, m, d) = (num(&c[1]), num(&c[2]), num(&c[3]));
let whole = c.get(0).unwrap();
if valid_date(y, m, d) {
add(whole.start(), whole.end(), format!("{y:04}{m:02}{d:02}"));
}
}
for c in DATE_NUMERIC.captures_iter(text) {
let (a, b, y) = (num(&c[1]), num(&c[2]), num(&c[3]));
let whole = c.get(0).unwrap();
// Day first or month first: either reading that is a real date
if valid_date(y, b, a) || valid_date(y, a, b) {
add(whole.start(), whole.end(), whole.as_str().to_string());
}
}
for c in DATE_WORDS.captures_iter(text) {
let whole = c.get(0).unwrap();
let (d, m, y) = match (c.get(1), c.get(4)) {
(Some(d), _) => (num(d.as_str()), month_number(&c[2]), num(&c[3])),
(_, Some(m)) => (num(&c[5]), month_number(m.as_str()), num(&c[6])),
_ => continue,
};
if valid_date(y, m, d) {
add(whole.start(), whole.end(), format!("{y:04}{m:02}{d:02}"));
}
}
}
// --- Passport -------------------------------------------------------------
static PASSPORT: LazyLock<Regex> = LazyLock::new(|| re(r"\b[A-Z0-9]{6,9}\b"));
const PASSPORT_WORDS: &[&str] = &[
"passport",
"passeport",
"reisepass",
"pasaporte",
"passaporto",
"paspoort",
"passnummer",
"pass-nr",
"passport no",
"pasaporte n.º",
"passaporte",
];
fn passport(text: &str, findings: &mut Findings) {
for m in PASSPORT.find_iter(text) {
let value = m.as_str();
if value.bytes().filter(u8::is_ascii_digit).count() >= 5
&& word_near(text, m.start(), m.end(), PASSPORT_WORDS)
{
findings.insert(value);
}
}
}
// --- Keys and credentials -------------------------------------------------
static PRIVATE_KEY: LazyLock<Regex> = LazyLock::new(|| {
re(
r"-----BEGIN (?:(?:RSA|EC|DSA|OPENSSH|ENCRYPTED|PGP) )?PRIVATE KEY(?: BLOCK)?-----\s*([A-Za-z0-9+/=:\s-]{0,64})",
)
});
fn private_key(text: &str, findings: &mut Findings) {
for c in PRIVATE_KEY.captures_iter(text) {
// Each key once, by the start of its body
let body: String = c[1].chars().filter(|c| !c.is_whitespace()).collect();
let whole = c.get(0).unwrap();
findings.insert(if body.is_empty() {
format!("@{}", whole.start())
} else {
body
});
}
}
/// Published token formats: AWS access key IDs, GitHub tokens, Slack
/// tokens, Stripe live secret and restricted keys, Google API keys.
static CREDENTIAL: LazyLock<Regex> = LazyLock::new(|| {
re(concat!(
r"\b(?:",
r"(?:AKIA|ASIA|ABIA|ACCA)[A-Z0-9]{16}",
r"|gh[pousr]_[A-Za-z0-9]{36}",
r"|github_pat_[A-Za-z0-9_]{82}",
r"|xox[abposr]-[A-Za-z0-9-]{10,72}",
r"|(?:sk|rk)_live_[A-Za-z0-9]{24,99}",
r"|AIza[0-9A-Za-z_-]{35}",
r")\b"
))
});
fn credentials(text: &str, findings: &mut Findings) {
for m in CREDENTIAL.find_iter(text) {
findings.insert(m.as_str());
}
}
#[cfg(test)]
mod tests {
use crate::mailflow::detectors::by_id;
fn count(id: &str, text: &str) -> usize {
by_id(id).unwrap().count(text)
}
#[test]
fn payment_cards() {
// Networks' and processors' published test numbers
let text = "Visa 4242 4242 4242 4242, MC 5555-5555-5555-4444, Amex 378282246310005, \
Discover 6011111111111117, JCB 3566002020360505, Diners 30569309025904, \
UnionPay 6200000000000005, Mastercard 2-series 2223003122003222";
assert_eq!(count("payment-card", text), 8);
// Luhn fails, wrong network length, inside a longer number
assert_eq!(count("payment-card", "4242424242424241"), 0);
assert_eq!(count("payment-card", "378282246310005 0"), 1);
assert_eq!(count("payment-card", "order 94242424242424242 shipped"), 0);
// The same number twice counts once
assert_eq!(
count("payment-card", "4242424242424242 and 4242-4242-4242-4242"),
1
);
// A card followed by a year
assert_eq!(count("payment-card", "card 4242 4242 4242 4242 2031"), 1);
}
#[test]
fn ibans() {
let text =
"Pay GB29 NWBK 6016 1331 9268 19 or de89370400440532013000 (NL91ABNA0417164300).";
assert_eq!(count("iban", text), 3);
assert_eq!(count("iban", "GB29 NWBK 6016 1331 9268 18"), 0);
// Runs into the next word: still found at the country's length
assert_eq!(count("iban", "IBAN NL91ABNA0417164300 BIC ABNANL2A"), 1);
}
#[test]
fn swift_codes_need_a_word() {
assert_eq!(count("swift-bic", "SWIFT: DEUTDEFF500"), 1);
assert_eq!(count("swift-bic", "BIC NWBKGB2L"), 1);
assert_eq!(count("swift-bic", "HAPPYDAYS DEUTDEFF"), 0);
// Not a country in positions 5–6
assert_eq!(count("swift-bic", "BIC DEUTZZFF"), 0);
}
#[test]
fn email_and_phone_lists() {
let list = "[email protected], [email protected], [email protected], [email protected]";
assert_eq!(count("email-addresses", list), 3);
assert_eq!(
count("phone-numbers", "+44 20 7946 0958, +1 (415) 555-2671"),
2
);
assert_eq!(count("phone-numbers", "call 020 7946 0958"), 0);
assert_eq!(count("phone-numbers", "Tel: 020 7946 0958"), 1);
assert_eq!(count("phone-numbers", "invoice 020 7946 0958"), 0);
// One number, not also its national tail
assert_eq!(count("phone-numbers", "Tel: +44 20 7946 0958"), 1);
}
#[test]
fn dates_of_birth() {
assert_eq!(count("date-of-birth", "DOB: 1984-02-29"), 1);
assert_eq!(count("date-of-birth", "Geburtsdatum 31.12.1970"), 1);
assert_eq!(count("date-of-birth", "born on March 3, 1962"), 1);
assert_eq!(count("date-of-birth", "date of birth 3 Mar 1962"), 1);
// Not a real date, no word, a meeting
assert_eq!(count("date-of-birth", "DOB: 1985-02-29"), 0);
assert_eq!(count("date-of-birth", "invoice 1984-02-29"), 0);
assert_eq!(count("date-of-birth", "Meeting on 12/05/2026"), 0);
}
#[test]
fn passports_need_a_word() {
assert_eq!(count("passport", "Passport number: 533380006"), 1);
assert_eq!(count("passport", "Reisepass C01X00T47"), 1);
assert_eq!(count("passport", "Order 533380006 shipped"), 0);
// Mostly letters: a word, not a number
assert_eq!(count("passport", "passport PASSWORD"), 0);
}
#[test]
fn keys_and_credentials() {
let key = "-----BEGIN OPENSSH PRIVATE KEY-----\nb3BlbnNzaC1rZXktdjEAAAAABG5vbmUAAAAEbm9uZQ\n-----END OPENSSH PRIVATE KEY-----";
assert_eq!(count("private-key", key), 1);
assert_eq!(count("private-key", "-----BEGIN PUBLIC KEY-----\nMFkw"), 0);
// Documentation examples of each format
let tokens = "AKIAIOSFODNN7EXAMPLE ghp_0123456789abcdefghijklmnopqrstuvwxyz \
AIzaSyA-0123456789abcdefghijklmnopqrstu";
assert_eq!(count("credentials", tokens), 3);
assert_eq!(count("credentials", "AKIA123 ghp_short"), 0);
}
}
@@ -1,281 +0,0 @@
/*
* SPDX-FileCopyrightText: 2026 Coffey Labs LLC
*
* SPDX-License-Identifier: AGPL-3.0-only
*/
//! Asian identifiers (§2.3): India's Aadhaar and PAN, China's resident ID,
//! Japan's My Number, Singapore's NRIC and FIN, and South Korea's resident
//! registration number.
use super::{
Detector, Findings, Region, Strength, digit_values, valid_date, valid_short_date, word_near,
};
use regex::Regex;
use std::sync::LazyLock;
pub static DETECTORS: &[Detector] = &[
Detector::new(
"in-aadhaar",
"India: Aadhaar",
Region::Asia,
Strength::Checked,
in_aadhaar,
),
Detector::new(
"in-pan",
"India: PAN",
Region::Asia,
Strength::NeedsWord,
in_pan,
),
Detector::new(
"cn-resident-id",
"China: resident ID",
Region::Asia,
Strength::Checked,
cn_resident_id,
),
Detector::new(
"jp-my-number",
"Japan: My Number",
Region::Asia,
Strength::Checked,
jp_my_number,
),
Detector::new(
"sg-nric",
"Singapore: NRIC and FIN",
Region::Asia,
Strength::Checked,
sg_nric,
),
Detector::new(
"kr-rrn",
"South Korea: resident registration number",
Region::Asia,
Strength::NeedsWord,
kr_rrn,
),
];
fn re(pattern: &str) -> Regex {
Regex::new(pattern).expect("detector pattern")
}
/// Twelve digits written in fours, or bare.
static TWELVE: LazyLock<Regex> = LazyLock::new(|| re(r"\b(\d{4})( ?)(\d{4})( ?)(\d{4})\b"));
const VERHOEFF_D: [[u8; 10]; 10] = [
[0, 1, 2, 3, 4, 5, 6, 7, 8, 9],
[1, 2, 3, 4, 0, 6, 7, 8, 9, 5],
[2, 3, 4, 0, 1, 7, 8, 9, 5, 6],
[3, 4, 0, 1, 2, 8, 9, 5, 6, 7],
[4, 0, 1, 2, 3, 9, 5, 6, 7, 8],
[5, 9, 8, 7, 6, 0, 4, 3, 2, 1],
[6, 5, 9, 8, 7, 1, 0, 4, 3, 2],
[7, 6, 5, 9, 8, 2, 1, 0, 4, 3],
[8, 7, 6, 5, 9, 3, 2, 1, 0, 4],
[9, 8, 7, 6, 5, 4, 3, 2, 1, 0],
];
const VERHOEFF_P: [[u8; 10]; 8] = [
[0, 1, 2, 3, 4, 5, 6, 7, 8, 9],
[1, 5, 7, 6, 2, 8, 3, 0, 9, 4],
[5, 8, 0, 3, 7, 9, 6, 1, 4, 2],
[8, 9, 1, 6, 0, 4, 3, 5, 2, 7],
[9, 4, 5, 3, 1, 2, 6, 8, 7, 0],
[4, 2, 8, 6, 5, 7, 3, 9, 0, 1],
[2, 7, 9, 3, 8, 0, 6, 4, 1, 5],
[7, 0, 4, 6, 9, 1, 3, 2, 5, 8],
];
/// The Verhoeff check (dihedral group D5).
pub fn verhoeff(n: &str) -> bool {
let mut c = 0u8;
for (i, b) in n.bytes().rev().enumerate() {
c = VERHOEFF_D[c as usize][VERHOEFF_P[i % 8][(b - b'0') as usize] as usize];
}
c == 0
}
const AADHAAR_WORDS: &[&str] = &["aadhaar", "aadhar", "uidai", "uid"];
fn in_aadhaar(text: &str, findings: &mut Findings) {
for c in TWELVE.captures_iter(text) {
let whole = c.get(0).unwrap();
let n = format!("{}{}{}", &c[1], &c[3], &c[5]);
let written = &c[2] == " " && &c[4] == " ";
// Never starts with 0 or 1
if !n.starts_with(['0', '1'])
&& verhoeff(&n)
&& (written || word_near(text, whole.start(), whole.end(), AADHAAR_WORDS))
{
findings.insert(n);
}
}
}
/// Five letters (the fourth names the holder's type), four digits, a letter.
static PAN: LazyLock<Regex> = LazyLock::new(|| re(r"\b[A-Z]{3}[ABCFGHLJPTK][A-Z]\d{4}[A-Z]\b"));
const PAN_WORDS: &[&str] = &["pan", "pan card", "permanent account number", "income tax"];
fn in_pan(text: &str, findings: &mut Findings) {
for m in PAN.find_iter(text) {
if word_near(text, m.start(), m.end(), PAN_WORDS) {
findings.insert(m.as_str());
}
}
}
/// Region, birth date `YYYYMMDD`, sequence, then the ISO 7064 MOD 11-2
/// check (0–9 or X).
static CN_ID: LazyLock<Regex> =
LazyLock::new(|| re(r"(?i)\b[1-8]\d{5}(\d{4})(\d{2})(\d{2})\d{3}[\dX]\b"));
pub fn cn_id_valid(id: &str) -> bool {
const WEIGHTS: [u32; 17] = [7, 9, 10, 5, 8, 4, 2, 1, 6, 3, 7, 9, 10, 5, 8, 4, 2];
const CHECKS: &[u8] = b"10X98765432";
let sum: u32 = digit_values(&id[..17])
.iter()
.zip(WEIGHTS)
.map(|(a, w)| a * w)
.sum();
CHECKS[(sum % 11) as usize] == id.as_bytes()[17].to_ascii_uppercase()
}
fn cn_resident_id(text: &str, findings: &mut Findings) {
for c in CN_ID.captures_iter(text) {
let id = c[0].to_ascii_uppercase();
let (y, m, d) = (num(&c[1]), num(&c[2]), num(&c[3]));
if valid_date(y, m, d) && cn_id_valid(&id) {
findings.insert(id);
}
}
}
/// My Number: weights 2–7 then 2–6 from the right; a remainder of 0 or 1
/// gives 0, else 11 minus it.
pub fn my_number_valid(n: &str) -> bool {
let d = digit_values(n);
let sum: u32 = (1..=11)
.map(|i| d[11 - i] * if i <= 6 { i as u32 + 1 } else { i as u32 - 5 })
.sum();
let check = match sum % 11 {
0 | 1 => 0,
r => 11 - r,
};
check == d[11]
}
const MY_NUMBER_WORDS: &[&str] = &[
"my number",
"mynumber",
"マイナンバー",
"個人番号",
"kojin bango",
];
fn jp_my_number(text: &str, findings: &mut Findings) {
for c in TWELVE.captures_iter(text) {
let whole = c.get(0).unwrap();
let n = format!("{}{}{}", &c[1], &c[3], &c[5]);
let written = &c[2] == " " && &c[4] == " ";
if my_number_valid(&n)
&& (written || word_near(text, whole.start(), whole.end(), MY_NUMBER_WORDS))
{
findings.insert(n);
}
}
}
static NRIC: LazyLock<Regex> = LazyLock::new(|| re(r"(?i)\b([STFGM])(\d{7})([A-Z])\b"));
/// Weights 2, 7, 6, 5, 4, 3, 2; T and G add 4, M adds 3; each series has its
/// own table of check letters.
fn nric_valid(prefix: u8, digits: &str, check: u8) -> bool {
let sum: u32 = digit_values(digits)
.iter()
.zip([2, 7, 6, 5, 4, 3, 2])
.map(|(a, w)| a * w)
.sum::<u32>()
+ match prefix {
b'T' | b'G' => 4,
b'M' => 3,
_ => 0,
};
let table: &[u8] = match prefix {
b'S' | b'T' => b"JZIHGFEDCBA",
b'F' | b'G' => b"XWUTRQPNMLK",
_ => b"KLJNPQRTUWX",
};
table[(sum % 11) as usize] == check
}
fn sg_nric(text: &str, findings: &mut Findings) {
for c in NRIC.captures_iter(text) {
let id = c[0].to_ascii_uppercase();
let bytes = id.as_bytes();
if nric_valid(bytes[0], &c[2], bytes[8]) {
findings.insert(id);
}
}
}
/// `YYMMDD-GNNNNNN`, the seventh digit giving sex and century.
static RRN: LazyLock<Regex> = LazyLock::new(|| re(r"\b(\d{2})(\d{2})(\d{2})-?([1-8])\d{6}\b"));
const RRN_WORDS: &[&str] = &["주민등록번호", "주민번호", "resident registration", "rrn"];
fn kr_rrn(text: &str, findings: &mut Findings) {
for c in RRN.captures_iter(text) {
let whole = c.get(0).unwrap();
if valid_short_date(num(&c[1]), num(&c[2]), num(&c[3]))
&& word_near(text, whole.start(), whole.end(), RRN_WORDS)
{
findings.insert(whole.as_str().replace('-', ""));
}
}
}
fn num(s: &str) -> u32 {
s.parse().unwrap_or(0)
}
#[cfg(test)]
mod tests {
use crate::mailflow::detectors::by_id;
fn count(id: &str, text: &str) -> usize {
by_id(id).unwrap().count(text)
}
#[test]
fn india() {
assert_eq!(count("in-aadhaar", "2345 6789 0124"), 1);
assert_eq!(count("in-aadhaar", "2345 6789 0125"), 0);
assert_eq!(count("in-aadhaar", "order 234567890124"), 0);
assert_eq!(count("in-aadhaar", "Aadhaar 234567890124"), 1);
assert_eq!(count("in-pan", "PAN: ABCPE1234F"), 1);
assert_eq!(count("in-pan", "ABCPE1234F"), 0);
}
#[test]
fn china_japan() {
assert_eq!(count("cn-resident-id", "11010519491231002X"), 1);
assert_eq!(count("cn-resident-id", "110105194912310021"), 0);
assert_eq!(count("cn-resident-id", "11010519491331002X"), 0);
assert_eq!(count("jp-my-number", "1234 5678 9018"), 1);
assert_eq!(count("jp-my-number", "1234 5678 9017"), 0);
assert_eq!(count("jp-my-number", "マイナンバー 123456789018"), 1);
}
#[test]
fn singapore_korea() {
assert_eq!(count("sg-nric", "S1234567D and T1234567J"), 2);
assert_eq!(count("sg-nric", "S1234567E"), 0);
assert_eq!(count("kr-rrn", "주민등록번호 800101-1234567"), 1);
assert_eq!(count("kr-rrn", "800101-1234567"), 0);
assert_eq!(count("kr-rrn", "RRN 801301-1234567"), 0);
}
}
@@ -1,128 +0,0 @@
/*
* SPDX-FileCopyrightText: 2026 Coffey Labs LLC
*
* SPDX-License-Identifier: AGPL-3.0-only
*/
//! Australian identifiers (§2.3): the ATO's Tax File Number and the Medicare
//! card number.
use super::{Detector, Findings, Region, Strength, word_near};
use regex::Regex;
use std::sync::LazyLock;
pub static DETECTORS: &[Detector] = &[
Detector::new(
"au-tfn",
"Australian Tax File Number",
Region::Australia,
Strength::Checked,
tfn,
),
Detector::new(
"au-medicare",
"Australian Medicare number",
Region::Australia,
Strength::Checked,
medicare,
),
];
fn re(pattern: &str) -> Regex {
Regex::new(pattern).expect("detector pattern")
}
/// `NNN NNN NNN` stands alone; bare digits (eight or nine) need a word.
static TFN: LazyLock<Regex> = LazyLock::new(|| re(r"\b(\d{3})( ?)(\d{3})( ?)(\d{2,3})\b"));
/// Weighted sum mod 11, with the ATO's weights for 9- and 8-digit numbers.
pub fn tfn_valid(n: &str) -> bool {
let weights: &[u32] = match n.len() {
9 => &[1, 4, 3, 7, 5, 8, 6, 9, 10],
8 => &[10, 7, 8, 4, 6, 3, 5, 1],
_ => return false,
};
n.bytes()
.zip(weights)
.map(|(b, w)| u32::from(b - b'0') * w)
.sum::<u32>()
% 11
== 0
}
const TFN_WORDS: &[&str] = &["tfn", "tax file number", "tax file no"];
fn tfn(text: &str, findings: &mut Findings) {
for c in TFN.captures_iter(text) {
let whole = c.get(0).unwrap();
let n = format!("{}{}{}", &c[1], &c[3], &c[5]);
let written = n.len() == 9 && c[2] == *" " && c[4] == *" ";
if tfn_valid(&n) && (written || word_near(text, whole.start(), whole.end(), TFN_WORDS)) {
findings.insert(n);
}
}
}
/// `NNNN NNNNN N` (and an optional issue number) stands alone; bare digits
/// need a word.
static MEDICARE: LazyLock<Regex> =
LazyLock::new(|| re(r"\b([2-6]\d{3})( ?)(\d{5})( ?)(\d)(?:[ -]?\d)?\b"));
/// The ninth digit is the weighted sum (1, 3, 7, 9, 1, 3, 7, 9) of the first
/// eight, mod 10.
pub fn medicare_valid(n: &str) -> bool {
let d: Vec<u32> = n.bytes().map(|b| u32::from(b - b'0')).collect();
d.len() >= 9
&& d[..8]
.iter()
.zip([1, 3, 7, 9, 1, 3, 7, 9])
.map(|(a, w)| a * w)
.sum::<u32>()
% 10
== d[8]
}
const MEDICARE_WORDS: &[&str] = &[
"medicare",
"medicare card",
"medicare no",
"medicare number",
];
fn medicare(text: &str, findings: &mut Findings) {
for c in MEDICARE.captures_iter(text) {
let whole = c.get(0).unwrap();
let n = format!("{}{}{}", &c[1], &c[3], &c[5]);
let written = c[2] == *" " && c[4] == *" ";
if medicare_valid(&n)
&& (written || word_near(text, whole.start(), whole.end(), MEDICARE_WORDS))
{
findings.insert(n);
}
}
}
#[cfg(test)]
mod tests {
use crate::mailflow::detectors::by_id;
fn count(id: &str, text: &str) -> usize {
by_id(id).unwrap().count(text)
}
#[test]
fn tax_file_numbers() {
assert_eq!(count("au-tfn", "TFN 123 456 782"), 1);
assert_eq!(count("au-tfn", "123 456 789"), 0);
assert_eq!(count("au-tfn", "order 123456782"), 0);
assert_eq!(count("au-tfn", "tax file number 123456782"), 1);
}
#[test]
fn medicare_numbers() {
assert_eq!(count("au-medicare", "2123 45670 1"), 1);
assert_eq!(count("au-medicare", "2123 45671 1"), 0);
assert_eq!(count("au-medicare", "ref 2123456701"), 0);
assert_eq!(count("au-medicare", "Medicare 2123456701"), 1);
}
}
@@ -1,66 +0,0 @@
/*
* SPDX-FileCopyrightText: 2026 Coffey Labs LLC
*
* SPDX-License-Identifier: AGPL-3.0-only
*/
//! Canadian identifiers (§2.3): the Social Insurance Number.
use super::{Detector, Findings, Region, Strength, checks, word_near};
use regex::Regex;
use std::sync::LazyLock;
pub static DETECTORS: &[Detector] = &[Detector::new(
"ca-sin",
"Canadian Social Insurance Number",
Region::Canada,
Strength::Checked,
sin,
)];
/// `NNN NNN NNN` or `NNN-NNN-NNN` stands alone; nine bare digits need a word.
static SIN: LazyLock<Regex> = LazyLock::new(|| {
Regex::new(r"\b(\d{3})([ -]?)(\d{3})([ -]?)(\d{3})\b").expect("detector pattern")
});
const SIN_WORDS: &[&str] = &[
"sin",
"social insurance",
"nas",
"numéro d'assurance sociale",
"assurance sociale",
];
fn sin(text: &str, findings: &mut Findings) {
for c in SIN.captures_iter(text) {
let whole = c.get(0).unwrap();
let n = format!("{}{}{}", &c[1], &c[3], &c[5]);
let written = !c[2].is_empty() && c[2] == c[4];
// 0 and 8 are never issued as a first digit
if !n.starts_with(['0', '8'])
&& checks::luhn(&n)
&& (written || word_near(text, whole.start(), whole.end(), SIN_WORDS))
{
findings.insert(n);
}
}
}
#[cfg(test)]
mod tests {
use crate::mailflow::detectors::by_id;
fn count(text: &str) -> usize {
by_id("ca-sin").unwrap().count(text)
}
#[test]
fn social_insurance_numbers() {
assert_eq!(count("130 692 544 and 193-456-787"), 2);
assert_eq!(count("130 692 545"), 0);
// The government's printed example starts with 0, never issued
assert_eq!(count("046 454 286"), 0);
assert_eq!(count("order 130692544"), 0);
assert_eq!(count("SIN: 130692544"), 1);
}
}
@@ -1,227 +0,0 @@
/*
* SPDX-FileCopyrightText: 2026 Coffey Labs LLC
*
* SPDX-License-Identifier: AGPL-3.0-only
*/
//! Check-digit algorithms, each from its public definition.
/// The Luhn check (ISO/IEC 7812-1, Annex B) over a string of ASCII digits.
pub fn luhn(digits: &str) -> bool {
if digits.len() < 2 || !digits.bytes().all(|b| b.is_ascii_digit()) {
return false;
}
let sum: u32 = digits
.bytes()
.rev()
.enumerate()
.map(|(i, b)| {
let d = u32::from(b - b'0');
if i % 2 == 1 {
let d = d * 2;
if d > 9 { d - 9 } else { d }
} else {
d
}
})
.sum();
sum.is_multiple_of(10)
}
/// ISO 13616 IBAN lengths, by country, from the IBAN registry.
const IBAN_LENGTHS: &[(&str, usize)] = &[
("AD", 24),
("AE", 23),
("AL", 28),
("AT", 20),
("AZ", 28),
("BA", 20),
("BE", 16),
("BG", 22),
("BH", 22),
("BI", 27),
("BR", 29),
("BY", 28),
("CH", 21),
("CR", 22),
("CY", 28),
("CZ", 24),
("DE", 22),
("DJ", 27),
("DK", 18),
("DO", 28),
("EE", 20),
("EG", 29),
("ES", 24),
("FI", 18),
("FK", 18),
("FO", 18),
("FR", 27),
("GB", 22),
("GE", 22),
("GI", 23),
("GL", 18),
("GR", 27),
("GT", 28),
("HN", 28),
("HR", 21),
("HU", 28),
("IE", 22),
("IL", 23),
("IQ", 23),
("IS", 26),
("IT", 27),
("JO", 30),
("KW", 30),
("KZ", 20),
("LB", 28),
("LC", 32),
("LI", 21),
("LT", 20),
("LU", 20),
("LV", 21),
("LY", 25),
("MC", 27),
("MD", 24),
("ME", 22),
("MK", 19),
("MN", 20),
("MR", 27),
("MT", 31),
("MU", 30),
("NI", 28),
("NL", 18),
("NO", 15),
("OM", 23),
("PK", 24),
("PL", 28),
("PS", 29),
("PT", 25),
("QA", 29),
("RO", 24),
("RS", 22),
("RU", 33),
("SA", 24),
("SC", 31),
("SD", 18),
("SE", 24),
("SI", 19),
("SK", 24),
("SM", 27),
("SO", 23),
("ST", 25),
("SV", 28),
("TL", 23),
("TN", 24),
("TR", 26),
("UA", 29),
("VA", 22),
("VG", 24),
("XK", 20),
("YE", 30),
];
/// The IBAN length for a country code, if the country uses IBANs.
pub fn iban_length(country: &str) -> Option<usize> {
IBAN_LENGTHS
.iter()
.find(|(code, _)| *code == country)
.map(|(_, len)| *len)
}
/// ISO 13616 / ISO 7064 MOD 97-10 over an IBAN with no spaces, upper case:
/// move the first four characters to the end, turn letters into 10–35, and
/// the number mod 97 must be 1. Also checks the country's length.
pub fn iban(iban: &str) -> bool {
if iban.len() < 5
|| !iban
.bytes()
.all(|b| b.is_ascii_uppercase() || b.is_ascii_digit())
{
return false;
}
if iban_length(&iban[..2]) != Some(iban.len())
|| !iban[2..4].bytes().all(|b| b.is_ascii_digit())
{
return false;
}
let mut remainder: u32 = 0;
for b in iban[4..].bytes().chain(iban[..4].bytes()) {
let value = if b.is_ascii_digit() {
u32::from(b - b'0')
} else {
u32::from(b - b'A') + 10
};
remainder = if value >= 10 {
(remainder * 100 + value) % 97
} else {
(remainder * 10 + value) % 97
};
}
remainder == 1
}
/// ISO 3166-1 alpha-2 country codes, for SWIFT/BIC positions 5–6.
const COUNTRIES: &str = "AD AE AF AG AI AL AM AO AQ AR AS AT AU AW AX AZ BA BB BD BE BF BG BH BI BJ \
BL BM BN BO BQ BR BS BT BV BW BY BZ CA CC CD CF CG CH CI CK CL CM CN CO CR CU CV CW CX CY CZ DE DJ \
DK DM DO DZ EC EE EG EH ER ES ET FI FJ FK FM FO FR GA GB GD GE GF GG GH GI GL GM GN GP GQ GR GS GT \
GU GW GY HK HM HN HR HT HU ID IE IL IM IN IO IQ IR IS IT JE JM JO JP KE KG KH KI KM KN KP KR KW KY \
KZ LA LB LC LI LK LR LS LT LU LV LY MA MC MD ME MF MG MH MK ML MM MN MO MP MQ MR MS MT MU MV MW MX \
MY MZ NA NC NE NF NG NI NL NO NP NR NU NZ OM PA PE PF PG PH PK PL PM PN PR PS PT PW PY QA RE RO RS \
RU RW SA SB SC SD SE SG SH SI SJ SK SL SM SN SO SR SS ST SV SX SY SZ TC TD TF TG TH TJ TK TL TM TN \
TO TR TT TV TW TZ UA UG UM US UY UZ VA VC VE VG VI VN VU WF WS XK YE YT ZA ZM ZW";
pub fn is_country(code: &str) -> bool {
code.len() == 2 && COUNTRIES.split(' ').any(|c| c == code)
}
#[cfg(test)]
mod tests {
use super::*;
#[test]
fn luhn_known_numbers() {
// Published test card numbers
for good in [
"4242424242424242",
"5555555555554444",
"378282246310005",
"79927398713",
] {
assert!(luhn(good), "{good}");
}
for bad in ["4242424242424241", "79927398710", "1", "12a4"] {
assert!(!luhn(bad), "{bad}");
}
}
#[test]
fn iban_registry_examples() {
// The IBAN registry's own examples
for good in [
"GB29NWBK60161331926819",
"DE89370400440532013000",
"FR1420041010050500013M02606",
"NL91ABNA0417164300",
"BE68539007547034",
"NO9386011117947",
"CH9300762011623852957",
] {
assert!(iban(good), "{good}");
}
for bad in [
"GB29NWBK60161331926818", // check fails
"GB29NWBK6016133192681", // too short for GB
"ZZ29NWBK60161331926819", // no such country
"DE8937040044053201300A", // letters where DE has none still fail mod 97
] {
assert!(!iban(bad), "{bad}");
}
}
#[test]
fn countries() {
assert!(is_country("DE") && is_country("US") && is_country("XK"));
assert!(!is_country("ZZ") && !is_country("D"));
}
}
@@ -1,646 +0,0 @@
/*
* SPDX-FileCopyrightText: 2026 Coffey Labs LLC
*
* SPDX-License-Identifier: AGPL-3.0-only
*/
//! European Union national identifiers (§2.3), each from its issuer's
//! published rules. An identifier that is only digits and whose check a
//! random number passes often (mod 10, mod 11) counts alone only in its
//! written form, and as bare digits only beside a word.
use super::{
Detector, Findings, Region, Strength, checks, digit_values, stands_alone, valid_short_date,
word_near,
};
use regex::Regex;
use std::sync::LazyLock;
pub static DETECTORS: &[Detector] = &[
Detector::new(
"de-tax-id",
"Germany: tax ID (Steuer-ID)",
Region::Eu,
Strength::Checked,
de_tax_id,
),
Detector::new(
"de-id-card",
"Germany: ID card number",
Region::Eu,
Strength::Checked,
de_id_card,
),
Detector::new(
"fr-nir",
"France: social security number (NIR)",
Region::Eu,
Strength::Checked,
fr_nir,
),
Detector::new(
"es-dni-nie",
"Spain: DNI and NIE",
Region::Eu,
Strength::Checked,
es_dni_nie,
),
Detector::new(
"it-codice-fiscale",
"Italy: codice fiscale",
Region::Eu,
Strength::Checked,
it_codice_fiscale,
),
Detector::new(
"nl-bsn",
"Netherlands: BSN",
Region::Eu,
Strength::Checked,
nl_bsn,
),
Detector::new(
"be-national-number",
"Belgium: national number",
Region::Eu,
Strength::Checked,
be_national_number,
),
Detector::new(
"pl-pesel",
"Poland: PESEL",
Region::Eu,
Strength::Checked,
pl_pesel,
),
Detector::new(
"se-personnummer",
"Sweden: personnummer",
Region::Eu,
Strength::Checked,
se_personnummer,
),
Detector::new(
"dk-cpr",
"Denmark: CPR number",
Region::Eu,
Strength::NeedsWord,
dk_cpr,
),
Detector::new(
"fi-hetu",
"Finland: personal identity code",
Region::Eu,
Strength::Checked,
fi_hetu,
),
Detector::new(
"ie-pps",
"Ireland: PPS number",
Region::Eu,
Strength::Checked,
ie_pps,
),
Detector::new(
"pt-nif",
"Portugal: NIF",
Region::Eu,
Strength::Checked,
pt_nif,
),
Detector::new(
"at-svnr",
"Austria: social insurance number",
Region::Eu,
Strength::Checked,
at_svnr,
),
];
fn re(pattern: &str) -> Regex {
Regex::new(pattern).expect("detector pattern")
}
fn num(s: &str) -> u32 {
s.parse().unwrap_or(0)
}
// --- Germany --------------------------------------------------------------
/// Eleven digits, written `86 095 742 719` on the BZSt's letters.
static DE_TAX: LazyLock<Regex> = LazyLock::new(|| re(r"\b\d{2}( ?)\d{3}( ?)\d{3}( ?)\d{3}\b"));
/// ISO 7064 MOD 11,10; no leading zero; in the first ten digits one digit
/// appears two or three times and every other at most once.
pub fn de_tax_id_valid(n: &str) -> bool {
let d = digit_values(n);
if d.len() != 11 || d[0] == 0 {
return false;
}
let mut counts = [0u8; 10];
for &x in &d[..10] {
counts[x as usize] += 1;
}
let repeated = counts.iter().filter(|&&c| c >= 2).count();
if repeated != 1 || counts.iter().any(|&c| c > 3) {
return false;
}
let mut product = 10;
for &x in &d[..10] {
let mut sum = (x + product) % 10;
if sum == 0 {
sum = 10;
}
product = (2 * sum) % 11;
}
let check = match 11 - product {
10 => 0,
c => c,
};
check == d[10]
}
const DE_TAX_WORDS: &[&str] = &[
"steuer-id",
"steueridentifikationsnummer",
"steuerliche identifikationsnummer",
"idnr",
"identifikationsnummer",
"tax id",
];
fn de_tax_id(text: &str, findings: &mut Findings) {
for c in DE_TAX.captures_iter(text) {
let whole = c.get(0).unwrap();
let written = [&c[1], &c[2], &c[3]].iter().all(|s| *s == " ");
let n: String = whole.as_str().replace(' ', "");
if de_tax_id_valid(&n)
&& (written || word_near(text, whole.start(), whole.end(), DE_TAX_WORDS))
{
findings.insert(n);
}
}
}
/// The ID card's document number: a letter from the card's alphabet, eight
/// more characters from it, then the check digit.
static DE_ID: LazyLock<Regex> =
LazyLock::new(|| re(r"\b[CFGHJKLMNPRTVWXYZ][CFGHJKLMNPRTVWXYZ0-9]{8}\d\b"));
/// ICAO 9303 check digit: weights 7, 3, 1; letters A=10 … Z=35.
pub fn icao_check(chars: &str, check: u32) -> bool {
let value = |c: char| c.to_digit(10).unwrap_or_else(|| c as u32 - 'A' as u32 + 10);
let sum: u32 = chars
.chars()
.zip([7, 3, 1].iter().cycle())
.map(|(c, w)| value(c) * w)
.sum();
sum % 10 == check
}
fn de_id_card(text: &str, findings: &mut Findings) {
for m in DE_ID.find_iter(text) {
let s = m.as_str();
if icao_check(&s[..9], num(&s[9..])) {
findings.insert(s);
}
}
}
// --- France ---------------------------------------------------------------
/// Sex, year, month, department (with Corsica's 2A and 2B), commune, order,
/// then the two-digit key, spaces allowed between groups.
static FR_NIR: LazyLock<Regex> = LazyLock::new(|| {
re(r"\b([1-478]) ?(\d{2}) ?(\d{2}) ?(\d{2}|2[AB]) ?(\d{3}) ?(\d{3}) ?(\d{2})\b")
});
fn fr_nir(text: &str, findings: &mut Findings) {
for c in FR_NIR.captures_iter(text) {
let month = num(&c[3]);
if !(matches!(month, 1..=12 | 20..=42 | 50..=99)) {
continue;
}
let department = match &c[4] {
"2A" => "19",
"2B" => "18",
d => d,
};
let body = format!(
"{}{}{}{}{}{}",
&c[1], &c[2], &c[3], department, &c[5], &c[6]
);
let Ok(value) = body.parse::<u64>() else {
continue;
};
if 97 - value % 97 == u64::from(num(&c[7])) {
findings.insert(format!(
"{}{}{}{}{}{}{}",
&c[1], &c[2], &c[3], &c[4], &c[5], &c[6], &c[7]
));
}
}
}
// --- Spain ----------------------------------------------------------------
static ES_ID: LazyLock<Regex> = LazyLock::new(|| re(r"(?i)\b([XYZ]?)[ -]?(\d{7,8})[ -]?([A-Z])\b"));
const DNI_LETTERS: &[u8] = b"TRWAGMYFPDXBNJZSQVHLCKE";
fn es_dni_nie(text: &str, findings: &mut Findings) {
for c in ES_ID.captures_iter(text) {
let prefix = c[1].to_ascii_uppercase();
let digits = &c[2];
// DNI: eight digits; NIE: X, Y or Z and seven digits
let number = match (prefix.as_str(), digits.len()) {
("", 8) => digits.to_string(),
("X", 7) => format!("0{digits}"),
("Y", 7) => format!("1{digits}"),
("Z", 7) => format!("2{digits}"),
_ => continue,
};
let letter = c[3].to_ascii_uppercase();
if DNI_LETTERS[(num(&number) % 23) as usize] == letter.as_bytes()[0] {
findings.insert(format!("{prefix}{digits}{letter}"));
}
}
}
// --- Italy ----------------------------------------------------------------
/// Surname and name letters, year, month letter, day, place code, check
/// letter; digits may be replaced by letters (omocodia).
static IT_CF: LazyLock<Regex> = LazyLock::new(|| {
let d = "[0-9LMNPQRSTUV]";
re(&format!(
r"(?i)\b[A-Z]{{6}}{d}{{2}}[ABCDEHLMPRST]{d}{{2}}[A-Z]{d}{{3}}[A-Z]\b"
))
});
/// The Ministry's odd-position values for 0–9 and A–Z.
const CF_ODD: [u32; 36] = [
1, 0, 5, 7, 9, 13, 15, 17, 19, 21, // 0-9
1, 0, 5, 7, 9, 13, 15, 17, 19, 21, 2, 4, 18, 20, 11, 3, 6, 8, 12, 14, 16, 10, 22, 25, 24,
23, // A-Z
];
pub fn codice_fiscale_valid(cf: &str) -> bool {
let index = |c: u8| {
if c.is_ascii_digit() {
(c - b'0') as usize
} else {
(c - b'A') as usize + 10
}
};
let even = |c: u8| {
if c.is_ascii_digit() {
u32::from(c - b'0')
} else {
u32::from(c - b'A')
}
};
let bytes = cf.as_bytes();
let sum: u32 = bytes[..15]
.iter()
.enumerate()
.map(|(i, &c)| {
if i % 2 == 0 {
CF_ODD[index(c)]
} else {
even(c)
}
})
.sum();
u32::from(bytes[15] - b'A') == sum % 26
}
fn it_codice_fiscale(text: &str, findings: &mut Findings) {
for m in IT_CF.find_iter(text) {
let cf = m.as_str().to_ascii_uppercase();
if codice_fiscale_valid(&cf) {
findings.insert(cf);
}
}
}
// --- Netherlands ----------------------------------------------------------
/// Nine digits, sometimes written `1112.22.333`.
static NL_BSN: LazyLock<Regex> = LazyLock::new(|| re(r"\b(\d{4})(\.?)(\d{2})(\.?)(\d{3})\b"));
/// The eleven test: weights 9 down to 2, and −1 for the last digit.
pub fn bsn_valid(n: &str) -> bool {
let d = digit_values(n);
let sum: i64 = d[..8]
.iter()
.zip((2..=9).rev())
.map(|(a, w)| i64::from(a * w))
.sum::<i64>()
- i64::from(d[8]);
sum != 0 && sum % 11 == 0
}
const BSN_WORDS: &[&str] = &[
"bsn",
"burgerservicenummer",
"sofinummer",
"sofi-nummer",
"citizen service number",
];
fn nl_bsn(text: &str, findings: &mut Findings) {
for c in NL_BSN.captures_iter(text) {
let whole = c.get(0).unwrap();
let n = format!("{}{}{}", &c[1], &c[3], &c[5]);
let written = &c[2] == "." && &c[4] == ".";
if bsn_valid(&n) && (written || word_near(text, whole.start(), whole.end(), BSN_WORDS)) {
findings.insert(n);
}
}
}
// --- Belgium --------------------------------------------------------------
/// `YY.MM.DD-XXX.CC` or eleven digits.
static BE_NN: LazyLock<Regex> =
LazyLock::new(|| re(r"\b(\d{2})\.?(\d{2})\.?(\d{2})-?(\d{3})\.?(\d{2})\b"));
fn be_national_number(text: &str, findings: &mut Findings) {
for c in BE_NN.captures_iter(text) {
let (month, day) = (num(&c[2]), num(&c[3]));
// Month 0 and day 0 mean unknown; bis numbers add 20 or 40 to the month
if !(month <= 12 || (20..=32).contains(&month) || (40..=52).contains(&month)) || day > 31 {
continue;
}
let body = format!("{}{}{}{}", &c[1], &c[2], &c[3], &c[4]);
let check = u64::from(num(&c[5]));
let before_2000 = 97 - body.parse::<u64>().unwrap_or(0) % 97;
let since_2000 = 97 - format!("2{body}").parse::<u64>().unwrap_or(0) % 97;
if check == before_2000 || check == since_2000 {
findings.insert(format!("{body}{}", &c[5]));
}
}
}
// --- Poland ---------------------------------------------------------------
static ELEVEN: LazyLock<Regex> = LazyLock::new(|| re(r"\b\d{11}\b"));
/// Weights 1, 3, 7, 9 repeating; the birth date encodes the century in the
/// month (+80 for the 1800s, +20 for the 2000s, and so on).
pub fn pesel_valid(n: &str) -> bool {
let d = digit_values(n);
let sum: u32 = d[..10]
.iter()
.zip([1, 3, 7, 9].iter().cycle())
.map(|(a, w)| a * w)
.sum();
let month = d[2] * 10 + d[3];
let (century, month) = match month {
81..=92 => (1800, month - 80),
1..=12 => (1900, month),
21..=32 => (2000, month - 20),
41..=52 => (2100, month - 40),
_ => return false,
};
let year = century + d[0] * 10 + d[1];
(10 - sum % 10) % 10 == d[10] && (1..=super::days_in(year, month)).contains(&(d[4] * 10 + d[5]))
}
const PESEL_WORDS: &[&str] = &["pesel", "numer pesel", "nr pesel"];
fn pl_pesel(text: &str, findings: &mut Findings) {
for m in ELEVEN.find_iter(text) {
if pesel_valid(m.as_str()) && word_near(text, m.start(), m.end(), PESEL_WORDS) {
findings.insert(m.as_str());
}
}
}
// --- Sweden ---------------------------------------------------------------
/// `YYMMDD-NNNN`, `YYYYMMDD-NNNN` (`+` after 100), or the bare digits.
static SE_PNR: LazyLock<Regex> =
LazyLock::new(|| re(r"\b(?:\d{2})?(\d{2})(\d{2})(\d{2})([-+]?)(\d{4})\b"));
const SE_WORDS: &[&str] = &[
"personnummer",
"personnr",
"person nr",
"samordningsnummer",
"pnr",
];
fn se_personnummer(text: &str, findings: &mut Findings) {
for c in SE_PNR.captures_iter(text) {
let whole = c.get(0).unwrap();
let (yy, month, day) = (num(&c[1]), num(&c[2]), num(&c[3]));
// Coordination numbers add 60 to the day
let day = if day > 60 { day - 60 } else { day };
let ten = format!("{}{}{}{}", &c[1], &c[2], &c[3], &c[5]);
let written = !c[4].is_empty();
if valid_short_date(yy, month, day)
&& checks::luhn(&ten)
&& (written || word_near(text, whole.start(), whole.end(), SE_WORDS))
{
findings.insert(ten);
}
}
}
// --- Denmark --------------------------------------------------------------
static DK_CPR: LazyLock<Regex> = LazyLock::new(|| re(r"\b(\d{2})(\d{2})(\d{2})-?(\d{4})\b"));
const CPR_WORDS: &[&str] = &["cpr", "cpr-nr", "cpr nr", "cpr-nummer", "personnummer"];
fn dk_cpr(text: &str, findings: &mut Findings) {
for c in DK_CPR.captures_iter(text) {
let whole = c.get(0).unwrap();
if valid_short_date(num(&c[3]), num(&c[2]), num(&c[1]))
&& word_near(text, whole.start(), whole.end(), CPR_WORDS)
{
findings.insert(whole.as_str().replace('-', ""));
}
}
}
// --- Finland --------------------------------------------------------------
static FI_HETU: LazyLock<Regex> =
LazyLock::new(|| re(r"(?i)\b(\d{2})(\d{2})(\d{2})[-+ABCDEFYXWVU](\d{3})([0-9A-Y])\b"));
const HETU_CHECK: &[u8] = b"0123456789ABCDEFHJKLMNPRSTUVWXY";
fn fi_hetu(text: &str, findings: &mut Findings) {
for c in FI_HETU.captures_iter(text) {
let (day, month, yy) = (num(&c[1]), num(&c[2]), num(&c[3]));
let n: u64 = format!("{}{}{}{}", &c[1], &c[2], &c[3], &c[4])
.parse()
.unwrap_or(0);
let check = c[5].to_ascii_uppercase().as_bytes()[0];
if valid_short_date(yy, month, day) && HETU_CHECK[(n % 31) as usize] == check {
findings.insert(c[0].to_ascii_uppercase());
}
}
}
// --- Ireland --------------------------------------------------------------
static IE_PPS: LazyLock<Regex> = LazyLock::new(|| re(r"(?i)\b(\d{7})([A-W])([ABHW]?)\b"));
const PPS_CHECK: &[u8] = b"WABCDEFGHIJKLMNOPQRSTUV";
fn ie_pps(text: &str, findings: &mut Findings) {
for c in IE_PPS.captures_iter(text) {
let mut sum: u32 = digit_values(&c[1])
.iter()
.zip((2..=8).rev())
.map(|(a, w)| a * w)
.sum();
// The second letter counts, times 9; W (the old form) counts as 0
let second = c[3].to_ascii_uppercase();
if let Some(&letter) = second.as_bytes().first()
&& letter != b'W'
{
sum += u32::from(letter - b'A' + 1) * 9;
}
let check = c[2].to_ascii_uppercase().as_bytes()[0];
if PPS_CHECK[(sum % 23) as usize] == check {
findings.insert(c[0].to_ascii_uppercase());
}
}
}
// --- Portugal -------------------------------------------------------------
static NINE: LazyLock<Regex> = LazyLock::new(|| re(r"\b\d{9}\b"));
/// Mod 11 over weights 9 down to 2; a check of 10 or 11 becomes 0.
pub fn nif_valid(n: &str) -> bool {
let d = digit_values(n);
let sum: u32 = d[..8].iter().zip((2..=9).rev()).map(|(a, w)| a * w).sum();
let check = match 11 - sum % 11 {
10 | 11 => 0,
c => c,
};
matches!(d[0], 1 | 2 | 3 | 5 | 6 | 8 | 9) && check == d[8]
}
const NIF_WORDS: &[&str] = &[
"nif",
"contribuinte",
"número de identificação fiscal",
"numero de contribuinte",
];
fn pt_nif(text: &str, findings: &mut Findings) {
for m in NINE.find_iter(text) {
if nif_valid(m.as_str()) && word_near(text, m.start(), m.end(), NIF_WORDS) {
findings.insert(m.as_str());
}
}
}
// --- Austria --------------------------------------------------------------
/// A serial and check digit, then the birth date: `1237 010180`.
static AT_SVNR: LazyLock<Regex> = LazyLock::new(|| re(r"\b(\d{3})(\d)( ?)(\d{2})(\d{2})(\d{2})\b"));
const SVNR_WORDS: &[&str] = &[
"sozialversicherungsnummer",
"svnr",
"sv-nr",
"sv-nummer",
"versicherungsnummer",
];
fn at_svnr(text: &str, findings: &mut Findings) {
for c in AT_SVNR.captures_iter(text) {
let whole = c.get(0).unwrap();
let n = format!("{}{}{}{}{}", &c[1], &c[2], &c[4], &c[5], &c[6]);
let d = digit_values(&n);
let sum: u32 = d
.iter()
.zip([3, 7, 9, 0, 5, 8, 4, 2, 1, 6])
.map(|(a, w)| a * w)
.sum();
let written = &c[3] == " ";
if d[0] != 0
&& sum % 11 == d[3]
&& valid_short_date(num(&c[6]), num(&c[5]), num(&c[4]))
&& (written || word_near(text, whole.start(), whole.end(), SVNR_WORDS))
&& stands_alone(text, whole.start(), whole.end())
{
findings.insert(n);
}
}
}
#[cfg(test)]
mod tests {
use crate::mailflow::detectors::by_id;
fn count(id: &str, text: &str) -> usize {
by_id(id).unwrap().count(text)
}
#[test]
fn germany() {
assert_eq!(count("de-tax-id", "86 095 742 719"), 1);
assert_eq!(count("de-tax-id", "Steuer-ID: 86095742719"), 1);
assert_eq!(count("de-tax-id", "Rechnung 86095742719"), 0);
assert_eq!(count("de-tax-id", "86 095 742 718"), 0);
// ICAO 9303's German specimen card
assert_eq!(count("de-id-card", "Ausweis T220001293"), 1);
assert_eq!(count("de-id-card", "T220001294"), 0);
}
#[test]
fn france_spain_italy() {
assert_eq!(count("fr-nir", "2 55 08 14 168 025 38"), 1);
assert_eq!(count("fr-nir", "255081416802539"), 0);
assert_eq!(count("es-dni-nie", "DNI 12345678Z, NIE X-1234567-L"), 2);
assert_eq!(count("es-dni-nie", "12345678A"), 0);
assert_eq!(count("it-codice-fiscale", "CF: RSSMRA85T10A562S"), 1);
assert_eq!(count("it-codice-fiscale", "RSSMRA85T10A562T"), 0);
}
#[test]
fn benelux() {
assert_eq!(count("nl-bsn", "1112.22.333"), 1);
assert_eq!(count("nl-bsn", "BSN 111222333"), 1);
assert_eq!(count("nl-bsn", "order 111222333"), 0);
assert_eq!(count("nl-bsn", "BSN 111222334"), 0);
assert_eq!(count("be-national-number", "85.07.30-033.28"), 1);
assert_eq!(count("be-national-number", "85073003329"), 0);
}
#[test]
fn nordics() {
assert_eq!(count("se-personnummer", "811218-9876"), 1);
assert_eq!(count("se-personnummer", "811218-9875"), 0);
assert_eq!(count("se-personnummer", "order 8112189876"), 0);
assert_eq!(count("se-personnummer", "personnummer 198112189876"), 1);
assert_eq!(count("dk-cpr", "CPR-nr: 010170-1234"), 1);
assert_eq!(count("dk-cpr", "010170-1234"), 0);
assert_eq!(count("dk-cpr", "CPR 320170-1234"), 0);
assert_eq!(count("fi-hetu", "131052-308T"), 1);
assert_eq!(count("fi-hetu", "131052-308U"), 0);
}
#[test]
fn poland_ireland_portugal_austria() {
assert_eq!(count("pl-pesel", "PESEL 44051401359, pesel 02070803628"), 2);
assert_eq!(count("pl-pesel", "PESEL 44051401358"), 0);
assert_eq!(count("pl-pesel", "44051401359"), 0);
assert_eq!(count("ie-pps", "PPS 1234567T and 1234567FA"), 2);
assert_eq!(count("ie-pps", "1234567U"), 0);
assert_eq!(count("pt-nif", "NIF 123456789"), 1);
assert_eq!(count("pt-nif", "NIF 123456788"), 0);
assert_eq!(count("at-svnr", "1237 010180"), 1);
assert_eq!(count("at-svnr", "SVNR 1237010180"), 1);
assert_eq!(count("at-svnr", "1238 010180"), 0);
}
}
@@ -1,116 +0,0 @@
/*
* SPDX-FileCopyrightText: 2026 Coffey Labs LLC
*
* SPDX-License-Identifier: AGPL-3.0-only
*/
//! European identifiers outside the EU (§2.3): Norway's national identity
//! number and Switzerland's AHV number.
use super::{Detector, Findings, Region, Strength, digit_values, valid_short_date};
use regex::Regex;
use std::sync::LazyLock;
pub static DETECTORS: &[Detector] = &[
Detector::new(
"no-fnr",
"Norway: national identity number",
Region::Europe,
Strength::Checked,
no_fnr,
),
Detector::new(
"ch-ahv",
"Switzerland: AHV number",
Region::Europe,
Strength::Checked,
ch_ahv,
),
];
static ELEVEN: LazyLock<Regex> =
LazyLock::new(|| Regex::new(r"\b\d{6} ?\d{5}\b").expect("detector pattern"));
/// Two mod 11 check digits over a birth date (D-numbers add 40 to the day,
/// H-numbers 40 to the month): strong enough to count alone.
pub fn fnr_valid(n: &str) -> bool {
let d = digit_values(n);
if d.len() != 11 {
return false;
}
let check =
|weights: &[u32]| match 11 - d.iter().zip(weights).map(|(a, w)| a * w).sum::<u32>() % 11 {
11 => Some(0),
10 => None,
c => Some(c),
};
let day = d[0] * 10 + d[1];
let month = d[2] * 10 + d[3];
let day = if day > 40 { day - 40 } else { day };
let month = if month > 40 { month - 40 } else { month };
valid_short_date(d[4] * 10 + d[5], month, day)
&& check(&[3, 7, 6, 1, 8, 9, 4, 5, 2]) == Some(d[9])
&& check(&[5, 4, 3, 2, 7, 6, 5, 4, 3, 2]) == Some(d[10])
}
fn no_fnr(text: &str, findings: &mut Findings) {
for m in ELEVEN.find_iter(text) {
let n = m.as_str().replace(' ', "");
if fnr_valid(&n) {
findings.insert(n);
}
}
}
/// `756.1234.5678.97`: the country prefix, then an EAN-13 check digit.
static AHV: LazyLock<Regex> = LazyLock::new(|| {
Regex::new(r"\b756[. ]?\d{4}[. ]?\d{4}[. ]?\d{2}\b").expect("detector pattern")
});
pub fn ean13_valid(n: &str) -> bool {
let d = digit_values(n);
if d.len() != 13 {
return false;
}
let sum: u32 = d[..12]
.iter()
.enumerate()
.map(|(i, x)| if i % 2 == 0 { *x } else { x * 3 })
.sum();
(10 - sum % 10) % 10 == d[12]
}
fn ch_ahv(text: &str, findings: &mut Findings) {
for m in AHV.find_iter(text) {
let n: String = m.as_str().chars().filter(char::is_ascii_digit).collect();
if ean13_valid(&n) {
findings.insert(n);
}
}
}
#[cfg(test)]
mod tests {
use crate::mailflow::detectors::by_id;
fn count(id: &str, text: &str) -> usize {
by_id(id).unwrap().count(text)
}
#[test]
fn norway() {
assert_eq!(count("no-fnr", "01019000083"), 1);
assert_eq!(count("no-fnr", "010190 00083"), 1);
assert_eq!(count("no-fnr", "01019000084"), 0);
// Not a date
assert_eq!(count("no-fnr", "32019000083"), 0);
}
#[test]
fn switzerland() {
// The federal example
assert_eq!(count("ch-ahv", "AHV 756.9217.0769.85"), 1);
assert_eq!(count("ch-ahv", "7569217076985"), 1);
assert_eq!(count("ch-ahv", "756.9217.0769.86"), 0);
}
}
@@ -1,278 +0,0 @@
/*
* SPDX-FileCopyrightText: 2026 Coffey Labs LLC
*
* SPDX-License-Identifier: AGPL-3.0-only
*/
//! Detectors (dlp-and-mail-flow-rules spec, §2.3): each finds one kind of
//! identifier in text and reports the distinct ones it found.
//!
//! A detector is one of two strengths:
//!
//! - **Checked**: the identifier carries a published check digit or
//! checksum, so a random number rarely passes; found on its own.
//! - **Needs a word**: the format alone is too common, so a candidate counts
//! only with a corroborating word within [`WINDOW`] characters either
//! side.
//!
//! Findings are distinct normalized values (digits only, upper case), so the
//! same card number pasted twice counts once. They stay in memory: callers
//! read only [`Findings::len`].
pub mod africa;
pub mod americas;
pub mod any;
pub mod asia;
pub mod australia;
pub mod canada;
pub mod checks;
pub mod eu;
pub mod europe;
pub mod templates;
pub mod uk;
pub mod us;
use ahash::AHashSet;
/// How far, in characters, a corroborating word may be from a candidate.
pub const WINDOW: usize = 50;
#[derive(Debug, Clone, Copy, PartialEq, Eq)]
pub enum Strength {
Checked,
NeedsWord,
}
#[derive(Debug, Clone, Copy, PartialEq, Eq)]
pub enum Region {
Any,
Us,
Uk,
Canada,
Australia,
Eu,
Europe,
Asia,
Americas,
Africa,
}
/// The distinct values one detector found.
#[derive(Debug, Default)]
pub struct Findings(AHashSet<String>);
impl Findings {
pub fn insert(&mut self, value: impl Into<String>) {
self.0.insert(value.into());
}
pub fn len(&self) -> usize {
self.0.len()
}
pub fn is_empty(&self) -> bool {
self.0.is_empty()
}
}
pub struct Detector {
/// Stable id, stored in rules: `payment-card`, `iban`, `us-ssn`.
pub id: &'static str,
pub name: &'static str,
pub region: Region,
pub strength: Strength,
find: fn(&str, &mut Findings),
}
impl Detector {
pub const fn new(
id: &'static str,
name: &'static str,
region: Region,
strength: Strength,
find: fn(&str, &mut Findings),
) -> Self {
Self {
id,
name,
region,
strength,
find,
}
}
/// Adds what this detector finds in `text` to `findings`. Call once per
/// piece of text (subject, each part, each attachment) with the same
/// `findings`, then read its length.
pub fn find(&self, text: &str, findings: &mut Findings) {
(self.find)(text, findings)
}
/// The distinct values found in one text.
pub fn count(&self, text: &str) -> usize {
let mut findings = Findings::default();
self.find(text, &mut findings);
findings.len()
}
}
/// Every detector, in the order the console lists them.
pub fn all() -> impl Iterator<Item = &'static Detector> {
[
any::DETECTORS,
us::DETECTORS,
uk::DETECTORS,
canada::DETECTORS,
australia::DETECTORS,
eu::DETECTORS,
europe::DETECTORS,
asia::DETECTORS,
americas::DETECTORS,
africa::DETECTORS,
]
.into_iter()
.flatten()
}
pub fn by_id(id: &str) -> Option<&'static Detector> {
all().find(|detector| detector.id == id)
}
/// Whether one of `words` appears, as a whole word and ignoring case, within
/// [`WINDOW`] characters before `start` or after `end` (byte offsets of the
/// candidate in `text`). The window is widened by the longest word, so a
/// word that reaches into it still counts whole.
pub fn word_near(text: &str, start: usize, end: usize, words: &[&str]) -> bool {
let reach = WINDOW + words.iter().map(|w| w.chars().count()).max().unwrap_or(0);
let before = text[..start]
.char_indices()
.rev()
.nth(reach - 1)
.map_or(0, |(i, _)| i);
let after = text[end..]
.char_indices()
.nth(reach)
.map_or(text.len(), |(i, _)| end + i);
let window = text[before..after].to_lowercase();
words.iter().any(|word| contains_word(&window, word))
}
/// Whether `word` (lower case) appears in `haystack` (lower case) with no
/// letter or digit on either side.
pub fn contains_word(haystack: &str, word: &str) -> bool {
haystack.match_indices(word).any(|(i, _)| {
let before_ok = haystack[..i]
.chars()
.next_back()
.is_none_or(|c| !c.is_alphanumeric());
let after_ok = haystack[i + word.len()..]
.chars()
.next()
.is_none_or(|c| !c.is_alphanumeric());
before_ok && after_ok
})
}
/// Whether the match at `start..end` stands alone: no digit or letter
/// directly before or after it, so `123-45-6789` isn't found inside a
/// longer run of digits.
pub fn stands_alone(text: &str, start: usize, end: usize) -> bool {
let before = text[..start].chars().next_back();
let after = text[end..].chars().next();
before.is_none_or(|c| !c.is_alphanumeric()) && after.is_none_or(|c| !c.is_alphanumeric())
}
/// Days in `month` of `year` (0 for a month that doesn't exist).
pub fn days_in(year: u32, month: u32) -> u32 {
match month {
1 | 3 | 5 | 7 | 8 | 10 | 12 => 31,
4 | 6 | 9 | 11 => 30,
2 if year.is_multiple_of(4) && (!year.is_multiple_of(100) || year.is_multiple_of(400)) => {
29
}
2 => 28,
_ => 0,
}
}
/// Whether `year`-`month`-`day` is a real date between 1900 and 2100.
pub fn valid_date(year: u32, month: u32, day: u32) -> bool {
(1900..=2100).contains(&year) && (1..=days_in(year, month)).contains(&day)
}
/// Whether a two-digit year, month and day make a real date in either the
/// 1900s or the 2000s.
pub fn valid_short_date(yy: u32, month: u32, day: u32) -> bool {
valid_date(1900 + yy, month, day) || valid_date(2000 + yy, month, day)
}
/// The value of each digit in `s`.
pub fn digit_values(s: &str) -> Vec<u32> {
s.bytes()
.filter(u8::is_ascii_digit)
.map(|b| u32::from(b - b'0'))
.collect()
}
/// The ASCII digits of `s`.
pub fn digits(s: &str) -> String {
s.chars().filter(char::is_ascii_digit).collect()
}
#[cfg(test)]
mod tests {
use super::*;
#[test]
fn words_are_whole_and_near() {
let text = "Your passport number is X1234567, thanks";
let start = text.find("X123").unwrap();
assert!(word_near(text, start, start + 8, &["passport"]));
assert!(!word_near(text, start, start + 8, &["pass"]));
let far = format!("passport{}X1234567", " ".repeat(60));
let start = far.find("X123").unwrap();
assert!(!word_near(&far, start, start + 8, &["passport"]));
}
#[test]
fn near_counts_characters_not_bytes() {
// 45 two-byte characters between the word and the candidate: within
// 50 characters, though over 50 bytes
let text = format!("passport {} X1234567", "é".repeat(45));
let start = text.find("X123").unwrap();
assert!(word_near(&text, start, start + 8, &["passport"]));
}
/// An ordinary business email: order, invoice and tracking numbers,
/// dates, amounts, a street address. Nothing here is an identifier, so
/// no detector may fire, except the contact ones on the signature.
#[test]
fn ordinary_mail_finds_nothing() {
let text = "Hi Dana,\n\nThanks for order 4471-2290 placed 2026-09-14. Invoice INV-2026-00917 \
for $12,480.00 is due 10/31/2026; PO 7731902 covers lines 1-14. Tracking \
1Z999AA10123456784, parcel 3 of 5, 12.5 kg, box 40x30x20 cm. Meeting moved to \
Tuesday 9:30-10:15 in room 2B, building 1177. Ticket #5520318, case 20260914-0042. \
Version 2026.9.28.4, build 118822, commit 5a73a118. Serial SN-88213-X. \
Ship to 1600 Amphitheatre Pkwy, Mountain View, CA 94043. Revenue grew 18% to \
1,204,332 units; see figures 3.1-3.4 and table 12.\n\nBest,\nSam\n\
Sam Rivera | +1 (415) 555-2671 | sam@example.com";
let quiet = ["email-addresses", "phone-numbers"];
for detector in all().filter(|d| !quiet.contains(&d.id)) {
assert_eq!(
detector.count(text),
0,
"{} fired on ordinary mail",
detector.id
);
}
}
#[test]
fn ids_are_unique() {
let mut seen = AHashSet::new();
for detector in all() {
assert!(seen.insert(detector.id), "duplicate id {}", detector.id);
assert!(by_id(detector.id).is_some());
}
}
}
@@ -1,97 +0,0 @@
/*
* SPDX-FileCopyrightText: 2026 Coffey Labs LLC
*
* SPDX-License-Identifier: AGPL-3.0-only
*/
//! Templates (§2.3): named sets of detectors, so a policy doesn't pick forty
//! one at a time. Each is named for what it finds, never for a law, and is a
//! starting point: once added to a rule, its detectors can be changed.
pub struct Template {
pub id: &'static str,
pub name: &'static str,
pub detectors: &'static [&'static str],
}
pub static TEMPLATES: &[Template] = &[
Template {
id: "payment-and-bank",
name: "Payment cards and bank accounts",
detectors: &["payment-card", "iban", "swift-bic", "us-aba-routing"],
},
Template {
id: "us-personal",
name: "US personal identifiers",
detectors: &[
"us-ssn",
"us-itin",
"us-ein",
"us-drivers-license",
"passport",
"date-of-birth",
],
},
Template {
id: "uk-personal",
name: "UK personal identifiers",
detectors: &["uk-nino", "uk-utr", "uk-nhs", "passport", "date-of-birth"],
},
Template {
id: "eu-national",
name: "EU national identifiers",
detectors: &[
"de-tax-id",
"de-id-card",
"fr-nir",
"es-dni-nie",
"it-codice-fiscale",
"nl-bsn",
"be-national-number",
"pl-pesel",
"se-personnummer",
"dk-cpr",
"fi-hetu",
"ie-pps",
"pt-nif",
"at-svnr",
],
},
Template {
id: "health",
name: "Health identifiers",
detectors: &["uk-nhs", "us-mbi", "us-npi", "us-dea", "au-medicare"],
},
Template {
id: "credentials",
name: "Credentials and keys",
detectors: &["private-key", "credentials"],
},
Template {
id: "contact-lists",
name: "Contact lists",
detectors: &["email-addresses", "phone-numbers"],
},
];
pub fn by_id(id: &str) -> Option<&'static Template> {
TEMPLATES.iter().find(|template| template.id == id)
}
#[cfg(test)]
mod tests {
use super::*;
#[test]
fn every_template_names_real_detectors() {
for template in TEMPLATES {
for id in template.detectors {
assert!(
super::super::by_id(id).is_some(),
"{}: no detector {id}",
template.id
);
}
}
}
}
@@ -1,155 +0,0 @@
/*
* SPDX-FileCopyrightText: 2026 Coffey Labs LLC
*
* SPDX-License-Identifier: AGPL-3.0-only
*/
//! United Kingdom identifiers (§2.3): HMRC's National Insurance number and
//! Unique Taxpayer Reference, and the NHS number.
use super::{Detector, Findings, Region, Strength, word_near};
use regex::Regex;
use std::sync::LazyLock;
pub static DETECTORS: &[Detector] = &[
Detector::new(
"uk-nino",
"UK National Insurance number",
Region::Uk,
Strength::Checked,
nino,
),
Detector::new(
"uk-nhs",
"UK NHS number",
Region::Uk,
Strength::Checked,
nhs,
),
Detector::new(
"uk-utr",
"UK Unique Taxpayer Reference",
Region::Uk,
Strength::NeedsWord,
utr,
),
];
fn re(pattern: &str) -> Regex {
Regex::new(pattern).expect("detector pattern")
}
/// Two letters, six digits (often in pairs), a suffix A–D.
static NINO: LazyLock<Regex> =
LazyLock::new(|| re(r"(?i)\b([A-Z])([A-Z]) ?(\d{2}) ?(\d{2}) ?(\d{2}) ?([A-D])\b"));
/// HMRC's rules: D, F, I, Q, U and V are never used; O never second; and
/// BG, GB, KN, NK, NT, TN and ZZ are never allocated.
fn nino_prefix(first: char, second: char) -> bool {
const NEVER: &str = "DFIQUV";
let pair: String = [first, second].iter().collect();
!NEVER.contains(first)
&& !NEVER.contains(second)
&& second != 'O'
&& !["BG", "GB", "KN", "NK", "NT", "TN", "ZZ"].contains(&pair.as_str())
}
fn nino(text: &str, findings: &mut Findings) {
for c in NINO.captures_iter(text) {
let first = c[1].to_ascii_uppercase().chars().next().unwrap();
let second = c[2].to_ascii_uppercase().chars().next().unwrap();
if nino_prefix(first, second) {
findings.insert(format!(
"{first}{second}{}{}{}{}",
&c[3],
&c[4],
&c[5],
c[6].to_ascii_uppercase()
));
}
}
}
/// `NNN NNN NNNN` stands alone; ten bare digits need a word.
static NHS: LazyLock<Regex> = LazyLock::new(|| re(r"\b(\d{3})([ -]?)(\d{3})([ -]?)(\d{4})\b"));
/// Mod 11: weights 10 down to 2 over the first nine digits; the check digit
/// is 11 minus the remainder (11 becomes 0; 10 is never issued).
pub fn nhs_valid(n: &str) -> bool {
let d: Vec<u32> = n.bytes().map(|b| u32::from(b - b'0')).collect();
let sum: u32 = d[..9].iter().zip((2..=10).rev()).map(|(a, w)| a * w).sum();
match 11 - sum % 11 {
11 => d[9] == 0,
10 => false,
check => d[9] == check,
}
}
const NHS_WORDS: &[&str] = &["nhs", "nhs number", "nhs no"];
fn nhs(text: &str, findings: &mut Findings) {
for c in NHS.captures_iter(text) {
let whole = c.get(0).unwrap();
let n = format!("{}{}{}", &c[1], &c[3], &c[5]);
let written = !c[2].is_empty() && c[2] == c[4];
if nhs_valid(&n) && (written || word_near(text, whole.start(), whole.end(), NHS_WORDS)) {
findings.insert(n);
}
}
}
static UTR: LazyLock<Regex> = LazyLock::new(|| re(r"\b\d{5} ?\d{5}\b"));
const UTR_WORDS: &[&str] = &[
"utr",
"unique taxpayer reference",
"tax reference",
"self assessment",
];
fn utr(text: &str, findings: &mut Findings) {
for m in UTR.find_iter(text) {
if word_near(text, m.start(), m.end(), UTR_WORDS) {
findings.insert(m.as_str().replace(' ', ""));
}
}
}
#[cfg(test)]
mod tests {
use crate::mailflow::detectors::by_id;
fn count(id: &str, text: &str) -> usize {
by_id(id).unwrap().count(text)
}
#[test]
fn national_insurance() {
assert_eq!(count("uk-nino", "NI: AB 12 34 56 C, ce123456d"), 2);
// Letters never used, pairs never allocated, a suffix past D
for bad in [
"QQ123456C",
"AO123456C",
"GB123456A",
"AB123456E",
"DA123456A",
] {
assert_eq!(count("uk-nino", bad), 0, "{bad}");
}
}
#[test]
fn nhs_numbers() {
// The NHS's own example
assert_eq!(count("uk-nhs", "943 476 5919"), 1);
assert_eq!(count("uk-nhs", "943 476 5918"), 0);
assert_eq!(count("uk-nhs", "order 9434765919"), 0);
assert_eq!(count("uk-nhs", "NHS number 9434765919"), 1);
}
#[test]
fn utr() {
assert_eq!(count("uk-utr", "UTR 12345 67890"), 1);
assert_eq!(count("uk-utr", "order 1234567890"), 0);
}
}
Loaded 100 of 404 files, more files were not shown because too many files have changed in this diff. Show more