Merge main into merge/upstream-v0.16.24
ci / fork-checks (pull_request) Successful in 21s
ci / build (pull_request) Successful in 35m18s

The schema, which both sides changed, merged as JSON with no conflicts.
The personal-data catalog (#83) gains upstream's new x:DnsServerPowerDns:
nothing personal but its API key, like the other DNS providers.
This commit is contained in:
2026-09-28 08:19:24 -07:00
15 changed files with 3463 additions and 5 deletions
+6
View File
@@ -40,6 +40,12 @@ jobs:
# nothing. CI never sees the difference; a release does. # nothing. CI never sees the difference; a release does.
- if: always() - if: always()
run: python3 tools/fork/context-check.py run: python3 tools/fork/context-check.py
# The personal-data catalog must classify every object and field the
# schema has, and name nothing that is gone.
- if: always()
run: python3 tools/fork/privacy-check.py
- if: always()
run: python3 -m unittest discover -s tools/fork/tests
build: build:
# Either runner (host1 or host2): the build needs no docker socket. # Either runner (host1 or host2): the build needs no docker socket.
+92 -1
View File
@@ -483,8 +483,16 @@ impl Tracers {
}; };
// Parse webhook events // Parse webhook events
// inbuxa: personal-data catalog, finding 1: an include list is
// sent as named; otherwise a webhook honors its level as a
// tracer does, and never sends a protocol's raw input or
// output (whole messages)
let level = Level::from(hook.level);
let named = (hook.events_policy == EventPolicy::Include)
.then(|| hook.events.iter().copied().collect::<AHashSet<_>>())
.unwrap_or_default();
apply_events(hook.events, hook.events_policy, |event_type| { apply_events(hook.events, hook.events_policy, |event_type| {
if event_type != EventType::Telemetry(TelemetryEvent::WebhookError) { if webhook_wants(event_type, level, &custom_levels, &named) {
tracer.interests.set(event_type); tracer.interests.set(event_type);
global_interests.set(event_type); global_interests.set(event_type);
} }
@@ -743,6 +751,31 @@ fn tracer_settings(tracer: &Tracer) -> u64 {
settings_hash(&tracer) settings_hash(&tracer)
} }
/// inbuxa: whether a webhook at `level` receives this event type. Its own
/// error event never, or a failing webhook would report itself to itself.
/// An event `named` in an include list always: naming it is the choice.
/// Otherwise (the exclude policy, the default) only events at or above its
/// level, as for a tracer, and never a protocol's raw input or output, which
/// carries whole messages and credentials.
fn webhook_wants(
event_type: EventType,
level: Level,
custom_levels: &AHashMap<EventType, Level>,
named: &AHashSet<EventType>,
) -> bool {
if event_type == EventType::Telemetry(TelemetryEvent::WebhookError) {
return false;
}
if named.contains(&event_type) {
return true;
}
let event_level = custom_levels
.get(&event_type)
.copied()
.unwrap_or(event_type.level());
level.is_contained(event_level) && !event_type.is_raw_io()
}
fn webhook_settings(hook: &WebHook) -> u64 { fn webhook_settings(hook: &WebHook) -> u64 {
let mut hook = hook.clone(); let mut hook = hook.clone();
in_place_reset!(hook); in_place_reset!(hook);
@@ -804,3 +837,61 @@ impl std::fmt::Debug for OtelMetrics {
.finish() .finish()
} }
} }
#[cfg(test)]
mod tests {
use super::*;
use trc::{AuthEvent, SmtpEvent};
fn wants(event: EventType, level: Level, named: &[EventType]) -> bool {
webhook_wants(
event,
level,
&AHashMap::new(),
&named.iter().copied().collect(),
)
}
#[test]
fn a_webhook_honors_its_level() {
let success = EventType::Auth(AuthEvent::Success);
assert!(wants(success, Level::Info, &[]));
assert!(!wants(success, Level::Error, &[]), "info is below error");
}
#[test]
fn raw_io_goes_out_only_when_named() {
let raw = EventType::Smtp(SmtpEvent::RawInput);
assert!(raw.is_raw_io());
// Not with the exclude policy, even at trace
assert!(!wants(raw, Level::Info, &[]));
assert!(!wants(raw, Level::Trace, &[]));
// Named in an include list, whatever the level
assert!(wants(raw, Level::Info, &[raw]));
}
#[test]
fn a_named_event_is_sent_whatever_its_level() {
let start = EventType::Smtp(SmtpEvent::ConnectionStart);
assert!(!Level::Info.is_contained(start.level()), "below info");
assert!(!wants(start, Level::Info, &[]));
assert!(wants(start, Level::Info, &[start]));
}
#[test]
fn a_custom_level_counts() {
let start = EventType::Smtp(SmtpEvent::ConnectionStart);
let custom = [(start, Level::Info)].into_iter().collect::<AHashMap<_, _>>();
assert!(webhook_wants(start, Level::Info, &custom, &AHashSet::new()));
// Raw I/O raised to info still needs naming
let raw = EventType::Smtp(SmtpEvent::RawInput);
let custom = [(raw, Level::Info)].into_iter().collect::<AHashMap<_, _>>();
assert!(!webhook_wants(raw, Level::Info, &custom, &AHashSet::new()));
}
#[test]
fn a_webhook_never_hears_its_own_errors() {
let own = EventType::Telemetry(TelemetryEvent::WebhookError);
assert!(!wants(own, Level::Trace, &[own]));
}
}
+101 -1
View File
@@ -14,7 +14,7 @@ use aws_lc_rs::{
use registry::{ use registry::{
schema::{ schema::{
enums::*, enums::*,
prelude::{ObjectType, SocketAddr}, prelude::{Object, ObjectType, SocketAddr},
structs::*, structs::*,
}, },
types::{duration::Duration, error::Error, list::List, map::Map}, types::{duration::Duration, error::Error, list::List, map::Map},
@@ -388,6 +388,28 @@ async fn insert_safe_defaults(bp: &mut Bootstrap) -> trc::Result<()> {
} }
} }
// inbuxa: personal-data catalog, defaults D2, D3, D4 and D6 (settled
// 2026-09-28): privacy-leaning values, for new installs only. A server
// with roles is not new, and keeps its settings whether saved or left at
// the default. Each singleton is read, changed and written back whole, so
// anything already in it stays.
#[cfg(not(feature = "test_mode"))]
if bp.registry.count_object(ObjectType::Role).await? == 0 {
let mut security = bp.setting_infallible::<Security>().await;
let mut classifier = bp.setting_infallible::<SpamClassifier>().await;
let mut pyzor = bp.setting_infallible::<SpamPyzor>().await;
let mut retention = bp.setting_infallible::<DataRetention>().await;
new_install_privacy_defaults(&mut security, &mut classifier, &mut pyzor, &mut retention);
for object in [
Object::from(security),
classifier.into(),
pyzor.into(),
retention.into(),
] {
bp.registry.write(RegistryWrite::insert(&object)).await?;
}
}
if bp.registry.count_object(ObjectType::Role).await? == 0 { if bp.registry.count_object(ObjectType::Role).await? == 0 {
let permissions = DefaultPermissions::default(); let permissions = DefaultPermissions::default();
let mut role_ids = Vec::with_capacity(4); let mut role_ids = Vec::with_capacity(4);
@@ -560,3 +582,81 @@ async fn insert_safe_defaults(bp: &mut Bootstrap) -> trc::Result<()> {
Ok(()) Ok(())
} }
/// inbuxa: the new-install values of defaults D2, D3, D4 and D6 from the
/// personal-data catalog spec. Automatic IP bans expire after 30 days instead
/// of never; spam training samples are kept 90 days instead of 180; Pyzor,
/// which sends a digest of each message's text to a public server, is off;
/// delivery history is kept 14 days instead of 30.
fn new_install_privacy_defaults(
security: &mut Security,
classifier: &mut SpamClassifier,
pyzor: &mut SpamPyzor,
retention: &mut DataRetention,
) {
const DAY: u64 = 24 * 60 * 60 * 1000;
let ban_period = Some(Duration::from_millis(30 * DAY));
security.auth_ban_period = ban_period;
security.abuse_ban_period = ban_period;
security.loiter_ban_period = ban_period;
security.scan_ban_period = ban_period;
classifier.hold_samples_for = Duration::from_millis(90 * DAY);
pyzor.enable = false;
retention.hold_traces_for = Some(Duration::from_millis(14 * DAY));
}
#[cfg(test)]
mod tests {
use super::*;
const DAY: u64 = 24 * 60 * 60 * 1000;
#[test]
fn new_installs_get_the_privacy_defaults() {
let (mut security, mut classifier, mut pyzor, mut retention) = (
Security::default(),
SpamClassifier::default(),
SpamPyzor::default(),
DataRetention::default(),
);
// What an install gets without them: bans that never lift, 180-day
// samples, Pyzor on, 30-day traces.
assert_eq!(security.auth_ban_period, None);
assert!(pyzor.enable);
new_install_privacy_defaults(&mut security, &mut classifier, &mut pyzor, &mut retention);
for period in [
security.auth_ban_period,
security.abuse_ban_period,
security.loiter_ban_period,
security.scan_ban_period,
] {
assert_eq!(period.map(|p| p.into_inner().as_millis() as u64), Some(30 * DAY));
}
assert_eq!(classifier.hold_samples_for.into_inner().as_millis() as u64, 90 * DAY);
assert!(!pyzor.enable);
assert_eq!(
retention.hold_traces_for.map(|p| p.into_inner().as_millis() as u64),
Some(14 * DAY)
);
}
#[test]
fn everything_else_in_the_settings_stays() {
let mut retention = DataRetention {
archive_deleted_items_for: Some(Duration::from_millis(7 * DAY)),
..Default::default()
};
let before = retention.clone();
new_install_privacy_defaults(
&mut Security::default(),
&mut SpamClassifier::default(),
&mut SpamPyzor::default(),
&mut retention,
);
assert_eq!(retention.archive_deleted_items_for, before.archive_deleted_items_for);
assert_eq!(retention.hold_metrics_for, before.hold_metrics_for);
assert_eq!(retention.expunge_trash_after, before.expunge_trash_after);
}
}
+31
View File
@@ -426,6 +426,37 @@ impl Server {
} }
} }
impl Server {
/// inbuxa: personal-data catalog, D2: removes bans whose period is over.
/// They already stop blocking when they expire, and go when settings are
/// next loaded; the daily clean-up makes sure a server that seldom
/// reloads doesn't keep them.
pub async fn purge_expired_blocked_ips(&self) -> trc::Result<()> {
let now = now() as i64;
let mut expired = Vec::new();
for ip in self.registry().list::<BlockedIp>().await? {
if ip.object.expires_at.as_ref().is_some_and(|at| at.timestamp() <= now) {
let address = ip.object.address.clone();
let object = Object {
inner: ip.object.into(),
revision: ip.revision,
};
self.registry()
.write(RegistryWrite::delete_object(ip.id, &object))
.await?;
expired.push(trc::Value::from(address.into_inner().0));
}
}
if !expired.is_empty() {
trc::event!(
Security(trc::SecurityEvent::IpBlockExpired),
Details = expired
);
}
Ok(())
}
}
impl BlockedIps { impl BlockedIps {
pub async fn parse(bp: &mut Bootstrap) -> Self { pub async fn parse(bp: &mut Bootstrap) -> Self {
let mut ips = Self::default(); let mut ips = Self::default();
+3 -1
View File
@@ -47532,7 +47532,9 @@ impl Default for WebHook {
level: TracingLevel::Info, level: TracingLevel::Info,
lossy: false, lossy: false,
events: Default::default(), events: Default::default(),
events_policy: EventPolicy::Exclude, // inbuxa: personal-data catalog, D7: a new webhook sends nothing
// until its events are chosen
events_policy: EventPolicy::Include,
} }
} }
} }
@@ -269,6 +269,11 @@ async fn store_maintenance(
trc::error!(err.details("Failed to re-apply account locks")); trc::error!(err.details("Failed to re-apply account locks"));
} }
// inbuxa: personal-data catalog, D2: bans past their period go
if let Err(err) = server.purge_expired_blocked_ips().await {
trc::error!(err.details("Failed to purge expired IP bans"));
}
// inbuxa: AU-7: audit records past their retention go; a // inbuxa: AU-7: audit records past their retention go; a
// failure leaves them for the next run // failure leaves them for the next run
if let Err(err) = server.audit_purge().await { if let Err(err) = server.audit_purge().await {
+647
View File
@@ -0,0 +1,647 @@
# Feature spec: personal-data catalog, compliance role, and the Compliance section's first pages
Status: approved 2026-09-28, with the answers under [Settled](#settled).
Phase 1 of the GDPR auditor foundation: the investigation and the design. Not a
rebuild of an upstream feature, so it has no line in SPEC.md §4's table.
## Provenance
Written for the record SPEC.md §3 rule 3 asks for. Sources, and nothing else:
| Source | License | Used for |
|---|---|---|
| This repository at `de275ba` (2026-09-28): the code, `resources/schema/schema.json.gz`, `tools/fork/`, `.gitea/workflows/` | AGPL-3.0-only | Every claim in the source map: each row names the code that writes the data |
| inbuxa-admin (the console) at `8e281e5` | AGPL-3.0-only | How navigation, hand-built pages and visibility work |
| docs.inbuxa.org (the inbuxa.org repository at `6b12467`) | Own | What the documentation says, for "Contradictions" |
| `inbuxa-drafts/specs/audit-hold-lock.md`, `legacy-protocols.md` | Own | The Compliance pages that already exist, and the permission-rollout notes |
| Regulation (EU) 2016/679 (GDPR), Articles 5, 25, 30 and 32 | Public law | The shape of a record of processing: categories, whose data, where, how long, who receives it |
No Enterprise-only file or snippet was used. Claims were checked by reading the
code; where a search found nothing, the row says so rather than asserting a
negative.
## What it is
Three things, the foundation for a GDPR auditor:
1. **A personal-data catalog.** Every place the server can store or send
personal data, with its categories, whose data it is, where it lives, what
bounds its retention, and the settings that control it. What a particular
server holds depends on how its operator configured it, so the catalog
describes *possibilities*; the server evaluates it against live settings to
say what *this* server does (Phase 3).
2. **A compliance role** that can see and act on compliance data without being
able to change server settings.
3. **Two console pages** under Compliance: **Overview** and **Data
inventory**.
The catalog records facts and the auditor reports findings. Neither judges:
nothing in code, docs, UI text or output claims the server meets a legal
standard, or that one is guaranteed. Nothing here changes a default; proposed
changes are listed under [Default profile](#default-profile) for John to
decide.
**Out of scope** (their own specs): retention policies, data subject requests,
the guided questionnaire, jurisdiction packs, mobile and ihasmail screens.
Legal hold and the audit log are built already (audit-hold-lock spec, phases
1–4); this spec places them in the Compliance navigation and gives the
compliance role access to them, nothing more.
## 1. Vocabulary
**Categories** (fixed; a source may have several):
| Category | Means |
|---|---|
| `identifier` | Names, email addresses, account and principal ids, login names |
| `contact` | Phone numbers, postal addresses, vCard details, contact addresses given for an object (an OAuth client's contacts) |
| `network` | IP addresses and ports, client hostnames (EHLO, PTR), user agents, device push URLs |
| `content` | Message bodies, subjects, attachments, events, contacts' cards, files, scripts, free-text descriptions and reasons |
| `metadata` | Timestamps, sizes, flags, envelopes, delivery status, message-ids, counts |
| `credential` | Passwords and their hashes, tokens, keys, secrets |
**Whose** (data subjects): `holder` (the account holder), `correspondent`
(people who send to or receive from the server, and people in contact cards
and events), `administrator` (people acting on the server).
**Where** (location): `data-store`, `blob-store`, `search-store`,
`in-memory-store` (the configured one; the data store by default),
`memory` (process RAM, per node), `log-file`, `external` (sent to an
endpoint). **Leaves the host** is a separate column: a store is local unless
the operator points it at a remote backend.
**Scope**: `tenant` (attributable to a tenant through the account or domain),
`server` (server-wide only).
## 2. Source map
Defaults are those of a new install: the struct defaults, plus what first boot
inserts (`crates/common/src/manager/defaults.rs`). "Unbounded" means no setting
or process removes the data on a schedule.
### 2.1 Mail, groupware and user content
| Source | Categories | Whose | Controlled by | Retention bound | Where / leaves host | Scope | Written by | Default |
|---|---|---|---|---|---|---|---|---|
| Emails, mailboxes, threads | content, identifier, metadata | holder, correspondent | always on; `x:Email.encryptAtRest`, `x:Email.maxMessages` | unbounded until deleted; Trash emptied after `x:DataRetention.expungeTrashAfter` | data + blob store; leaves when sent | tenant | `email/src/message/ingest.rs` `email_ingest`; `delivery.rs` `deliver_message`; `message/delete.rs` `emails_delete`, `emails_auto_expunge` | collected; Trash 30 d |
| Email submissions | metadata (envelopes) | holder, correspondent | always on | `x:DataRetention.expungeSubmissionsAfter` | data store | tenant | `message/delete.rs` `purge_email_submissions` (purge) | 3 d |
| Full-text index: mail | content, identifier, metadata | holder, correspondent | `x:Search.indexEmail`, `indexEmailFields` | follows the item | search store (external if Elasticsearch/Meilisearch) | tenant | `services/src/task_manager/index.rs` `build_email_document` | indexed, all fields |
| Calendars, events, iTIP | content, identifier, metadata | holder, correspondent | `x:CalendarScheduling.enable`, `httpRsvpEnable`; `x:CalendarAlarm.enable` | unbounded; scheduling inbox `expungeSchedulingInboxAfter`, share notices `expungeShareNotifyAfter` | data + blob store; **iMIP mails event details to external attendees** | tenant | `dav/src/calendar/update.rs`; `jmap/src/calendar_event/set.rs`; `groupware/src/calendar/storage.rs`; `services/src/task_manager/imip.rs` | collected; 30 d / 30 d |
| Contacts, address books | contact, identifier | correspondent, holder | always on | unbounded | data store | tenant | `dav/src/card/update.rs`; `jmap/src/contact/set.rs`; `groupware/src/contact/storage.rs` | collected |
| Full-text index: calendars, contacts | content, contact, identifier | correspondent, holder | `x:Search.indexCalendar`/`Fields`, `indexContacts`/`Fields` | follows the item | search store | tenant | `index.rs` `build_calendar_document`, `build_contact_document` | indexed |
| Files (FileNode, WebDAV) | content, metadata | holder | `x:FileStorage.maxSize`, `maxFiles` | unbounded | data + blob store | tenant | `dav/src/file/update.rs`; `jmap/src/file/set.rs`; `groupware/src/file/storage.rs` | collected |
| Sieve scripts, vacation responses | content | holder (scripts may name correspondents) | user-set | unbounded | data + blob store | tenant | `jmap/src/sieve/set.rs`; `managesieve/src/op/putscript.rs`; `jmap/src/vacation/set.rs` | when set |
| Sieve "seen" ids (vacation, duplicate) | identifier (blake3 hash) | correspondent | used only by scripts that call vacation/duplicate | the script's TTL | in-memory store | tenant | `email/src/sieve/ingest.rs` | when used |
| Identities | identifier, content (signatures) | holder | `x:Email.maxIdentities` | unbounded | data store | tenant | `jmap/src/identity/set.rs` | collected |
| Push subscriptions | network (push URL), credential (keys); **EmailPush can carry email properties to the push URL** | holder | client opt-in; `x:Jmap.maxSubscriptions` | expire within 7 days | data store → **external push service** | tenant | `jmap/src/push/set.rs`; `services/src/state_manager/http.rs` | client-driven |
### 2.2 Records kept by inbuxa's own features
These live in inbuxa's own key space (`SUBSPACE_INBUXA`) unless noted.
`Store::danger_destroy_account` doesn't clear that space, so a row's
"survives account deletion" note matters.
| Source | Categories | Whose | Controlled by | Retention bound | Where | Scope | Written by | Default |
|---|---|---|---|---|---|---|---|---|
| Archived items (undelete) | content, identifier, metadata | holder, correspondent | `x:DataRetention.archiveDeletedItemsFor`; a legal hold forces keeping | each item's `archivedUntil`; **held: year 9999** | registry (data store) + blob store | tenant | `features/src/undelete/records.rs` `insert`, `set_deadline`, `remove_expired`; `undelete/email.rs` `note`; `groupware/src/inbuxa.rs`; `services/.../index.rs` `archive_noted` | **off** |
| Deleted accounts kept | identifier (name, addresses), **credential (the whole account record, password hashes included)**, metadata | holder | `x:DataRetention.archiveDeletedAccountsFor`; a hold forces keeping | `kept_until`; held: year 9999 | data store | tenant | `jmap/src/inbuxa/deleted_account.rs` `keep`; `features/src/undelete/data.rs` `set_kept_account` | **off** |
| Legal holds | identifier, content (reference, description, reasons) | administrator, holder | `sysLegalHold*` | **unbounded, by design** | data store | server (scope may name tenants) | `features/src/hold/mod.rs` `create`, `update`, `pin_account`, `keep_moved` | none |
| Hold exports: records | identifier, content (reason), metadata | administrator | `sysLegalHoldExport` | **unbounded, by design** | data store | server | `hold/mod.rs` `create_export`, `update_export` | none |
| Hold exports: ZIP | content (everything the hold covers) | holder, correspondent | as above | `x:Jmap.uploadTtl`, then the next blob purge (`blobCleanupSchedule`) | blob store; **leaves when downloaded** | server | `jmap/src/inbuxa/hold_export.rs` `build` → `put_jmap_blob` | 1 h + up to a day |
| Account locks, delegations | identifier, content (reason), metadata | holder, administrator | `sysAccountLock*` | until the lock is removed; **no removal on account deletion found** | data store | tenant | `features/src/lock/mod.rs` `set`, `remove` | none |
| Audit log | identifier, **network (sign-in IPs)**, content (setting diffs, secrets redacted; reasons) | administrator, holder (as target), correspondent (e.g. banned IPs) | always on; `inbuxa:AuditSettings.keepForDays` (not a registry setting) | `keepForDays`, minimum 90; records about held accounts kept while held; **kept after account deletion** | data store (hash-chained) | tenant-filterable | `common/src/audit.rs` `audit_append`, `audit_sign_in`, `audit_sign_in_failed`, `audit_foreign_access`, `audit_delegate`; `jmap/src/inbuxa/audit.rs` `recorded`; purge `audit_purge` | **collected, 730 d** |
| Audit exports | as the audit log | as above | `sysAuditExport` | `x:Jmap.uploadTtl`, then the next blob purge | blob store; leaves when downloaded | server or tenant | `jmap/src/inbuxa/audit_log.rs` `build_export` | 1 h + up to a day |
| Masked addresses | identifier, content (description, site), metadata (last message) | holder | `x:Email.maxMaskedAddresses` | the registry object goes with the account; **the address tombstone is kept forever by design**; its `Mm`/`Mc` records aren't cleared on deletion (search found no clearing) | registry + data store | tenant | `features/src/masked_email/ops.rs`; `data.rs` `set_address`, `set_record`, `log_change` | when used (5 max) |
| Legacy-protocol last use | metadata (one timestamp per account and protocol; no IP) | holder | none | **unbounded; not cleared on account deletion** | data store | tenant | `features/src/security/legacy_use.rs` `record` (from `common/src/network/legacy.rs` `admit_legacy_session`) | **collected** |
### 2.3 Mail transport and reports
| Source | Categories | Whose | Controlled by | Retention bound | Where / leaves host | Scope | Written by | Default |
|---|---|---|---|---|---|---|---|---|
| Queue (`x:QueuedMessage`) | content, metadata, network (`receivedFromIp`) | holder, correspondent | always on | delivered, or `x:MtaDeliveryExpirationTtl.expire` | data + blob store; delivered out | server | `smtp/src/queue/spool.rs` `queue`, `save_changes`, `remove` | 3 d max |
| DSNs | content (**original headers**), metadata | correspondent, holder | `x:DsnReportSettings` | sent through the queue | **leaves** | server | `smtp/src/queue/dsn.rs` `send_dsn` | on |
| Received DMARC, TLS-RPT, ARF reports | network (source IPs), identifier (envelope/header from); ARF: **message and headers**, original recipient, user agent | correspondent | `x:ReportSettings.inboundReportAddresses`, `inboundReportForwarding` | `x:DataRetention.holdMtaReportsFor` | data store; also forwarded to the mailbox | tenant (by domain) | `smtp/src/reporting/analysis.rs` | collected, 30 d |
| DMARC aggregate reports we send | network (source IPs), identifier (domains) | correspondent | `x:DmarcReportSettings.aggregateSendFrequency`; only to domains publishing `rua` | kept until sent | data store → **external `rua`** | server | `smtp/src/reporting/dmarc.rs` | daily |
| DMARC/SPF/DKIM failure reports we send | content (**raw headers**), network, identifier | correspondent | `x:DmarcReportSettings.failureSendFrequency`, `x:SpfReportSettings`, `x:DkimReportSettings`; only when requested | through the queue | **external `ruf`** | server | `reporting/dmarc.rs` `send_dmarc_report`; `spf.rs`; `dkim.rs` | on, rate-limited |
| TLS-RPT we send | network (remote MX hosts) | operators of other servers | `x:TlsReportSettings.sendFrequency` | until sent (deletion path not confirmed) | **external** | server | `smtp/src/reporting/tls.rs` | daily |
| Relay (smart host) | everything in the message | holder, correspondent | `x:MtaRoute` (Relay) | receiver's | **external** | server | `smtp/src/outbound` | none |
| Milters, MTA hooks | network (client IP, PTR, HELO), identifier (SASL login), **full message** | holder, correspondent | `x:MtaMilter`, `x:MtaHook` | receiver's | **external** | server | `smtp/src/inbound/milter`, `inbound/hooks/mod.rs` | none |
### 2.4 Telemetry and logs
| Source | Categories | Whose | Controlled by | Retention bound | Where / leaves host | Scope | Written by | Default |
|---|---|---|---|---|---|---|---|---|
| Stored traces (`x:Trace`) | network (IP, EHLO), identifier (envelope addresses, account names), metadata (message-ids, results) | correspondent, holder | `x:TracingStore`; `x:EventTracingLevel` | `x:DataRetention.holdTracesFor` (unset = unbounded) | data store, or a remote PostgreSQL/MySQL/FDB | server (records name accounts) | `common/src/telemetry/tracers/store.rs` `spawn_store_tracer`; purge `purge_spans` | **collected** (first boot inserts `TracingStore::Default` although the struct default is off); 30 d |
| Trace search index | same keywords, lowercased | as above | `x:Search.indexTelemetry`, `indexTracingFields` | purged with the traces | search store | server | `index.rs` `trace_search_document` | indexed |
| Stored metrics | none personal (aggregates) | — | `x:MetricsStore`, `x:Metrics` | `holdMetricsFor` | data store | server | `telemetry/metrics/store.rs` | collected, 90 d |
| Log file | network (**client IP on every in-session line**, from the span), identifier (MAIL FROM, RCPT TO, account names), metadata (message-ids). No subject key exists; bodies only in trace-level raw-I/O events | correspondent, holder, administrator | `x:TracerLog` (`level`, `events`, `path`, `rotate`) | **unbounded: rotated daily, never deleted** | log file (`/var/log/inbuxa`); read back as `x:Log` and fed to Explain | server | `common/src/telemetry/tracers/log.rs` | **collected, info level** |
| Console / journald tracers | as the log file | as above | `x:Tracer` Stdout/Journal | the host's journal policy | stdout / journald (leaves only if the host forwards it) | server | `tracers/stdout.rs`, `journald.rs` | none (recovery mode adds a console tracer) |
| OpenTelemetry tracer | as the log file, every event and span key | as above | `x:Tracer` OtelHttp/OtelGrpc (`level`, `events`, `endpoint`) | receiver's | **external** | server | `tracers/otel.rs` | none |
| Webhooks | as the log file; **can include raw SMTP input (message content) and model replies** — see [Findings](#findings) 1 | as above | `x:WebHook` (`events`, `eventsPolicy`, `url`) | receiver's; retried up to `discardAfter` | **external** | server | `telemetry/webhooks/mod.rs` `post_webhook_events` | none |
| Alerts | contact (the addresses an administrator sets) | administrator | `x:Alert` | as mail | mail; may leave | server | `telemetry/alerts.rs` | none |
| Failed tasks | identifier (account names, e.g. a failed account destroy), content (iTIP messages) | holder, correspondent | none | **not confirmed: failed tasks are rescheduled at `u64::MAX` and no purge was found** | data store | server | `services/src/task_manager/manager.rs` | when a task fails |
### 2.5 Sign-in and network defence
| Source | Categories | Whose | Controlled by | Retention bound | Where | Scope | Written by | Default |
|---|---|---|---|---|---|---|---|---|
| Accounts, groups, lists, domains, tenants (registry) | identifier, contact, metadata | holder | registry | the object's life | data store | tenant | registry writes (`jmap/src/registry/`) | — |
| Passwords, app passwords, API keys | credential (argon2id hashes), network (`allowedIps`), metadata | holder | `x:Authentication.passwordHashAlgorithm`; `maxAppPasswords`, `maxApiKeys`; `expiresAt` | the account's life or `expiresAt` | data store | tenant | `jmap/src/registry/mapping/account.rs` | — |
| OAuth clients | identifier, contact (`contacts`), credential (secret hash) | administrator, third-party developer | `x:OidcProvider.requireClientRegistration` | `expiresAt` or unbounded | data store | tenant | `http/src/auth/oauth/registration.rs` | none |
| OAuth codes and tokens | credential | holder | `x:OidcProvider.*Expiry` | tokens are sealed and stateless (not stored); codes in the in-memory store with TTL | in-memory store | tenant | `http/src/auth/oauth/auth.rs`, `token.rs` | codes 10 min |
| Rate-limit state | network, identifier (login names) | holder, correspondent | `x:Http.rateLimit*`, `x:Imap.maxRequestRate`, `x:Security.*BanRate` | the rate's period | in-memory store | server | `common/src/auth/rate_limit.rs`, `network/security.rs` | on |
| Greylist | identifier (sender/recipient pairs, plain) | correspondent, holder | `x:SpamSettings.greylistFor` | that period | in-memory store | server | `smtp/src/inbound/rcpt.rs` | **off** |
| Automatic IP bans (`x:BlockedIp`) | network | correspondent, holder | `x:Security.authBanRate`, `abuseBanRate`, `loiterBanRate`, `scanBanRate` (on); `*BanPeriod` (**no default**) | **unbounded: a ban with no period never expires**; an expired ban's record goes when settings next load; each ban is also an audit record | data store (registry) | server | `common/src/network/security.rs` `block_ip` | **collected, permanent** |
| Allowed IPs | network | administrator's choice | manual; `expiresAt` | optional | data store | server | registry | none |
### 2.6 Spam filter and AI
| Source | Categories | Whose | Controlled by | Retention bound | Where / leaves host | Scope | Written by | Default |
|---|---|---|---|---|---|---|---|---|
| Training samples (`x:SpamTrainingSample`) | content (**the whole message, pinned**), identifier (`from`), content (`subject`) | correspondent, holder | `x:SpamClassifier.model`, `learnSpamFromTraps`, `learnSpamFromRblHits`, `learnHamFromCard`, `learnHamFromReply`; users' Junk moves | `x:SpamClassifier.holdSamplesFor`; **outlives the user deleting the mail** | data + blob store | tenant for users' samples; **server, unattributed, for SMTP autolearn** | `email/src/message/ingest.rs` `add_spam_sample`; `smtp/src/queue/spool.rs` | **collected, 180 d** |
| Trainer state (`INBUXA_SPAM_TRAIN_DATA`) | metadata (blob hashes with account ids; hashed tokens that include addresses and domains) | correspondent, holder | as above | **unbounded (overwritten on training, never expired)** | blob store | server | `spam-filter/src/modules/classifier.rs` `spam_train` | after 100 + 100 samples |
| Trained model | hashed weights, not directly personal | — | as above | unbounded | blob store | server | same | same |
| AI spam classification | content (subject + up to `inbuxa:AiLimits.maxContentBytes` of text) | correspondent | `x:SpamLlm`, `x:AiModel` (`url`), `inbuxa:AiLimits` | the endpoint's; the verdict is written into the stored message | **external endpoint** (local or hosted; a remote URL only warns) | server | `spam-filter/src/analysis/llm.rs`; `common/src/enterprise/llm.rs` | **off** |
| Sieve `llm_prompt` | whatever a script sends | any | `x:AiModel`; `interactAi` | endpoint's | **external** | per account | `enterprise/llm.rs` `sieve_prompt` | no model |
| Explain | identifier (return path, recipients), network (trace IPs), metadata (remote replies, log details); never contents or raw I/O | correspondent, holder | `inbuxa:AiLimits.explainEnabled`, `explainModelId` | answer cache: memory, per node, 24 h, 1000 answers | **external endpoint** + memory | server | `jmap/src/inbuxa/explanation.rs`; `features/src/ai/explain/memory.rs` | no model |
| DNSBL lookups | network (client IPs), identifier (domains; **SHA-1 of addresses** to msbl.org), metadata (MD5 of URLs) | correspondent | bundled `x:SpamDnsblServer` rules; `x:SpamDnsblSettings` | receiver's | **external (DNS)** | server | spam-filter DNSBL module | **on: 17 of 18 bundled lists** |
| Pyzor | content (SHA-1 digest of normalized body) | correspondent | `x:SpamPyzor` | receiver's | **external: `public.pyzor.org`** | server | `spam-filter/src/modules/pyzor.rs` | **on** |
| URL-shortener expansion | network (fetches links, which can carry per-recipient tracking tokens) | correspondent, holder | bundled redirector list | — | **external (HTTP)** | server | `spam-filter/src/analysis/url.rs` | **on** |
### 2.7 Stores and directories an operator can point elsewhere
| Registry object | Receives | Default |
|---|---|---|
| `x:DataStore` | everything above that says data store | RocksDB, local |
| `x:BlobStore` (S3, Azure, FS, SQL, FDB) | message bodies, attachments, files, samples, exports | the data store |
| `x:SearchStore` (Elasticsearch, Meilisearch, SQL, FDB) | full text of mail, contacts, calendars; trace keywords | the data store |
| `x:InMemoryStore` (Redis) | rate limits, greylist, OAuth codes, Sieve seen ids | the data store |
| `x:TracingStore`, `x:MetricsStore` | traces, metrics | the data store |
| `x:Coordinator` (Kafka, NATS, Zenoh, Redis) | cluster broadcasts (whether personal: not confirmed) | off |
| `x:Directory` (LDAP, SQL, OIDC) | login names, passwords to verify, address lookups | internal |
| `x:AcmeProvider`, `x:DnsServer` | hostnames, a role address; no personal data found | Let's Encrypt if chosen; manual DNS |
## Findings
Facts from the investigation that matter beyond the catalog. Each is a finding,
not a judgment; what to do about each is John's call.
1. **A webhook ignores levels and, left at its defaults, receives every
event, message content included.** `config/telemetry.rs` builds a webhook's
interests with `apply_events` alone: no level check. With the default
`eventsPolicy` (exclude) and no events listed, that is every event type,
including trace-level `smtp.raw-input` (the raw SMTP bytes, `DATA`
included) and `ai.llm-response`. The audit-log docs suggest a webhook to
pass records to a SIEM. Probably upstream behavior; not yet checked
against upstream.
2. **Log files are never deleted.** Daily rotation opens a new file; nothing
removes old ones, and no logrotate configuration ships. At the default
level every in-session line carries the client IP.
3. **Automatic IP bans are permanent.** No `*BanPeriod` has a default, so a
ban never expires. (Corrected 2026-09-28: a ban that does expire stops
blocking, and its record is deleted when settings are next loaded, in
`BlockedIps::parse`; the investigation missed that path.)
4. **Some records outlive the account.** Deleting an account doesn't clear
inbuxa's own key space: legacy-protocol last use, account locks, masked
address records and audit records stay (audit records by design).
5. **A kept deleted account keeps its password hashes** for
`archiveDeletedAccountsFor`, or indefinitely under a legal hold.
6. **Spam training keeps whole messages for 180 days**, after the user
deleted them, and SMTP autolearn keeps them with no account attribution, so
they can't be found by person. The trainer state never expires.
7. **Traces are on in a new install** although the struct default says off:
first boot inserts the tracing and metrics stores.
8. **Data leaves the host by default** through the spam filter (DNSBL queries
with IPs, domains and hashed addresses to 17 lists; Pyzor body digests;
shortener URL fetches) and through reports (DMARC aggregate daily; failure
reports with headers where requested). No AI, webhook, OpenTelemetry,
milter, hook or relay endpoint is configured by default.
9. **Export files outlive their stated lifetime by up to a day.** The link
expires with `uploadTtl`; the bytes stay until the next blob purge
(04:00 daily).
10. **Failed tasks may stay forever** (rescheduled at `u64::MAX`; no purge
found), holding account names or iTIP messages. Not confirmed by a test.
## 3. How the schema is produced, and what the catalog works from
The prompt says not to hand-edit generated files and to work from the source
of truth. What the investigation found:
- **`resources/schema/schema.json.gz` and the registry Rust code
(`crates/registry/src/schema/*.rs`, headed "auto-generated") come from
upstream's generator, which isn't in this repository or upstream's public
tree.** Nothing here regenerates them.
- **The fork edits both by hand.** Since the v0.16.22 import the gz has
changed in 18 commits: the Compliance layout (Audit Log, Legal Holds, Locked
Accounts), the Local AI and Legacy Protocols pages, the fork's permissions
(`sysAudit*`, `sysAccountLock*`, `sysLegalHold*`, `sysAiExplain`), always
paired with `enums.rs`/`enums_impl.rs` edits marked `// inbuxa:`. Scripted
edits go through `tools/fork/renames.py` (`schema_bytes`), which writes the gz
with `gzip.compress(compresslevel=9, mtime=0)` and the `.sha256` sidecar as
unpadded URL-safe base64 of the gz's SHA-256.
- **`/api/schema`** (`crates/http/src/api/mod.rs`, `"schema"` arm) serves the
embedded gz as-is to any authenticated caller, unfiltered; the console
hides what a person can't use from the permission list `/api/account`
returns.
- **The schema carries 150 objects and 2,102 properties** (316 field
schemas). About 34 objects hold data about people, roughly 150–180
properties counting nested structs; about 20 configuration objects carry
about 100 credential or contact properties. inbuxa's own objects
(`inbuxa:AuditEvent`, `LegalHold`, `HoldExport`, `AccountLock`,
`DeletedAccount`, `ProtocolPolicy`, `TenantProtocolPolicy`,
`Explanation`, `AiLimits`, Fastmail `MaskedEmail`) aren't in the schema
at all: they are defined in `crates/jmap-proto/src/object/inbuxa_*.rs`.
So the source of truth for *what exists* is the schema (for registry objects)
and `jmap-proto` (for inbuxa's own). Neither should carry the catalog:
**Decision (Settled 1):** the catalog is a sidecar, and the schema and generated
code are read, never annotated, by it. The fork's existing hand edits to the
schema (layout entries, permissions) continue as today for the Compliance
pages and the new permissions, because the console's navigation comes from
the schema and there is no other way in; each is marked in the commit and
listed in the strip report as now.
## 4. The catalog: format and location
**Location:** `resources/privacy/catalog.toml`, a fork-owned file. The server
embeds it (`include_str!`) and parses it at start, as it does the schema.
**Why a sidecar** rather than annotating in place:
- The schema and registry code are upstream's generated output; annotations
there are overwritten or conflict on every import (SPEC.md §2.2, §2.3).
- Many sources aren't registry objects: log files, telemetry exporters,
inbuxa's key-space records, DNSBL and Pyzor, the in-memory cache. An
in-place annotation has nowhere to put them.
- One file is one diff to review when an import brings new fields, and it is
what the CI check and the strip report read.
**Shape.** Three kinds of entry:
```toml
# A registry object: a default for its properties, and the ones that differ.
[object."x:Account"]
default = "none" # properties not listed hold no personal data
whose = ["holder"]
where = ["data-store"]
scope = "tenant"
retention = "object-life"
[object."x:Account".properties]
name = ["identifier"]
emailAddress = ["identifier"]
aliases = ["identifier"]
description = ["content"]
credentials = ["credential", "network"] # secrets, allowedIps
locale = ["metadata"]
# One of inbuxa's own JMAP objects (not in the schema).
[object."inbuxa:AuditEvent"]
default = "none"
whose = ["administrator", "holder", "correspondent"]
where = ["data-store"]
scope = "tenant"
retention = { setting = "inbuxa:AuditSettings.keepForDays" }
[object."inbuxa:AuditEvent".properties]
actor = ["identifier"]
remoteIp = ["network"]
changes = ["content"]
reason = ["content"]
# A source that is no object: files, exporters, lookups, caches.
[source."log-file"]
categories = ["network", "identifier", "metadata"]
whose = ["correspondent", "holder", "administrator"]
where = ["log-file"]
scope = "server"
enabled_by = ["x:TracerLog.enable"]
captures = ["x:TracerLog.level", "x:TracerLog.events", "x:EventTracingLevel"]
retention = "unbounded"
leaves_host = false
written_by = ["crates/common/src/telemetry/tracers/log.rs"]
[source."spam-dnsbl"]
categories = ["network", "identifier", "metadata"]
whose = ["correspondent"]
where = ["external"]
scope = "server"
enabled_by = ["x:SpamDnsblServer.enable"]
retention = "receiver"
leaves_host = true
written_by = ["crates/spam-filter/src/modules/dnsbl.rs"]
```
Rules: `retention` is `unbounded`, `object-life`, `receiver` (it left the
host), or `{ setting = "<object>.<property>" }`; every setting named is a real
`x:` or `inbuxa:` property. Every object in the schema and every `inbuxa:`
object has an entry, even if only `default = "none"` (configuration objects
with nothing personal). The file holds facts only: no wording about legal
status.
**Evaluation (Phase 3).** Each entry's `enabled_by`, `retention` and
`leaves_host` are evaluated against the live registry: a store pointed at a
remote backend makes everything in it `leaves_host = true`; an unset
retention setting reads `unbounded`; an external endpoint lists its URL's
host as a candidate processor.
## 5. The CI check
`tools/fork/privacy-check.py`, modeled on `name-check.py` and
`notice-check.py` (plain Python, no dependencies, exit 1 on findings). Wired
into the `fork-checks` job in `.gitea/workflows/ci.yml`, beside them. It fails
when:
1. an object in `schema.json.gz` `fields`, or an `inbuxa:` object in
`crates/jmap-proto/src/object/`, has no catalog entry;
2. a property whose schema format is `emailAddress`, `ipAddress`,
`ipNetwork`, `secret`/`secretText`, or whose type is `x:SecretKey*` or
`x:HttpAuth`, is covered only by an object's `default = "none"` — those
must be classified explicitly, so a new personal field can't hide behind a
default;
3. an entry names an object, property or setting that no longer exists
(stale);
4. an entry uses a category, subject or location outside the vocabulary.
`tools/fork/strip.py` gains `privacy_flags(tree)`, beside `schema_flags`:
objects and properties new in an import and not in the catalog, reported in
`STRIP-REPORT.md` under "Unclassified in the privacy catalog", the way
Enterprise flags are reported. Informational, as the Enterprise list is; CI
is what fails.
Tests (Phase 2): the check passes on main; fails on a synthetic unclassified
field; fails on a synthetic `emailAddress` property covered only by a
default; fails on a stale entry.
## Default profile
**Today's defaults in a new install** that collect or send personal data:
| Source | Collected | Retention |
|---|---|---|
| Traces + trace index | yes | 30 days |
| Log file | yes, info level | **unbounded** |
| Audit log | yes | 730 days |
| Spam training samples (whole messages) | yes | 180 days |
| Spam trainer state | yes, after 200 samples | **unbounded** |
| Automatic IP bans | yes | **unbounded** |
| Legacy-protocol last use | yes | **unbounded** |
| Received DMARC/TLS/ARF reports | yes | 30 days |
| DNSBL lookups (17 lists), Pyzor, URL-shortener fetches | yes | leave the host |
| Outbound DMARC aggregate and failure reports | yes, where the domain asks | leave the host |
| Trash | yes | 30 days |
| Undelete, deleted-account keeping | **off** | — |
| AI classification, Explain | **off** (no model) | — |
| Greylisting | **off** | — |
**Changes for new installs only.** Settled (5): all seven are built, in
Phase 3; existing servers keep their settings. D7 is also covered by the bug
fix for finding 1 (Settled 6).
**As built (2026-09-28).** D2, D3, D4, D6 are written on first boot of a new
install only (no roles yet), each singleton read and written back whole
(`manager/defaults.rs`, `new_install_privacy_defaults`); expired bans are
also purged daily (`purge_expired_blocked_ips`). D7 changes the default for
webhooks created from now on; stored webhooks keep theirs (the registry
stores every field). **Settled (John, 2026-09-28):** D1 becomes a
fork-owned log-retention setting, kept in inbuxa's own storage as audit
retention is (not a field on `x:TracerLog`, which is also stored inside
`x:Bootstrap` with fields after it, so a new field would change that
object's stored format); new installs 30 days, existing servers keep every
file as today. D5 is built after the v0.16.24 import lands, on its reworked
spam-rules loader, which keeps each blocklist's on/off state.
| # | Change | Trade-off |
|---|---|---|
| D1 | A **log retention** setting on `x:TracerLog` (delete rotated files older than N days), default 30 days for new installs | Needs code (a new field, so a schema edit); older logs gone for troubleshooting; operators wanting longer set it |
| D2 | A default **ban period**, e.g. 30 days, for the four ban rates, and a purge of expired `x:BlockedIp` | A persistent attacker is re-banned after expiry rather than kept out; permanent bans of shared/NAT addresses stop being permanent |
| D3 | **Spam sample retention** 180 → 90 days | Fewer samples to retrain from; the classifier is retrained regularly, so the effect on accuracy is likely small (not measured) |
| D4 | **Pyzor off** by default | One fewer spam signal; no body digests leave the host |
| D5 | **msbl.org EBL off** by default (the list that receives hashed addresses) | One fewer signal on address-based spam; no hashed addresses leave the host |
| D6 | **Trace retention** 30 → 14 days | Shorter delivery history in **Emails › History** and for support |
| D7 | Webhooks default to `eventsPolicy = include` with no events, and never receive raw-I/O events unless listed by name | A webhook does nothing until events are chosen; closes finding 1 |
Separately, **not defaults but gaps** a later spec should close (listed so
they aren't lost): clearing inbuxa's own records on account deletion
(finding 4); an expiry for the spam trainer state; a purge for failed tasks
(finding 10, once confirmed); removing export blobs when their link expires
(finding 9).
## 6. The inventory, over JMAP (Phase 3)
Under `urn:inbuxa:jmap`, read-only.
### `inbuxa:DataInventory/get`
A singleton, evaluated on request.
Request: `{ accountId, ids: null | ["singleton"], properties? }`.
Response `list[0]`:
```json
{
"id": "singleton",
"evaluatedAt": "2026-09-28T10:00:00Z",
"catalogVersion": "2026.9.28.3",
"sources": [
{
"id": "log-file",
"kind": "source",
"categories": ["network", "identifier", "metadata"],
"whose": ["correspondent", "holder", "administrator"],
"where": ["log-file"],
"collected": true,
"retention": { "unbounded": true },
"leavesHost": false,
"scope": "server",
"controlledBy": ["x:TracerLog.enable", "x:TracerLog.level"],
"endpoints": []
},
{
"id": "x:Trace",
"kind": "object",
"categories": ["network", "identifier", "metadata"],
"collected": true,
"retention": { "days": 30, "setting": "x:DataRetention.holdTracesFor" },
"leavesHost": false,
"scope": "server",
"endpoints": []
}
],
"processors": [
{ "host": "public.pyzor.org", "receives": ["content"], "sources": ["spam-pyzor"] }
]
}
```
`collected` is false when the source's switch is off; `retention` reads the
live setting; `leavesHost` is true when the source's location is a remote
store or an external endpoint; `processors` lists each external host once,
with what it receives — candidates, since whether a host is a processor in
law is the operator's determination.
**Permission:** `sysComplianceGet`. **Tenant scoping:** inside a tenant,
`sources` holds only `scope = "tenant"` entries, evaluated with the tenant's
own settings where it has them (its protocol switches, its domains' stores
are the server's); server-scope sources and `processors` are left out, because
they describe the whole server. A tenant principal without the permission gets
`forbidden`.
### `inbuxa:InventorySnapshot/get`, `/query`
A dated copy of the evaluated inventory, taken when a setting the catalog
references changes (hooked where the audit log already sees registry writes)
and at most once a day otherwise. `query` filters by `after`/`before`,
newest first. Fields: `id`, `takenAt`, `trigger` (`setting-changed` with the
setting, or `daily`), `summary` (counts: sources collected, unbounded,
leaving the host, processors), and on `get` the full inventory as above.
Kept for `inbuxa:AuditSettings.keepForDays`, so history is as long as the
audit log's. Same permission and tenant scoping as the inventory.
## 7. The compliance role
### What it holds
A new built-in role, **Compliance Officer**, at server level:
| Permission | New? | Gives |
|---|---|---|
| `sysComplianceGet` | new | Overview, Data inventory, snapshots |
| `sysAuditGet`, `sysAuditExport` | existing | read and export the audit log, verify the chain |
| `sysLegalHoldGet`, `sysLegalHoldCreate`, `sysLegalHoldUpdate`, `sysLegalHoldExport` | existing | place, widen, release and export holds |
| `sysAccountLockGet` | existing | see locks and delegations |
| read (`*Get`, `*Query`) on accounts, groups, lists, domains, tenants, roles | existing | know who and what the records refer to |
It holds **no** `*Create`/`*Update`/`*Destroy` on registry objects, no
`sysAuditSettingsUpdate`, no `impersonate`, no `fetchAnyBlob`. It can't
change a server setting.
**Settled (2):** the officer *places and releases* holds, because that is
the job; every placing, widening and release is already recorded with its
reason, under the officer's own identity (AU-12), and a release can't be
undone silently (LH-10).
A **Tenant Compliance Officer** (inside a tenant: `sysComplianceGet`,
`sysAuditGet`, `sysAuditExport`, `sysAccountLockGet`, reads) is built with
it (Settled 3). It has no hold permissions, since legal holds are
server-only by the tenant ceiling (LH-13), and it sees the tenant's slice of
the inventory only (§6).
### Reaching existing servers
New permissions get ids 673 onward and `COUNT` grows (`enums.rs`,
`enums_impl.rs`, and the schema's `Permission` enum, a marked hand edit as
before). `DefaultPermissions` places `sysComplianceGet` with superusers;
`granted_permissions.rs` `ADMIN_GRANTS` adds it once to stored administrator
roles, as it did for the audit and hold permissions. The **role object**
itself is seeded only on a fresh install (`defaults.rs`, `count == 0`); for
existing servers a one-time step modeled on `granted_permissions.rs` creates
it once, recorded so that deleting it sticks. That step is new code.
### Keeping administrators reviewable
What already holds: every administrator's change to the registry and to
inbuxa's own objects is recorded before it is allowed, under their identity,
with a reason where required; sign-ins by anyone holding `sys*` permissions
are recorded; records can't be deleted over JMAP; the chain is verifiable;
changing retention is itself recorded.
What this adds: the compliance officer reads all of it, server administrators'
actions included, without being an administrator. The Overview (§8) puts on
top anything that weakens review: audit retention shortened, a tracer or the
audit export webhook removed, a change to who holds the compliance or
administrator roles, a hold released — each with who and when.
**Settled (4):** a server administrator can still shorten audit retention to
90 days; that is recorded and surfaced on the Overview, not gated on a second
person, since a server may have only one. Anyone with shell access can edit
the store directly; that is outside what the server can review.
## 8. The Compliance section in the console
The navigation comes from the schema's `layouts` (`/api/schema`); a
hand-built page is a `CustomComponent/<Name>` link there, a render branch in
`MainContent.tsx`, and a visibility rule in `layout.ts` `checkSpecialLink`.
The section shows when at least one of its pages is visible to the person
(Phase 4 confirms the container behaves so with every child hidden).
Planned navigation, in order:
| Page | This task | Built as | Visible with |
|---|---|---|---|
| **Overview** | **builds** | hand-built (`CustomComponent/ComplianceOverview`) | `sysComplianceGet` |
| **Data inventory** | **builds** | hand-built (`CustomComponent/DataInventory`) | `sysComplianceGet` |
| Retention | later spec | likely hand-built over `x:DataRetention`, `x:SpamClassifier`, `x:TracerLog` and `inbuxa:AuditSettings`, since they're spread across objects | — |
| Legal holds | exists | hand-built | `sysLegalHoldGet` |
| Audit log | exists | hand-built | `sysAuditGet` |
| Locked accounts | exists | hand-built | `sysAccountLockGet` |
| Data subject requests | later spec | hand-built | — |
| Records and documents | later spec | could be schema-driven if it becomes a registry object | — |
| Jurisdiction packs | later spec | hand-built | — |
**Decision:** later pages are left out of the navigation until
built, not shown disabled: a disabled entry reads as a feature that exists.
**Overview** shows the latest snapshot's findings as facts ("Log files are
kept with no limit", "Sent to a host outside this network: public.pyzor.org"),
the review items from §7, and the snapshot history (when and why the inventory
changed). **Data inventory** lists the evaluated sources, filterable by
source, category and location, with each external endpoint listed as a
candidate processor. UI text states facts and never claims a legal standard is met.
## Contradictions with the docs and SPEC.md
1. **docs, Security › "What the AI features do and do not send":** "message
content is never written to the logs". True at the default level. The
model's reply to the spam classifier is logged as `ai.llm-response` at
trace level (up to 1,024 characters, which can restate the message), so a
trace-level tracer writes it, and any webhook at its defaults receives it,
along with raw SMTP input (finding 1).
2. **docs, Audit log:** suggests a webhook to pass records to a SIEM, without
saying a webhook's `level` is ignored and its default policy sends every
event.
3. **docs, Monitoring:** "rotated daily" is right, but doesn't say old files
are never removed (finding 2). Not strictly a contradiction; an omission
that matters here.
4. **SPEC.md §2.3** calls `schema.json.gz` upstream's published schema and
the checklist for rebuilt features. It doesn't say the fork now hand-edits
it (layout, permissions); §2.2's strip step would overwrite those edits on
an import unless they're re-applied. Worth a sentence in SPEC.md.
5. **The schema's `TracingStore`/`MetricsStore` defaults** say disabled; first
boot turns both on. The docs describe the real behavior; the schema default
(what the console shows as the default) doesn't.
The rest checked out: tracing 30 days, metrics 90 days, the Explain cache (in
memory, a day), audit retention (two years, minimum 90), undelete off by
default.
## Settled
John's answers, 2026-09-28, to the questions this spec asked:
1. **The catalog is a sidecar**, `resources/privacy/catalog.toml`; the
Compliance pages and the new permission are hand-added to the schema as
the audit and hold work did.
2. **The Compliance Officer places and releases legal holds.** It's their
role.
3. **The Tenant Compliance Officer is built now**, with the server role
(Phase 3), not after data subject requests.
4. **Shortening audit retention is recorded and surfaced**, not gated on a
second person: a server may have only one.
5. **All seven defaults, D1–D7, for new installs.** Built in Phase 3.
6. **Finding 1 is fixed now**, as a bug, separately from this work.
7. **Snapshots are kept as long as the audit log** (`keepForDays`).
## Phases
1. **This spec.** Stop for approval.
2. **The catalog and its check:** `resources/privacy/catalog.toml`,
`tools/fork/privacy-check.py` in `fork-checks`, the strip report section,
and the tests in §5.
3. **Server:** `sysComplianceGet`, the Compliance Officer and Tenant
Compliance Officer roles (with the existing-server step),
`inbuxa:DataInventory`, `inbuxa:InventorySnapshot`, and defaults D1–D7
for new installs; tests with several configurations (defaults, a remote
store, a hosted AI endpoint, telemetry off), tenant scoping, refusal
without the permission.
4. **Console:** Overview and Data inventory, the navigation entries, a PR
linking this spec.
File diff suppressed because it is too large Load Diff
Binary file not shown.
+1 -1
View File
@@ -1 +1 @@
6SH8D-bnOr8ynvSjjkUDkpWWy-QxJQdfZc3KaDlE-hk ZjiCQWnwHbujcF7hNolPZQa-VtrNTNfc_HbSuZZ_TiY
+15
View File
@@ -221,6 +221,21 @@ pub async fn test(test: &mut TestServer) {
) )
.await; .await;
// inbuxa: personal-data catalog, D2: the daily clean-up removes the
// expired ban's record, without waiting for settings to reload
test.server.purge_expired_blocked_ips().await.unwrap();
assert_eq!(
admin
.registry_query_ids(
ObjectType::BlockedIp,
[(Property::Address, "10.0.0.2")],
Vec::<&str>::new(),
)
.await,
Vec::<Id>::new(),
"the expired ban's record is gone"
);
// Make sure the IP remains unblocked after reload // Make sure the IP remains unblocked after reload
admin.registry_create_object(Action::ReloadBlockedIps).await; admin.registry_create_object(Action::ReloadBlockedIps).await;
validate_password_with_ip( validate_password_with_ip(
+19
View File
@@ -72,6 +72,25 @@ tools/fork/notice-check.py --fix # add it where it's missing
Run `--fix` after resolving an upstream merge: a conflict resolved by taking Run `--fix` after resolving an upstream merge: a conflict resolved by taking
upstream's side can drop a notice the file had. upstream's side can drop a notice the file had.
## privacy-check.py
Fails when the personal-data catalog (`resources/privacy/catalog.toml`) and
the code disagree: an object in the schema or one of inbuxa's own JMAP
objects with no entry, a property the schema types as an address, an IP or
a secret left to its object's default, or an entry naming an object,
property, setting or code path that no longer exists. CI runs it beside the
name check, with its tests (`python3 -m unittest discover -s tools/fork/tests`).
See `docs/spec/features/personal-data-catalog.md`.
```bash
tools/fork/privacy-check.py # exit 1 on any finding
tools/fork/privacy-check.py --unlisted # starting entries for what's missing
```
After an upstream import, the strip report lists what is new and unclassified
under "Unclassified in the privacy catalog". `--unlisted` types each from the
schema alone; read the field's description before trusting it.
## record-compat.py ## record-compat.py
Records what the `*_compat` tests compare against, from the Enterprise Records what the `*_compat` tests compare against, from the Enterprise
+228
View File
@@ -0,0 +1,228 @@
#!/usr/bin/env python3
# SPDX-FileCopyrightText: 2026 Coffey Labs
# SPDX-License-Identifier: AGPL-3.0-only
"""
Fail when the personal-data catalog and the code disagree.
tools/fork/privacy-check.py # check; exit 1 on any finding
tools/fork/privacy-check.py --unlisted # print catalog entries for what's missing
The catalog (`resources/privacy/catalog.toml`, spec
`docs/spec/features/personal-data-catalog.md`) says, for every object and
source, what personal data it can hold. An upstream import can bring objects
and fields nobody has classified, and a refactor can leave the catalog naming
things that are gone; either way the catalog stops being true without anyone
noticing, so this runs in CI on every push and pull request. It fails when:
1. an object in the schema's `fields`, or one of inbuxa's own JMAP objects,
has no catalog entry;
2. a property the schema types as an email address, an IP address or
network, or a secret is covered only by its object's `default` -- it
must be listed, so a new personal field can't hide behind a default;
3. an entry names an object, property, setting or code path that doesn't
exist (stale);
4. an entry uses a word outside the catalog's own vocabulary.
When it fails on a new object or field, classify it: `--unlisted` prints a
starting entry for each, typed from the schema alone. Read the property's
description before trusting it.
"""
import argparse
import gzip
import json
import os
import re
import sys
import tomllib
ROOT = os.path.dirname(os.path.dirname(os.path.dirname(os.path.abspath(__file__))))
SCHEMA = 'resources/schema/schema.json.gz'
CATALOG = 'resources/privacy/catalog.toml'
OBJECTS_DIR = 'crates/jmap-proto/src/object'
METHODS = 'crates/jmap-proto/src/request/method.rs'
SENSITIVE_FORMATS = {
'emailAddress': 'identifier',
'ipAddress': 'network',
'ipNetwork': 'network',
'secret': 'credential',
'secretText': 'credential',
}
INBUXA_OBJECT = re.compile(r'"(inbuxa:[A-Z][A-Za-z]*)"')
PROPERTY_NAME = re.compile(r'=> "([a-z][A-Za-z0-9]*)"')
def sensitive(type_):
"""The category a property's schema type alone implies, or None."""
fmt = type_.get('format') or (type_.get('class') or {}).get('format')
if fmt in SENSITIVE_FORMATS:
return SENSITIVE_FORMATS[fmt]
name = type_.get('objectName') or (type_.get('class') or {}).get('objectName') or ''
if name.startswith(('x:SecretKey', 'x:SecretText')) or name == 'x:HttpAuth':
return 'credential'
return None
def load_schema(root):
with gzip.open(os.path.join(root, SCHEMA)) as f:
return json.load(f)['fields']
def load_inbuxa_objects(root):
"""inbuxa's own objects, from the method names jmap-proto parses."""
with open(os.path.join(root, METHODS), encoding='utf-8') as f:
return set(INBUXA_OBJECT.findall(f.read()))
def inbuxa_properties(root, file_name):
"""The property names a jmap-proto object file maps, or None if it's gone."""
path = os.path.join(root, OBJECTS_DIR, file_name)
if not os.path.isfile(path):
return None
with open(path, encoding='utf-8') as f:
return set(PROPERTY_NAME.findall(f.read()))
def findings(root, catalog):
"""Everything wrong with `catalog` against the tree at `root`."""
fields = load_schema(root)
inbuxa = load_inbuxa_objects(root)
vocab = catalog.get('vocabulary', {})
objects = catalog.get('object', {})
sources = catalog.get('source', {})
out = []
inbuxa_props = {}
def vocabulary(where, key, values):
allowed = set(vocab.get(key, []))
for value in values if isinstance(values, list) else [values]:
if value not in allowed:
out.append(f'{where}: "{value}" is not in vocabulary.{key}')
def setting_exists(where, setting):
obj, _, prop = setting.partition('.')
if obj in fields:
if prop not in fields[obj].get('properties', {}):
out.append(f'{where}: setting {setting} names no property of {obj}')
elif obj in inbuxa:
props = inbuxa_props.get(obj)
if props is not None and prop not in props:
out.append(f'{where}: setting {setting} names no property of {obj}')
else:
out.append(f'{where}: setting {setting} names no object')
def common(where, entry):
for key in ('whose', 'where', 'scope'):
if key in entry:
vocabulary(where, key, entry[key])
retention = entry.get('retention')
if isinstance(retention, dict):
setting_exists(where, retention.get('setting', ''))
elif retention is not None:
vocabulary(where, 'retention', retention)
for key in ('enabled_by', 'captures'):
for setting in entry.get(key, []):
setting_exists(where, setting)
# inbuxa objects' properties first, so settings can name them
for name, entry in objects.items():
if name.startswith('inbuxa:'):
inbuxa_props[name] = inbuxa_properties(root, entry.get('file', ''))
# 1: nothing unlisted
for name in sorted(fields):
if name not in objects:
out.append(f'{name}: schema object has no catalog entry')
for name in sorted(inbuxa):
if name not in objects:
out.append(f'{name}: inbuxa object has no catalog entry')
for name, entry in sorted(objects.items()):
props = entry.get('properties', {})
if name.startswith('inbuxa:'):
known = inbuxa_props.get(name)
if name not in inbuxa:
out.append(f'{name}: no such inbuxa object (stale)')
if known is None:
out.append(f'{name}: file "{entry.get("file", "")}" not found in {OBJECTS_DIR}')
known = set()
elif name in fields:
known = set(fields[name].get('properties', {}))
else:
out.append(f'{name}: no such schema object (stale)')
continue
if entry.get('default') != 'none':
out.append(f'{name}: default must be "none"')
for prop, categories in props.items():
if prop not in known:
out.append(f'{name}.{prop}: no such property (stale)')
vocabulary(f'{name}.{prop}', 'categories', categories)
common(name, entry)
# 2: typed-sensitive properties are listed
if name in fields:
for prop, spec in fields[name].get('properties', {}).items():
if prop not in props and sensitive(spec['type']):
out.append(f'{name}.{prop}: typed as {sensitive(spec["type"])} but not listed')
for name, entry in sorted(sources.items()):
where = f'source "{name}"'
vocabulary(where, 'categories', entry.get('categories', []))
common(where, entry)
if 'leaves_host' not in entry:
out.append(f'{where}: leaves_host is missing')
for path in entry.get('written_by', []):
if not os.path.exists(os.path.join(root, path)):
out.append(f'{where}: written_by {path} doesn\'t exist (stale)')
return out
def unlisted(root, catalog):
"""Starting catalog entries for unlisted objects and properties."""
fields = load_schema(root)
objects = catalog.get('object', {})
lines = []
for name in sorted(fields):
entry = objects.get(name)
listed = (entry or {}).get('properties', {})
missing = {
p: sensitive(spec['type'])
for p, spec in fields[name].get('properties', {}).items()
if p not in listed and sensitive(spec['type'])
}
if entry is None or missing:
lines.append(f'[object.{json.dumps(name)}]' if entry is None else f'# add to {name}:')
if entry is None:
lines.append('default = "none"')
if missing:
if entry is None:
lines.append(f'[object.{json.dumps(name)}.properties]')
for prop, category in sorted(missing.items()):
lines.append(f'{prop} = ["{category}"]')
lines.append('')
return lines
def main(argv=None):
parser = argparse.ArgumentParser(description=__doc__.split('\n')[1])
parser.add_argument('--unlisted', action='store_true', help='print entries for what is missing')
parser.add_argument('--root', default=ROOT, help=argparse.SUPPRESS)
args = parser.parse_args(argv)
with open(os.path.join(args.root, CATALOG), 'rb') as f:
catalog = tomllib.load(f)
if args.unlisted:
print('\n'.join(unlisted(args.root, catalog)))
return 0
found = findings(args.root, catalog)
if found:
print(f'privacy check: {len(found)} finding(s) in {CATALOG}:')
for line in found:
print(f' {line}')
print('Classify new objects and fields (--unlisted helps); remove what no longer exists.')
return 1
print(f'privacy check: clean ({len(catalog.get("object", {}))} objects, '
f'{len(catalog.get("source", {}))} sources).')
return 0
if __name__ == '__main__':
sys.exit(main())
+43 -1
View File
@@ -373,6 +373,39 @@ def schema_flags(tree):
return {'objects': objects, 'fields': fields} return {'objects': objects, 'fields': fields}
def privacy_flags(tree):
"""Objects and properties new in this import that the personal-data
catalog doesn't classify (docs/spec/features/personal-data-catalog.md).
Informational, like the Enterprise flags: privacy-check.py is what fails
CI once the import is merged."""
import importlib.util
import tomllib
here = Path(__file__).resolve().parent
catalog_path = here.parents[1] / 'resources' / 'privacy' / 'catalog.toml'
current_path = here.parents[1] / 'resources' / 'schema' / 'schema.json.gz'
path = tree / 'resources' / 'schema' / 'schema.json.gz'
if not (path.is_file() and catalog_path.is_file() and current_path.is_file()):
return None
spec = importlib.util.spec_from_file_location('privacy_check', here / 'privacy-check.py')
check = importlib.util.module_from_spec(spec)
spec.loader.exec_module(check)
catalog = tomllib.loads(catalog_path.read_text(encoding='utf-8')).get('object', {})
current = json.loads(gzip.decompress(current_path.read_bytes())).get('fields', {})
upstream = json.loads(gzip.decompress(path.read_bytes())).get('fields', {})
objects = sorted(name for name in upstream if name not in catalog)
fields = []
for name, spec_ in upstream.items():
if name not in catalog:
continue
listed = catalog[name].get('properties', {})
known = current.get(name, {}).get('properties', {})
for prop, p in spec_.get('properties', {}).items():
if prop not in known and prop not in listed:
typed = check.sensitive(p.get('type', {}))
fields.append(f'{name}.{prop}' + (f' ({typed})' if typed else ''))
return {'objects': objects, 'fields': sorted(fields)}
def remaining_hooks(tree): def remaining_hooks(tree):
gates, checks = {}, {} gates, checks = {}, {}
for path in tree.rglob('*.rs'): for path in tree.rglob('*.rs'):
@@ -456,6 +489,8 @@ def write_report(out_dir, report):
] ]
if r['schema']: if r['schema']:
md.append(f'- Upstream schema flags {len(r["schema"]["objects"])} objects and {len(r["schema"]["fields"])} fields as Enterprise') md.append(f'- Upstream schema flags {len(r["schema"]["objects"])} objects and {len(r["schema"]["fields"])} fields as Enterprise')
if r.get('privacy'):
md.append(f'- Unclassified in the privacy catalog: {len(r["privacy"]["objects"])} objects and {len(r["privacy"]["fields"])} new fields')
md += ['', '## Removed files', ''] + [f'- `{f}`' for f in r['removed_files']] md += ['', '## Removed files', ''] + [f'- `{f}`' for f in r['removed_files']]
md += ['', '## Removed snippets', ''] + [f'- `{f}`: {n}' for f, n in r['removed_snippets'].items()] md += ['', '## Removed snippets', ''] + [f'- `{f}`: {n}' for f, n in r['removed_snippets'].items()]
md += ['', '## Dangling module declarations removed', ''] + [f'- `{d["file"]}:{d["line"]}`: `mod {d["module"]}` ({" / ".join(d["lines"])})' for d in r['dangling_mods']] md += ['', '## Dangling module declarations removed', ''] + [f'- `{d["file"]}:{d["line"]}`: `mod {d["module"]}` ({" / ".join(d["lines"])})' for d in r['dangling_mods']]
@@ -463,6 +498,13 @@ def write_report(out_dir, report):
if r['schema']: if r['schema']:
md += ['', '## Flagged Enterprise in upstream\'s schema', '', '**Objects:** ' + ', '.join(f'`{o}`' for o in r['schema']['objects']), md += ['', '## Flagged Enterprise in upstream\'s schema', '', '**Objects:** ' + ', '.join(f'`{o}`' for o in r['schema']['objects']),
'', '**Fields:** ' + ', '.join(f'`{f}`' for f in r['schema']['fields'])] '', '**Fields:** ' + ', '.join(f'`{f}`' for f in r['schema']['fields'])]
if r.get('privacy'):
md += ['', '## Unclassified in the privacy catalog', '',
'New in this import and not in `resources/privacy/catalog.toml`. `tools/fork/privacy-check.py` '
'fails CI on these once merged; `--unlisted` prints a starting entry. A type in brackets is what '
'the schema alone says the field holds.', '',
'**Objects:** ' + (', '.join(f'`{o}`' for o in r['privacy']['objects']) or 'none'),
'', '**Fields:** ' + (', '.join(f'`{f}`' for f in r['privacy']['fields']) or 'none')]
md += ['', '## Third-party code', '', md += ['', '## Third-party code', '',
'Comments in the stripped tree that name another copyright holder, another license, or a source the ' 'Comments in the stripped tree that name another copyright holder, another license, or a source the '
'code came from. Files marked **new** aren\'t in THIRD-PARTY.md yet.', ''] 'code came from. Files marked **new** aren\'t in THIRD-PARTY.md yet.', '']
@@ -537,7 +579,7 @@ def main():
'removed_files': removed_files, 'removed_snippets': removed_snippets, 'removed_files': removed_files, 'removed_snippets': removed_snippets,
'cargo_edits': edits, 'dangling_mods': dangling, 'problems': problems, 'cargo_edits': edits, 'dangling_mods': dangling, 'problems': problems,
'feature_gates': gates, 'edition_checks': checks, 'feature_gates': gates, 'edition_checks': checks,
'schema': schema_flags(tree), 'third_party': others, 'third_party_unlisted': new_others, 'schema': schema_flags(tree), 'privacy': privacy_flags(tree), 'third_party': others, 'third_party_unlisted': new_others,
'renames': renames, 'build': build, 'ossify_log': log, 'renames': renames, 'build': build, 'ossify_log': log,
} }
write_report(args.out, report) write_report(args.out, report)
+177
View File
@@ -0,0 +1,177 @@
# SPDX-FileCopyrightText: 2026 Coffey Labs
# SPDX-License-Identifier: AGPL-3.0-only
"""Tests for tools/fork/privacy-check.py: python3 -m unittest discover tools/fork/tests"""
import gzip
import importlib.util
import json
import os
import tempfile
import tomllib
import unittest
HERE = os.path.dirname(os.path.abspath(__file__))
spec = importlib.util.spec_from_file_location('privacy_check', os.path.join(HERE, '..', 'privacy-check.py'))
check = importlib.util.module_from_spec(spec)
spec.loader.exec_module(check)
VOCAB = """
[vocabulary]
categories = ["identifier", "contact", "network", "content", "metadata", "credential"]
whose = ["holder", "correspondent", "administrator"]
where = ["data-store", "blob-store", "search-store", "in-memory-store", "memory", "log-file", "external"]
scope = ["tenant", "server"]
retention = ["unbounded", "object-life", "receiver"]
"""
GOOD = VOCAB + """
[object."x:Widget"]
default = "none"
whose = ["holder"]
where = ["data-store"]
scope = "tenant"
retention = { setting = "x:Widget.keepFor" }
[object."x:Widget".properties]
owner = ["identifier"]
[object."inbuxa:Gadget"]
file = "inbuxa_gadget.rs"
default = "none"
[object."inbuxa:Gadget".properties]
remoteIp = ["network"]
[source."widget-log"]
categories = ["network"]
where = ["log-file"]
scope = "server"
retention = "unbounded"
enabled_by = ["x:Widget.enable", "inbuxa:Gadget.remoteIp"]
leaves_host = false
written_by = ["crates/widget.rs"]
"""
def make_tree(fields):
root = tempfile.mkdtemp()
os.makedirs(os.path.join(root, 'resources/schema'))
with gzip.open(os.path.join(root, check.SCHEMA), 'wt') as f:
json.dump({'fields': fields}, f)
os.makedirs(os.path.join(root, check.OBJECTS_DIR))
with open(os.path.join(root, check.OBJECTS_DIR, 'inbuxa_gadget.rs'), 'w') as f:
f.write('Property::RemoteIp => "remoteIp",\nProperty::Id => "id",\n')
os.makedirs(os.path.join(root, 'crates/jmap-proto/src/request'), exist_ok=True)
with open(os.path.join(root, check.METHODS), 'w') as f:
f.write('"inbuxa:Gadget" => MethodObject::Gadget,\n')
with open(os.path.join(root, 'crates/widget.rs'), 'w') as f:
f.write('')
return root
def prop(type_, fmt=None):
t = {'type': type_}
if fmt:
t['format'] = fmt
return {'type': t}
FIELDS = {
'x:Widget': {'properties': {
'owner': prop('string', 'emailAddress'),
'enable': prop('boolean'),
'keepFor': prop('number', 'duration'),
'label': prop('string', 'string'),
}},
}
class PrivacyCheck(unittest.TestCase):
def run_check(self, catalog, fields=FIELDS):
return check.findings(make_tree(fields), tomllib.loads(catalog))
def test_the_repository_passes(self):
with open(os.path.join(check.ROOT, check.CATALOG), 'rb') as f:
catalog = tomllib.load(f)
self.assertEqual(check.findings(check.ROOT, catalog), [])
def test_a_consistent_catalog_passes(self):
self.assertEqual(self.run_check(GOOD), [])
def test_an_unclassified_object_fails(self):
fields = dict(FIELDS, **{'x:Sprocket': {'properties': {'size': prop('number', 'size')}}})
found = self.run_check(GOOD, fields)
self.assertEqual(found, ['x:Sprocket: schema object has no catalog entry'])
def test_an_address_hidden_behind_the_default_fails(self):
fields = {'x:Widget': {'properties': dict(FIELDS['x:Widget']['properties'],
contactEmail=prop('string', 'emailAddress'))}}
found = self.run_check(GOOD, fields)
self.assertEqual(found, ['x:Widget.contactEmail: typed as identifier but not listed'])
def test_a_secret_in_a_set_or_object_counts(self):
fields = {'x:Widget': {'properties': dict(
FIELDS['x:Widget']['properties'],
ips={'type': {'type': 'set', 'class': {'type': 'string', 'format': 'ipNetwork'}}},
key={'type': {'type': 'object', 'objectName': 'x:SecretKey'}},
)}}
found = self.run_check(GOOD, fields)
self.assertIn('x:Widget.ips: typed as network but not listed', found)
self.assertIn('x:Widget.key: typed as credential but not listed', found)
def test_a_stale_property_fails(self):
found = self.run_check(GOOD.replace('owner = ["identifier"]', 'owner = ["identifier"]\ngone = ["content"]'))
self.assertEqual(found, ['x:Widget.gone: no such property (stale)'])
def test_a_stale_object_fails(self):
found = self.run_check(GOOD + '\n[object."x:Removed"]\ndefault = "none"\n')
self.assertEqual(found, ['x:Removed: no such schema object (stale)'])
def test_a_stale_setting_fails(self):
found = self.run_check(GOOD.replace('x:Widget.keepFor', 'x:Widget.keepForever'))
self.assertEqual(found, ['x:Widget: setting x:Widget.keepForever names no property of x:Widget'])
def test_a_stale_code_path_fails(self):
found = self.run_check(GOOD.replace('crates/widget.rs', 'crates/gone.rs'))
self.assertEqual(found, ['source "widget-log": written_by crates/gone.rs doesn\'t exist (stale)'])
def test_an_unlisted_inbuxa_object_fails(self):
catalog = VOCAB + """
[object."x:Widget"]
default = "none"
[object."x:Widget".properties]
owner = ["identifier"]
"""
found = self.run_check(catalog)
self.assertEqual(found, ['inbuxa:Gadget: inbuxa object has no catalog entry'])
def test_words_outside_the_vocabulary_fail(self):
found = self.run_check(GOOD.replace('owner = ["identifier"]', 'owner = ["personal"]'))
self.assertEqual(found, ['x:Widget.owner: "personal" is not in vocabulary.categories'])
def test_unlisted_prints_a_starting_entry(self):
fields = dict(FIELDS, **{'x:Sprocket': {'properties': {'mail': prop('string', 'emailAddress')}}})
lines = check.unlisted(make_tree(fields), tomllib.loads(GOOD))
self.assertIn('[object."x:Sprocket"]', lines)
self.assertIn('mail = ["identifier"]', lines)
class StripReport(unittest.TestCase):
def test_an_import_reports_what_is_new_and_unclassified(self):
import pathlib
strip_spec = importlib.util.spec_from_file_location('strip', os.path.join(HERE, '..', 'strip.py'))
strip = importlib.util.module_from_spec(strip_spec)
strip_spec.loader.exec_module(strip)
with gzip.open(os.path.join(check.ROOT, check.SCHEMA)) as f:
upstream = json.load(f)
upstream['fields']['x:NewThing'] = {'properties': {}}
upstream['fields']['x:UserAccount']['properties']['backupEmail'] = prop('string', 'emailAddress')
tree = pathlib.Path(tempfile.mkdtemp())
(tree / 'resources/schema').mkdir(parents=True)
with gzip.open(tree / 'resources/schema/schema.json.gz', 'wt') as f:
json.dump(upstream, f)
self.assertEqual(strip.privacy_flags(tree), {
'objects': ['x:NewThing'],
'fields': ['x:UserAccount.backupEmail (identifier)'],
})
if __name__ == '__main__':
unittest.main()