Add the personal-data catalog and the check that keeps it true #83

Merged
jcoffey-dev merged 1 commits from feature/privacy-catalog into main 2026-09-28 14:32:24 +00:00
6 changed files with 2563 additions and 1 deletions
+6
View File
@@ -40,6 +40,12 @@ jobs:
# nothing. CI never sees the difference; a release does.
- if: always()
run: python3 tools/fork/context-check.py
# The personal-data catalog must classify every object and field the
# schema has, and name nothing that is gone.
- if: always()
run: python3 tools/fork/privacy-check.py
- if: always()
run: python3 -m unittest discover -s tools/fork/tests
build:
# Either runner (host1 or host2): the build needs no docker socket.
File diff suppressed because it is too large Load Diff
+19
View File
@@ -68,6 +68,25 @@ tools/fork/notice-check.py --fix # add it where it's missing
Run `--fix` after resolving an upstream merge: a conflict resolved by taking
upstream's side can drop a notice the file had.
## privacy-check.py
Fails when the personal-data catalog (`resources/privacy/catalog.toml`) and
the code disagree: an object in the schema or one of inbuxa's own JMAP
objects with no entry, a property the schema types as an address, an IP or
a secret left to its object's default, or an entry naming an object,
property, setting or code path that no longer exists. CI runs it beside the
name check, with its tests (`python3 -m unittest discover -s tools/fork/tests`).
See `docs/spec/features/personal-data-catalog.md`.
```bash
tools/fork/privacy-check.py # exit 1 on any finding
tools/fork/privacy-check.py --unlisted # starting entries for what's missing
```
After an upstream import, the strip report lists what is new and unclassified
under "Unclassified in the privacy catalog". `--unlisted` types each from the
schema alone; read the field's description before trusting it.
## record-compat.py
Records what the `*_compat` tests compare against, from the Enterprise
+228
View File
@@ -0,0 +1,228 @@
#!/usr/bin/env python3
# SPDX-FileCopyrightText: 2026 Coffey Labs
# SPDX-License-Identifier: AGPL-3.0-only
"""
Fail when the personal-data catalog and the code disagree.
tools/fork/privacy-check.py # check; exit 1 on any finding
tools/fork/privacy-check.py --unlisted # print catalog entries for what's missing
The catalog (`resources/privacy/catalog.toml`, spec
`docs/spec/features/personal-data-catalog.md`) says, for every object and
source, what personal data it can hold. An upstream import can bring objects
and fields nobody has classified, and a refactor can leave the catalog naming
things that are gone; either way the catalog stops being true without anyone
noticing, so this runs in CI on every push and pull request. It fails when:
1. an object in the schema's `fields`, or one of inbuxa's own JMAP objects,
has no catalog entry;
2. a property the schema types as an email address, an IP address or
network, or a secret is covered only by its object's `default` -- it
must be listed, so a new personal field can't hide behind a default;
3. an entry names an object, property, setting or code path that doesn't
exist (stale);
4. an entry uses a word outside the catalog's own vocabulary.
When it fails on a new object or field, classify it: `--unlisted` prints a
starting entry for each, typed from the schema alone. Read the property's
description before trusting it.
"""
import argparse
import gzip
import json
import os
import re
import sys
import tomllib
ROOT = os.path.dirname(os.path.dirname(os.path.dirname(os.path.abspath(__file__))))
SCHEMA = 'resources/schema/schema.json.gz'
CATALOG = 'resources/privacy/catalog.toml'
OBJECTS_DIR = 'crates/jmap-proto/src/object'
METHODS = 'crates/jmap-proto/src/request/method.rs'
SENSITIVE_FORMATS = {
'emailAddress': 'identifier',
'ipAddress': 'network',
'ipNetwork': 'network',
'secret': 'credential',
'secretText': 'credential',
}
INBUXA_OBJECT = re.compile(r'"(inbuxa:[A-Z][A-Za-z]*)"')
PROPERTY_NAME = re.compile(r'=> "([a-z][A-Za-z0-9]*)"')
def sensitive(type_):
"""The category a property's schema type alone implies, or None."""
fmt = type_.get('format') or (type_.get('class') or {}).get('format')
if fmt in SENSITIVE_FORMATS:
return SENSITIVE_FORMATS[fmt]
name = type_.get('objectName') or (type_.get('class') or {}).get('objectName') or ''
if name.startswith(('x:SecretKey', 'x:SecretText')) or name == 'x:HttpAuth':
return 'credential'
return None
def load_schema(root):
with gzip.open(os.path.join(root, SCHEMA)) as f:
return json.load(f)['fields']
def load_inbuxa_objects(root):
"""inbuxa's own objects, from the method names jmap-proto parses."""
with open(os.path.join(root, METHODS), encoding='utf-8') as f:
return set(INBUXA_OBJECT.findall(f.read()))
def inbuxa_properties(root, file_name):
"""The property names a jmap-proto object file maps, or None if it's gone."""
path = os.path.join(root, OBJECTS_DIR, file_name)
if not os.path.isfile(path):
return None
with open(path, encoding='utf-8') as f:
return set(PROPERTY_NAME.findall(f.read()))
def findings(root, catalog):
"""Everything wrong with `catalog` against the tree at `root`."""
fields = load_schema(root)
inbuxa = load_inbuxa_objects(root)
vocab = catalog.get('vocabulary', {})
objects = catalog.get('object', {})
sources = catalog.get('source', {})
out = []
inbuxa_props = {}
def vocabulary(where, key, values):
allowed = set(vocab.get(key, []))
for value in values if isinstance(values, list) else [values]:
if value not in allowed:
out.append(f'{where}: "{value}" is not in vocabulary.{key}')
def setting_exists(where, setting):
obj, _, prop = setting.partition('.')
if obj in fields:
if prop not in fields[obj].get('properties', {}):
out.append(f'{where}: setting {setting} names no property of {obj}')
elif obj in inbuxa:
props = inbuxa_props.get(obj)
if props is not None and prop not in props:
out.append(f'{where}: setting {setting} names no property of {obj}')
else:
out.append(f'{where}: setting {setting} names no object')
def common(where, entry):
for key in ('whose', 'where', 'scope'):
if key in entry:
vocabulary(where, key, entry[key])
retention = entry.get('retention')
if isinstance(retention, dict):
setting_exists(where, retention.get('setting', ''))
elif retention is not None:
vocabulary(where, 'retention', retention)
for key in ('enabled_by', 'captures'):
for setting in entry.get(key, []):
setting_exists(where, setting)
# inbuxa objects' properties first, so settings can name them
for name, entry in objects.items():
if name.startswith('inbuxa:'):
inbuxa_props[name] = inbuxa_properties(root, entry.get('file', ''))
# 1: nothing unlisted
for name in sorted(fields):
if name not in objects:
out.append(f'{name}: schema object has no catalog entry')
for name in sorted(inbuxa):
if name not in objects:
out.append(f'{name}: inbuxa object has no catalog entry')
for name, entry in sorted(objects.items()):
props = entry.get('properties', {})
if name.startswith('inbuxa:'):
known = inbuxa_props.get(name)
if name not in inbuxa:
out.append(f'{name}: no such inbuxa object (stale)')
if known is None:
out.append(f'{name}: file "{entry.get("file", "")}" not found in {OBJECTS_DIR}')
known = set()
elif name in fields:
known = set(fields[name].get('properties', {}))
else:
out.append(f'{name}: no such schema object (stale)')
continue
if entry.get('default') != 'none':
out.append(f'{name}: default must be "none"')
for prop, categories in props.items():
if prop not in known:
out.append(f'{name}.{prop}: no such property (stale)')
vocabulary(f'{name}.{prop}', 'categories', categories)
common(name, entry)
# 2: typed-sensitive properties are listed
if name in fields:
for prop, spec in fields[name].get('properties', {}).items():
if prop not in props and sensitive(spec['type']):
out.append(f'{name}.{prop}: typed as {sensitive(spec["type"])} but not listed')
for name, entry in sorted(sources.items()):
where = f'source "{name}"'
vocabulary(where, 'categories', entry.get('categories', []))
common(where, entry)
if 'leaves_host' not in entry:
out.append(f'{where}: leaves_host is missing')
for path in entry.get('written_by', []):
if not os.path.exists(os.path.join(root, path)):
out.append(f'{where}: written_by {path} doesn\'t exist (stale)')
return out
def unlisted(root, catalog):
"""Starting catalog entries for unlisted objects and properties."""
fields = load_schema(root)
objects = catalog.get('object', {})
lines = []
for name in sorted(fields):
entry = objects.get(name)
listed = (entry or {}).get('properties', {})
missing = {
p: sensitive(spec['type'])
for p, spec in fields[name].get('properties', {}).items()
if p not in listed and sensitive(spec['type'])
}
if entry is None or missing:
lines.append(f'[object.{json.dumps(name)}]' if entry is None else f'# add to {name}:')
if entry is None:
lines.append('default = "none"')
if missing:
if entry is None:
lines.append(f'[object.{json.dumps(name)}.properties]')
for prop, category in sorted(missing.items()):
lines.append(f'{prop} = ["{category}"]')
lines.append('')
return lines
def main(argv=None):
parser = argparse.ArgumentParser(description=__doc__.split('\n')[1])
parser.add_argument('--unlisted', action='store_true', help='print entries for what is missing')
parser.add_argument('--root', default=ROOT, help=argparse.SUPPRESS)
args = parser.parse_args(argv)
with open(os.path.join(args.root, CATALOG), 'rb') as f:
catalog = tomllib.load(f)
if args.unlisted:
print('\n'.join(unlisted(args.root, catalog)))
return 0
found = findings(args.root, catalog)
if found:
print(f'privacy check: {len(found)} finding(s) in {CATALOG}:')
for line in found:
print(f' {line}')
print('Classify new objects and fields (--unlisted helps); remove what no longer exists.')
return 1
print(f'privacy check: clean ({len(catalog.get("object", {}))} objects, '
f'{len(catalog.get("source", {}))} sources).')
return 0
if __name__ == '__main__':
sys.exit(main())
+43 -1
View File
@@ -373,6 +373,39 @@ def schema_flags(tree):
return {'objects': objects, 'fields': fields}
def privacy_flags(tree):
"""Objects and properties new in this import that the personal-data
catalog doesn't classify (docs/spec/features/personal-data-catalog.md).
Informational, like the Enterprise flags: privacy-check.py is what fails
CI once the import is merged."""
import importlib.util
import tomllib
here = Path(__file__).resolve().parent
catalog_path = here.parents[1] / 'resources' / 'privacy' / 'catalog.toml'
current_path = here.parents[1] / 'resources' / 'schema' / 'schema.json.gz'
path = tree / 'resources' / 'schema' / 'schema.json.gz'
if not (path.is_file() and catalog_path.is_file() and current_path.is_file()):
return None
spec = importlib.util.spec_from_file_location('privacy_check', here / 'privacy-check.py')
check = importlib.util.module_from_spec(spec)
spec.loader.exec_module(check)
catalog = tomllib.loads(catalog_path.read_text(encoding='utf-8')).get('object', {})
current = json.loads(gzip.decompress(current_path.read_bytes())).get('fields', {})
upstream = json.loads(gzip.decompress(path.read_bytes())).get('fields', {})
objects = sorted(name for name in upstream if name not in catalog)
fields = []
for name, spec_ in upstream.items():
if name not in catalog:
continue
listed = catalog[name].get('properties', {})
known = current.get(name, {}).get('properties', {})
for prop, p in spec_.get('properties', {}).items():
if prop not in known and prop not in listed:
typed = check.sensitive(p.get('type', {}))
fields.append(f'{name}.{prop}' + (f' ({typed})' if typed else ''))
return {'objects': objects, 'fields': sorted(fields)}
def remaining_hooks(tree):
gates, checks = {}, {}
for path in tree.rglob('*.rs'):
@@ -456,6 +489,8 @@ def write_report(out_dir, report):
]
if r['schema']:
md.append(f'- Upstream schema flags {len(r["schema"]["objects"])} objects and {len(r["schema"]["fields"])} fields as Enterprise')
if r.get('privacy'):
md.append(f'- Unclassified in the privacy catalog: {len(r["privacy"]["objects"])} objects and {len(r["privacy"]["fields"])} new fields')
md += ['', '## Removed files', ''] + [f'- `{f}`' for f in r['removed_files']]
md += ['', '## Removed snippets', ''] + [f'- `{f}`: {n}' for f, n in r['removed_snippets'].items()]
md += ['', '## Dangling module declarations removed', ''] + [f'- `{d["file"]}:{d["line"]}`: `mod {d["module"]}` ({" / ".join(d["lines"])})' for d in r['dangling_mods']]
@@ -463,6 +498,13 @@ def write_report(out_dir, report):
if r['schema']:
md += ['', '## Flagged Enterprise in upstream\'s schema', '', '**Objects:** ' + ', '.join(f'`{o}`' for o in r['schema']['objects']),
'', '**Fields:** ' + ', '.join(f'`{f}`' for f in r['schema']['fields'])]
if r.get('privacy'):
md += ['', '## Unclassified in the privacy catalog', '',
'New in this import and not in `resources/privacy/catalog.toml`. `tools/fork/privacy-check.py` '
'fails CI on these once merged; `--unlisted` prints a starting entry. A type in brackets is what '
'the schema alone says the field holds.', '',
'**Objects:** ' + (', '.join(f'`{o}`' for o in r['privacy']['objects']) or 'none'),
'', '**Fields:** ' + (', '.join(f'`{f}`' for f in r['privacy']['fields']) or 'none')]
md += ['', '## Third-party code', '',
'Comments in the stripped tree that name another copyright holder, another license, or a source the '
'code came from. Files marked **new** aren\'t in THIRD-PARTY.md yet.', '']
@@ -537,7 +579,7 @@ def main():
'removed_files': removed_files, 'removed_snippets': removed_snippets,
'cargo_edits': edits, 'dangling_mods': dangling, 'problems': problems,
'feature_gates': gates, 'edition_checks': checks,
'schema': schema_flags(tree), 'third_party': others, 'third_party_unlisted': new_others,
'schema': schema_flags(tree), 'privacy': privacy_flags(tree), 'third_party': others, 'third_party_unlisted': new_others,
'renames': renames, 'build': build, 'ossify_log': log,
}
write_report(args.out, report)
+177
View File
@@ -0,0 +1,177 @@
# SPDX-FileCopyrightText: 2026 Coffey Labs
# SPDX-License-Identifier: AGPL-3.0-only
"""Tests for tools/fork/privacy-check.py: python3 -m unittest discover tools/fork/tests"""
import gzip
import importlib.util
import json
import os
import tempfile
import tomllib
import unittest
HERE = os.path.dirname(os.path.abspath(__file__))
spec = importlib.util.spec_from_file_location('privacy_check', os.path.join(HERE, '..', 'privacy-check.py'))
check = importlib.util.module_from_spec(spec)
spec.loader.exec_module(check)
VOCAB = """
[vocabulary]
categories = ["identifier", "contact", "network", "content", "metadata", "credential"]
whose = ["holder", "correspondent", "administrator"]
where = ["data-store", "blob-store", "search-store", "in-memory-store", "memory", "log-file", "external"]
scope = ["tenant", "server"]
retention = ["unbounded", "object-life", "receiver"]
"""
GOOD = VOCAB + """
[object."x:Widget"]
default = "none"
whose = ["holder"]
where = ["data-store"]
scope = "tenant"
retention = { setting = "x:Widget.keepFor" }
[object."x:Widget".properties]
owner = ["identifier"]
[object."inbuxa:Gadget"]
file = "inbuxa_gadget.rs"
default = "none"
[object."inbuxa:Gadget".properties]
remoteIp = ["network"]
[source."widget-log"]
categories = ["network"]
where = ["log-file"]
scope = "server"
retention = "unbounded"
enabled_by = ["x:Widget.enable", "inbuxa:Gadget.remoteIp"]
leaves_host = false
written_by = ["crates/widget.rs"]
"""
def make_tree(fields):
root = tempfile.mkdtemp()
os.makedirs(os.path.join(root, 'resources/schema'))
with gzip.open(os.path.join(root, check.SCHEMA), 'wt') as f:
json.dump({'fields': fields}, f)
os.makedirs(os.path.join(root, check.OBJECTS_DIR))
with open(os.path.join(root, check.OBJECTS_DIR, 'inbuxa_gadget.rs'), 'w') as f:
f.write('Property::RemoteIp => "remoteIp",\nProperty::Id => "id",\n')
os.makedirs(os.path.join(root, 'crates/jmap-proto/src/request'), exist_ok=True)
with open(os.path.join(root, check.METHODS), 'w') as f:
f.write('"inbuxa:Gadget" => MethodObject::Gadget,\n')
with open(os.path.join(root, 'crates/widget.rs'), 'w') as f:
f.write('')
return root
def prop(type_, fmt=None):
t = {'type': type_}
if fmt:
t['format'] = fmt
return {'type': t}
FIELDS = {
'x:Widget': {'properties': {
'owner': prop('string', 'emailAddress'),
'enable': prop('boolean'),
'keepFor': prop('number', 'duration'),
'label': prop('string', 'string'),
}},
}
class PrivacyCheck(unittest.TestCase):
def run_check(self, catalog, fields=FIELDS):
return check.findings(make_tree(fields), tomllib.loads(catalog))
def test_the_repository_passes(self):
with open(os.path.join(check.ROOT, check.CATALOG), 'rb') as f:
catalog = tomllib.load(f)
self.assertEqual(check.findings(check.ROOT, catalog), [])
def test_a_consistent_catalog_passes(self):
self.assertEqual(self.run_check(GOOD), [])
def test_an_unclassified_object_fails(self):
fields = dict(FIELDS, **{'x:Sprocket': {'properties': {'size': prop('number', 'size')}}})
found = self.run_check(GOOD, fields)
self.assertEqual(found, ['x:Sprocket: schema object has no catalog entry'])
def test_an_address_hidden_behind_the_default_fails(self):
fields = {'x:Widget': {'properties': dict(FIELDS['x:Widget']['properties'],
contactEmail=prop('string', 'emailAddress'))}}
found = self.run_check(GOOD, fields)
self.assertEqual(found, ['x:Widget.contactEmail: typed as identifier but not listed'])
def test_a_secret_in_a_set_or_object_counts(self):
fields = {'x:Widget': {'properties': dict(
FIELDS['x:Widget']['properties'],
ips={'type': {'type': 'set', 'class': {'type': 'string', 'format': 'ipNetwork'}}},
key={'type': {'type': 'object', 'objectName': 'x:SecretKey'}},
)}}
found = self.run_check(GOOD, fields)
self.assertIn('x:Widget.ips: typed as network but not listed', found)
self.assertIn('x:Widget.key: typed as credential but not listed', found)
def test_a_stale_property_fails(self):
found = self.run_check(GOOD.replace('owner = ["identifier"]', 'owner = ["identifier"]\ngone = ["content"]'))
self.assertEqual(found, ['x:Widget.gone: no such property (stale)'])
def test_a_stale_object_fails(self):
found = self.run_check(GOOD + '\n[object."x:Removed"]\ndefault = "none"\n')
self.assertEqual(found, ['x:Removed: no such schema object (stale)'])
def test_a_stale_setting_fails(self):
found = self.run_check(GOOD.replace('x:Widget.keepFor', 'x:Widget.keepForever'))
self.assertEqual(found, ['x:Widget: setting x:Widget.keepForever names no property of x:Widget'])
def test_a_stale_code_path_fails(self):
found = self.run_check(GOOD.replace('crates/widget.rs', 'crates/gone.rs'))
self.assertEqual(found, ['source "widget-log": written_by crates/gone.rs doesn\'t exist (stale)'])
def test_an_unlisted_inbuxa_object_fails(self):
catalog = VOCAB + """
[object."x:Widget"]
default = "none"
[object."x:Widget".properties]
owner = ["identifier"]
"""
found = self.run_check(catalog)
self.assertEqual(found, ['inbuxa:Gadget: inbuxa object has no catalog entry'])
def test_words_outside_the_vocabulary_fail(self):
found = self.run_check(GOOD.replace('owner = ["identifier"]', 'owner = ["personal"]'))
self.assertEqual(found, ['x:Widget.owner: "personal" is not in vocabulary.categories'])
def test_unlisted_prints_a_starting_entry(self):
fields = dict(FIELDS, **{'x:Sprocket': {'properties': {'mail': prop('string', 'emailAddress')}}})
lines = check.unlisted(make_tree(fields), tomllib.loads(GOOD))
self.assertIn('[object."x:Sprocket"]', lines)
self.assertIn('mail = ["identifier"]', lines)
class StripReport(unittest.TestCase):
def test_an_import_reports_what_is_new_and_unclassified(self):
import pathlib
strip_spec = importlib.util.spec_from_file_location('strip', os.path.join(HERE, '..', 'strip.py'))
strip = importlib.util.module_from_spec(strip_spec)
strip_spec.loader.exec_module(strip)
with gzip.open(os.path.join(check.ROOT, check.SCHEMA)) as f:
upstream = json.load(f)
upstream['fields']['x:NewThing'] = {'properties': {}}
upstream['fields']['x:UserAccount']['properties']['backupEmail'] = prop('string', 'emailAddress')
tree = pathlib.Path(tempfile.mkdtemp())
(tree / 'resources/schema').mkdir(parents=True)
with gzip.open(tree / 'resources/schema/schema.json.gz', 'wt') as f:
json.dump(upstream, f)
self.assertEqual(strip.privacy_flags(tree), {
'objects': ['x:NewThing'],
'fields': ['x:UserAccount.backupEmail (identifier)'],
})
if __name__ == '__main__':
unittest.main()