linux/include/trace/events/landlock.h
Mickaël Salaün e7e0a54300
landlock: Widen ruleset versions to 64 bits
Tracepoint consumers use a ruleset ID and version to identify the
successful landlock_add_rule(2) call prefix used to create a domain.
LANDLOCK_MAX_NUM_RULES bounds distinct stored rules, not successful
calls: re-adding already-present rights for an object or port succeeds
without increasing num_rules.  Because every successful call increments
the version, these calls can wrap the 32-bit counter and give different
prefixes the same trace identity.

Widen the counter and its trace fields to 64 bits so the counter cannot
wrap in practice, while preserving the successful-call semantics.
Saturating would alias all subsequent histories, while rejecting a call
at the limit would change otherwise valid syscall behavior solely for
trace metadata.

Cc: Günther Noack <gnoack@google.com>
Cc: Steven Rostedt <rostedt@goodmis.org>
Fixes: 63747c9477 ("landlock: Add landlock_add_rule_fs and landlock_add_rule_net tracepoints")
Reviewed-by: Günther Noack <gnoack@google.com>
Link: https://patch.msgid.link/20260922132615.1025945-1-mic@digikod.net
Signed-off-by: Mickaël Salaün <mic@digikod.net>
2026-09-22 20:52:13 +02:00

1061 lines
38 KiB
C

/* SPDX-License-Identifier: GPL-2.0 */
/*
* Copyright © 2025 Microsoft Corporation
* Copyright © 2026 Cloudflare, Inc.
*/
#undef TRACE_SYSTEM
#define TRACE_SYSTEM landlock
#if !defined(_TRACE_LANDLOCK_H) || defined(TRACE_HEADER_MULTI_READ)
#define _TRACE_LANDLOCK_H
#include <linux/in.h>
#include <linux/in6.h>
#include <linux/landlock.h>
#include <linux/socket.h>
#include <linux/string.h>
#include <linux/string_helpers.h>
#include <linux/tracepoint.h>
#include <linux/trace_seq.h>
#include <net/af_unix.h>
enum landlock_request_type;
struct dentry;
struct landlock_blockers;
struct landlock_domain;
struct landlock_hierarchy;
struct landlock_rule;
struct landlock_ruleset;
struct path;
struct sock;
struct task_struct;
static_assert(sizeof(access_mask_t) <= sizeof(u64));
TRACE_DEFINE_ENUM(LANDLOCK_REQUEST_FS_CHANGE_TOPOLOGY);
TRACE_DEFINE_ENUM(LANDLOCK_REQUEST_FS_ACCESS);
TRACE_DEFINE_ENUM(LANDLOCK_REQUEST_NET_ACCESS);
#ifdef CREATE_TRACE_POINTS
/* About 6 KiB, leaving about 2 KiB for sibling helpers and fixed fields. */
#define TRACE_UNTRUSTED_STR_OUTPUT_SIZE \
(TRACE_SEQ_BUFFER_SIZE - TRACE_SEQ_BUFFER_SIZE / 4)
/*
* A raw UTF-8 ellipsis (…) marks truncation and cannot collide with escaped
* input: ESCAPE_NAP renders every non-ASCII input byte in octal.
*/
#define TRACE_TRUNCATION_MARKER "\xe2\x80\xa6"
/*
* Escapes @len bytes of an untrusted string into the trace sequence @p so it
* cannot inject field separators or control characters into the ftrace text
* output, and can be unambiguously recovered. Called from the TP_printk() of
* the tracepoints that expose paths and process names. @len is passed by the
* caller (rather than derived with strlen()) so a name that is not
* NUL-terminated or carries embedded NUL bytes (an abstract socket name) is
* escaped in full instead of being truncated at the first NUL.
*
* Strings that exceed the output limit retain the largest complete escaped
* prefix followed by the truncation marker.
*
* Return: a pointer into @p's buffer, or NULL if @src is NULL or the fixed
* output reservation is unavailable.
*/
static inline const char *
__trace_print_untrusted_str(struct trace_seq *p, const char *src, size_t len)
{
const unsigned int escape_flags = ESCAPE_SPACE | ESCAPE_SPECIAL |
ESCAPE_NAP | ESCAPE_APPEND |
ESCAPE_OCTAL;
const size_t marker_len = sizeof(TRACE_TRUNCATION_MARKER) - 1;
size_t buf_size, prefix_len, prefix_size;
int escaped_size;
char *buf;
const char *ret;
buf_size = seq_buf_get_buf(&p->seq, &buf);
if (!src || buf_size < TRACE_UNTRUSTED_STR_OUTPUT_SIZE)
return NULL;
ret = trace_seq_buffer_ptr(p);
escaped_size = string_escape_mem(src, len, buf,
TRACE_UNTRUSTED_STR_OUTPUT_SIZE,
escape_flags, " ='\"\\");
if (likely(escaped_size < TRACE_UNTRUSTED_STR_OUTPUT_SIZE)) {
seq_buf_commit(&p->seq, escaped_size);
trace_seq_putc(p, 0);
return ret;
}
prefix_len = 0;
prefix_size = 0;
while (prefix_len < len) {
const char *const src_char = src + prefix_len;
int char_size;
char_size = string_escape_mem(src_char, 1, NULL, 0,
escape_flags, " ='\"\\");
if (char_size > TRACE_UNTRUSTED_STR_OUTPUT_SIZE - marker_len -
1 - prefix_size)
break;
prefix_size += char_size;
prefix_len++;
}
escaped_size = string_escape_mem(src, prefix_len, buf, prefix_size,
escape_flags, " ='\"\\");
if (WARN_ON_ONCE(escaped_size != prefix_size))
return NULL;
memcpy(buf + prefix_size, TRACE_TRUNCATION_MARKER, marker_len);
seq_buf_commit(&p->seq, prefix_size + marker_len);
trace_seq_putc(p, 0);
return ret;
}
/*
* Fills the dense per-domain-layer array layers (one access mask per layer,
* indexed by level - 1) from rule's sparse layer stack, keeping only the
* requested rights (access_request). Layers with no matching rule entry get
* a zero mask. Shared by the check_rule_inode and check_rule_net_port events.
*
* rule->layers is sorted by ascending level, with levels in the domain's
* [1, num_layers] range (see landlock_merge_ruleset()), so every entry maps
* to a slot. A leftover entry would be a malformed rule; the zero-filled
* slots keep the output and the array bounds safe regardless.
*/
static inline void
__trace_landlock_fill_layers(access_mask_t *const layers,
const size_t num_layers,
const struct landlock_rule *const rule,
const access_mask_t access_request)
{
size_t i = 0;
for (size_t level = 1; level <= num_layers; level++) {
access_mask_t grants = 0;
if (i < rule->num_layers && level == rule->layers[i].level) {
grants = rule->layers[i].access & access_request;
i++;
}
layers[level - 1] = grants;
}
/* A leftover entry means an out-of-range or unsorted rule level. */
WARN_ON_ONCE(i < rule->num_layers);
}
/*
* Renders the dense per-domain-layer access array as symbolic flag names for
* the grants field: layers wrapped in "{}", flags within a layer joined by
* "|", layers separated by ",", an empty layer rendered as nothing.
* Open-codes the flag walk because trace_print_flags_seq() NUL-terminates per
* call and so cannot be chained into a single field. The shared names table
* covers every access right, so masked bits are always named. Returns the
* trace_seq position like __print_flags().
*/
static inline const char *__trace_landlock_print_layers(
struct trace_seq *p, const access_mask_t *const layers,
const size_t num_layers, const struct trace_print_flags *const names,
const size_t names_size)
{
const char *const ret = trace_seq_buffer_ptr(p);
trace_seq_putc(p, '{');
for (size_t i = 0; i < num_layers; i++) {
access_mask_t mask = layers[i];
bool first = true;
if (i)
trace_seq_putc(p, ',');
for (size_t j = 0; mask && j < names_size; j++) {
if ((mask & names[j].mask) != names[j].mask)
continue;
if (!first)
trace_seq_putc(p, '|');
trace_seq_puts(p, names[j].name);
mask &= ~names[j].mask;
first = false;
}
}
trace_seq_putc(p, '}');
trace_seq_putc(p, 0);
return ret;
}
#endif /* CREATE_TRACE_POINTS */
/* clang-format off */
/* Maps a shared _LANDLOCK_*_NAMES entry to a __print_flags() pair. */
#define _LANDLOCK_NAME_ENTRY(mask, name) { mask, name }
#define _LANDLOCK_FS_BLOCKER_TYPE_NAMES \
{ LANDLOCK_REQUEST_FS_CHANGE_TOPOLOGY, "change_topology" }
/**
* DOC: Landlock trace events
*
* These guarantees and constraints hold for every Landlock tracepoint.
* A new tracepoint must uphold them, and an eBPF consumer can rely on
* them.
*
* Decision context
* ~~~~~~~~~~~~~~~~
*
* A denial event identifies the domain whose policy denied the request, the
* Landlock operation and policy object that were checked, and the blocker or
* domain relationship responsible for the denial. When tracing starts with
* sandbox construction, ruleset and domain events provide the policy history
* needed to interpret these identifiers. The denying domain is the subject
* that enforced the policy, not necessarily current. Generic tracepoints can
* provide additional operational context.
*
* Lifecycle consistency
* ~~~~~~~~~~~~~~~~~~~~~~
*
* Lifecycle emission is balanced: a creation event always has a matching
* deallocation event and vice versa. A consumer that observes an object's
* complete lifetime can model it from this pair; one that attaches late or
* loses exported records must reconcile incomplete state. A creation event
* fires while the object is still private to the calling thread
* (landlock_create_ruleset fires before the ruleset's file descriptor is
* installed, so it cannot race a concurrent :manpage:`close(2)`); if fd
* installation later fails and the ruleset is freed, free_ruleset still
* fires, keeping the pair balanced. The domain pair (create_domain and
* free_domain) is balanced the same way: create_domain fires when the
* domain is created (under the ruleset lock, before thread-sync), and
* free_domain fires when it is freed. A rare thread-sync failure aborts
* the just-created domain, which then emits both events (its creation, then
* an immediate free). Denial events fire only for denials that actually
* happen.
*
* Pointer access
* ~~~~~~~~~~~~~~
*
* All pointer arguments in TP_PROTO are guaranteed non-NULL by the
* caller, but pointers reached through them may still be NULL (e.g.,
* hierarchy->parent at a root domain) and must be checked. eBPF programs
* read these pointers via BTF for richer introspection than the
* TP_STRUCT__entry fields, which serve TP_printk display only.
*
* Mutable object pointers are passed while the caller holds the object's
* lock, so TP_fast_assign and a BTF reader see the exact object the event
* reports, a snapshot no concurrent writer can change: add_rule holds the
* modified ruleset's lock, and create_domain holds the ruleset lock across
* the emission (before the thread-sync wait) so the inspected ruleset is
* the one merged into the domain. Objects immutable at the emission site
* (a domain after creation, a hierarchy at its last reference) need no
* lock. A few values that no held lock protects are a best-effort
* lockless snapshot instead: a task's comm, and the deny_access_net struct
* sock (whose network hook holds no socket lock), matching how the sched
* and signal trace events sample comm.
*
* Field encoding
* ~~~~~~~~~~~~~~
*
* Fields that mirror the Landlock UAPI preserve their widths and endianness
* (e.g. network ports are u64 in host endianness, like
* landlock_net_port_attr.port). Per-event details, such as where a value
* is byte-swapped, live in the field's own kdoc.
*
* Rule-check fields
* ~~~~~~~~~~~~~~~~~
*
* The check_rule events fire during an access check, once per matching
* rule, before the final allow-or-deny verdict. They share domain (the
* enforcing domain being evaluated), access_request (the access mask being
* checked), and rule (the matching rule, with per-layer access masks).
*
* Denial fields
* ~~~~~~~~~~~~~
*
* Every denial event shares three fields. domain is the ID of the
* innermost domain that blocked the access. same_exec tells whether the
* current task is the same executable that entered that domain. logged is
* the domain's audit-logging decision for this denial (its log_status is
* enabled and the per-execution flag selected by same_exec is set); a
* stateless ftrace filter can select the denials the domain submits to
* audit with logged==1, without reconstructing it from the per-execution
* log flags. Denial events order their fields as domain, same_exec,
* logged, then blockers (deny_access events only), then the type-specific
* object fields, then any variable-length field.
*
* Relational referents
* ~~~~~~~~~~~~~~~~~~~~~
*
* A scope or ptrace verdict compares two domains, so the other party's
* domain is part of the decision context. It is exposed as a scalar
* domain ID (0 when that party is unsandboxed): target_domain (signal),
* peer_domain (abstract unix socket), tracee_domain (ptrace). With both
* IDs in the stream, a consumer that tracked domain creation can relate
* the two parties without kernel-internal state. The ID is a scalar
* snapshot, not a live domain pointer that could dangle: an optional
* relational referent is a scalar (0 sentinel), not a nullable pointer.
* Nonzero IDs are unique within one boot. For ptrace, same_exec instead
* describes the tracer, even for
* PTRACE_TRACEME, and may differ from the current task.
*
* Blocker fields
* ~~~~~~~~~~~~~~
*
* The filesystem and network blocker arguments identify the request type
* and carry its final missing access subset when applicable. The type
* determines how to interpret the access value.
*/
/*
* Prints a per-layer access mask array (the dynamic array @array) as symbolic
* flag names using the shared @flag_names list (a _LANDLOCK_*_NAMES macro).
* Stays outside CREATE_TRACE_POINTS: TP_printk is expanded in the print-output
* pass where that macro is undefined.
*/
#define __print_landlock_layers(array, flag_names...) \
({ \
static const struct trace_print_flags __layer_names[] = { \
flag_names \
}; \
__trace_landlock_print_layers( \
p, __get_dynamic_array(array), \
__get_dynamic_array_len(array) / \
sizeof(access_mask_t), \
__layer_names, ARRAY_SIZE(__layer_names)); \
})
/**
* landlock_create_ruleset - New ruleset created
*
* @ruleset: Newly created ruleset (never NULL); not yet shared via an fd,
* so no lock is needed.
*
* Emitted by sys_landlock_create_ruleset() while the new ruleset is still
* private to the calling thread, before its file descriptor is installed,
* so it cannot race a concurrent :manpage:`close(2)`. Balanced by a
* matching landlock_free_ruleset event.
*/
TRACE_EVENT(landlock_create_ruleset,
TP_PROTO(const struct landlock_ruleset *ruleset),
TP_ARGS(ruleset),
TP_STRUCT__entry(
__field( u64, ruleset_id )
__field( u64, ruleset_version )
__field( access_mask_t, handled_fs )
__field( access_mask_t, handled_net )
__field( access_mask_t, scoped )
),
TP_fast_assign(
__entry->ruleset_id = ruleset->id;
__entry->ruleset_version = ruleset->version;
__entry->handled_fs = ruleset->handled_masks.fs;
__entry->handled_net = ruleset->handled_masks.net;
__entry->scoped = ruleset->handled_masks.scope;
),
TP_printk("ruleset=%llx.%llu handled_fs=%s handled_net=%s scoped=%s",
__entry->ruleset_id, __entry->ruleset_version,
__print_flags(__entry->handled_fs, "|", _LANDLOCK_ACCESS_FS_NAMES),
__print_flags(__entry->handled_net, "|", _LANDLOCK_ACCESS_NET_NAMES),
__print_flags(__entry->scoped, "|", _LANDLOCK_SCOPE_NAMES))
);
/**
* landlock_free_ruleset - Ruleset freed
*
* @ruleset: Ruleset being freed (never NULL); at its last reference, so no
* lock is needed.
*
* Emitted when a ruleset's last reference is dropped (typically when
* the creating process closes the ruleset file descriptor). Fires even
* when file-descriptor installation failed after creation, keeping the
* create/free pair balanced.
*/
TRACE_EVENT(landlock_free_ruleset,
TP_PROTO(const struct landlock_ruleset *ruleset),
TP_ARGS(ruleset),
TP_STRUCT__entry(
__field( u64, ruleset_id )
__field( u64, ruleset_version )
),
TP_fast_assign(
__entry->ruleset_id = ruleset->id;
__entry->ruleset_version = ruleset->version;
),
TP_printk("ruleset=%llx.%llu",
__entry->ruleset_id, __entry->ruleset_version)
);
/**
* landlock_add_rule_path_beneath - Path-beneath rule added to a ruleset
*
* @ruleset: Source ruleset (never NULL).
* @flags: Complete validated landlock_add_rule_flags value supplied by this
* successful call, not the rule's accumulated quiet state.
* @access_rights: Canonical per-call access mask passed to
* landlock_insert_rule() after normalization, not the raw
* sys_landlock_add_rule() argument or accumulated rule.
* @path: Filesystem path for the rule (never NULL).
* @pathname: Resolved absolute path string (never NULL; error placeholder
* on resolution failure).
*
* Emitted by sys_landlock_add_rule() under the modified ruleset's lock, so
* the reported ruleset is a stable snapshot that no concurrent writer can
* change.
*/
TRACE_EVENT(landlock_add_rule_path_beneath,
TP_PROTO(const struct landlock_ruleset *ruleset, u32 flags,
u64 access_rights, const struct path *path,
const char *pathname),
TP_ARGS(ruleset, flags, access_rights, path, pathname),
TP_STRUCT__entry(
__field( u64, ruleset_id )
__field( u64, ruleset_version )
__field( access_mask_t, access_rights )
__field( dev_t, dev )
__field( ino_t, ino )
__string( pathname, pathname )
),
TP_fast_assign(
lockdep_assert_held(&ruleset->lock);
__entry->ruleset_id = ruleset->id;
__entry->ruleset_version = ruleset->version;
__entry->access_rights = access_rights;
__entry->dev = path->dentry->d_sb->s_dev;
/*
* The inode number may not be the user-visible one,
* but it will be the same used by audit.
*/
__entry->ino = d_backing_inode(path->dentry)->i_ino;
__assign_str(pathname);
),
TP_printk("ruleset=%llx.%llu access_rights=%s dev=%u:%u ino=%lu path=%s",
__entry->ruleset_id, __entry->ruleset_version,
__print_flags(__entry->access_rights, "|", _LANDLOCK_ACCESS_FS_NAMES),
MAJOR(__entry->dev), MINOR(__entry->dev), __entry->ino,
__trace_print_untrusted_str(p, __get_str(pathname),
__get_dynamic_array_len(pathname) - 1))
);
/**
* landlock_add_rule_net_port - Network-port rule added to a ruleset
*
* @ruleset: Source ruleset (never NULL).
* @flags: Complete validated landlock_add_rule_flags value supplied by this
* successful call, not the rule's accumulated quiet state.
* @access_rights: Canonical per-call access mask passed to
* landlock_insert_rule() after normalization, not the raw
* sys_landlock_add_rule() argument or accumulated rule.
* @port: Network port in host endianness, forwarded directly from
* &landlock_net_port_attr.port.
*
* Emitted by sys_landlock_add_rule() under the modified ruleset's lock, so
* the reported ruleset is a stable snapshot that no concurrent writer can
* change.
*/
TRACE_EVENT(landlock_add_rule_net_port,
TP_PROTO(const struct landlock_ruleset *ruleset, u32 flags,
u64 access_rights, u64 port),
TP_ARGS(ruleset, flags, access_rights, port),
TP_STRUCT__entry(
__field( u64, ruleset_id )
__field( u64, ruleset_version )
__field( access_mask_t, access_rights )
__field( u64, port )
),
TP_fast_assign(
lockdep_assert_held(&ruleset->lock);
__entry->ruleset_id = ruleset->id;
__entry->ruleset_version = ruleset->version;
__entry->access_rights = access_rights;
__entry->port = port;
),
TP_printk("ruleset=%llx.%llu access_rights=%s port=%llu",
__entry->ruleset_id, __entry->ruleset_version,
__print_flags(__entry->access_rights, "|", _LANDLOCK_ACCESS_NET_NAMES),
__entry->port)
);
/**
* landlock_create_domain - New domain created
*
* @domain: Newly created domain (never NULL, immutable after creation).
* @domain->hierarchy->id is its unique ID, shared with the
* landlock_enforce_domain and landlock_free_domain events;
* @domain->hierarchy->details holds the requesting process.
* @ruleset: Source ruleset frozen into the domain (never NULL). The
* ruleset lock is held across the emission, so a BPF program
* reading it via BTF sees the exact merged ruleset;
* @ruleset->id / @ruleset->version identify it.
*
* Emitted by sys_landlock_restrict_self() once, in the requesting
* thread's context, right after the merge and before thread-sync. The
* flags-only path (ruleset_fd == -1) creates no domain and does not
* emit this event. Paired with the per-thread landlock_enforce_domain
* (join on @domain->hierarchy->id) and balanced by a matching
* landlock_free_domain event.
*/
TRACE_EVENT(landlock_create_domain,
TP_PROTO(const struct landlock_domain *domain,
const struct landlock_ruleset *ruleset),
TP_ARGS(domain, ruleset),
TP_STRUCT__entry(
__field( u64, domain_id )
__field( u64, parent_id )
__field( u64, ruleset_id )
__field( u64, ruleset_version )
),
TP_fast_assign(
lockdep_assert_held(&ruleset->lock);
__entry->domain_id = domain->hierarchy->id;
__entry->parent_id = domain->hierarchy->parent ?
domain->hierarchy->parent->id : 0;
__entry->ruleset_id = ruleset->id;
__entry->ruleset_version = ruleset->version;
),
TP_printk("domain=%llx parent=%llx ruleset=%llx.%llu",
__entry->domain_id, __entry->parent_id,
__entry->ruleset_id, __entry->ruleset_version)
);
/**
* landlock_enforce_domain - Domain enforced on a thread
*
* @domain: Domain now enforced on the current thread (never NULL,
* immutable; read locklessly). Correlate to
* landlock_create_domain via @domain->hierarchy->id for the
* source ruleset and requesting thread, or read
* @domain->hierarchy->details for the requesting process.
* @complete: Set on the single event that concludes the operation, after
* all its other enforcements; filter on it for one event per
* operation.
* @process_wide: The enforcement covers every eligible (non-exiting)
* thread of the process: set when the caller used
* %LANDLOCK_RESTRICT_SELF_TSYNC or the process is
* single-threaded. A lone thread whose group still
* holds a zombie leader is not counted single-threaded,
* so process_wide == 0 never proves the opposite.
* @no_new_privs: The enforcing thread's no_new_privs state at
* enforcement time: 1 if set (by a prior
* :manpage:`prctl(2)` %PR_SET_NO_NEW_PRIVS or by
* %LANDLOCK_RESTRICT_SELF_NO_NEW_PRIVS), 0 if the domain
* was enforced with %CAP_SYS_ADMIN instead.
*
* Emitted for each thread sys_landlock_restrict_self() enforces the
* domain on, in that thread's own context, right after its
* commit_creds(), so it fires only once the thread is irreversibly
* enforcing the domain (aborted operations emit none). Not
* balanced; every enforcement falls between the domain's
* landlock_create_domain and landlock_free_domain events.
*
* @complete == 1 && @process_wide == 1 means the whole process is
* sandboxed by @domain, durably (Landlock domains are monotonic and
* inherited on :manpage:`clone(2)`).
*/
TRACE_EVENT(landlock_enforce_domain,
TP_PROTO(const struct landlock_domain *domain, bool complete,
bool process_wide, bool no_new_privs),
TP_ARGS(domain, complete, process_wide, no_new_privs),
TP_STRUCT__entry(
__field( u64, domain_id )
__field( bool, complete )
__field( bool, process_wide )
__field( bool, no_new_privs )
),
TP_fast_assign(
__entry->domain_id = domain->hierarchy->id;
__entry->complete = complete;
__entry->process_wide = process_wide;
__entry->no_new_privs = no_new_privs;
),
TP_printk("domain=%llx complete=%d process_wide=%d no_new_privs=%d",
__entry->domain_id, __entry->complete, __entry->process_wide,
__entry->no_new_privs)
);
/**
* landlock_free_domain - Domain freed
*
* @hierarchy: Hierarchy node being freed (never NULL).
*
* Emitted when the hierarchy node's last reference is dropped: its
* refcount reaches zero after all child domains have released their
* parent reference. A committed domain is
* freed from a kworker via landlock_put_domain_deferred() (the credential
* free path runs in RCU context, where sleeping is forbidden), so the
* current task is not the sandboxed task that triggered the free. Balanced
* by a matching landlock_create_domain event.
*/
TRACE_EVENT(landlock_free_domain,
TP_PROTO(const struct landlock_hierarchy *hierarchy),
TP_ARGS(hierarchy),
TP_STRUCT__entry(
__field( u64, domain_id )
__field( u64, denials )
),
TP_fast_assign(
__entry->domain_id = hierarchy->id;
__entry->denials = atomic64_read(&hierarchy->num_denials);
),
TP_printk("domain=%llx denials=%llu",
__entry->domain_id, __entry->denials)
);
/**
* landlock_check_rule_inode - Inode rule evaluated during access check
*
* @domain: Enforcing domain (never NULL).
* @rule: Matching rule with per-layer access masks (never NULL).
* @access_request: Access mask evaluated against the rule (the domain's
* handled mask during rename/link double-checks).
* @dentry: Filesystem dentry being checked (never NULL).
*
* Emitted for each rule that matches during a filesystem access check.
* The grants array shows the requested rights the rule grants at each
* domain layer. See Documentation/trace/events-landlock.rst for how to
* interpret it.
*/
TRACE_EVENT(landlock_check_rule_inode,
TP_PROTO(const struct landlock_domain *domain,
const struct landlock_rule *rule,
u64 access_request, const struct dentry *dentry),
TP_ARGS(domain, rule, access_request, dentry),
TP_STRUCT__entry(
__field( u64, domain_id )
__field( access_mask_t, access_request )
__field( dev_t, dev )
__field( ino_t, ino )
__dynamic_array(access_mask_t, grants,
domain->num_layers)
),
TP_fast_assign(
__entry->domain_id = domain->hierarchy->id;
__entry->access_request = access_request;
__entry->dev = dentry->d_sb->s_dev;
__entry->ino = d_backing_inode(dentry)->i_ino;
__trace_landlock_fill_layers(__get_dynamic_array(grants),
__get_dynamic_array_len(grants) /
sizeof(access_mask_t),
rule,
(access_mask_t)access_request);
),
TP_printk("domain=%llx access_request=%s dev=%u:%u ino=%lu grants=%s",
__entry->domain_id,
__print_flags(__entry->access_request, "|", _LANDLOCK_ACCESS_FS_NAMES),
MAJOR(__entry->dev), MINOR(__entry->dev), __entry->ino,
__print_landlock_layers(grants, _LANDLOCK_ACCESS_FS_NAMES))
);
/**
* landlock_check_rule_net_port - Network port rule evaluated
*
* @domain: Enforcing domain (never NULL).
* @rule: Matching rule with per-layer access masks (never NULL).
* @access_request: Access mask being requested.
* @port: Network port being checked (host endianness).
*
* Emitted for each rule that matches during a network access check. The
* grants array shows the requested rights the rule grants at each domain
* layer. See Documentation/trace/events-landlock.rst for how to
* interpret it.
*/
TRACE_EVENT(landlock_check_rule_net_port,
TP_PROTO(const struct landlock_domain *domain,
const struct landlock_rule *rule,
u64 access_request, u64 port),
TP_ARGS(domain, rule, access_request, port),
TP_STRUCT__entry(
__field( u64, domain_id )
__field( access_mask_t, access_request )
__field( u64, port )
__dynamic_array(access_mask_t, grants,
domain->num_layers)
),
TP_fast_assign(
__entry->domain_id = domain->hierarchy->id;
__entry->access_request = access_request;
__entry->port = port;
__trace_landlock_fill_layers(__get_dynamic_array(grants),
__get_dynamic_array_len(grants) /
sizeof(access_mask_t),
rule,
(access_mask_t)access_request);
),
TP_printk("domain=%llx access_request=%s port=%llu grants=%s",
__entry->domain_id,
__print_flags(__entry->access_request, "|", _LANDLOCK_ACCESS_NET_NAMES),
__entry->port,
__print_landlock_layers(grants, _LANDLOCK_ACCESS_NET_NAMES))
);
/**
* landlock_deny_access_fs - Filesystem access denied
*
* @hierarchy: Denying domain's hierarchy node (never NULL); its id is the
* domain field.
* @same_exec: Whether the current task entered the denying domain itself.
* @logged: The domain's audit-logging decision for this denial.
* @blockers: Request type and final missing access subset (never NULL).
* @path: Filesystem path that was denied (never NULL).
* @pathname: Resolved path string (never NULL; an error placeholder on
* resolution failure).
*
* Emitted when a Landlock domain denies a filesystem access.
*/
TRACE_EVENT(landlock_deny_access_fs,
TP_PROTO(const struct landlock_hierarchy *hierarchy, bool same_exec,
bool logged, const struct landlock_blockers *blockers,
const struct path *path, const char *pathname),
TP_ARGS(hierarchy, same_exec, logged, blockers, path, pathname),
TP_STRUCT__entry(
__field( u64, domain_id )
__field( bool, same_exec )
__field( bool, logged )
__field( enum landlock_request_type, blockers_type )
__field( access_mask_t, blockers_access )
__field( dev_t, dev )
__field( ino_t, ino )
__string( pathname, pathname )
),
TP_fast_assign(
const struct inode *inode = d_backing_inode(path->dentry);
__entry->domain_id = hierarchy->id;
__entry->same_exec = same_exec;
__entry->logged = logged;
__entry->blockers_type = blockers->type;
__entry->blockers_access = blockers->access;
__entry->dev = path->dentry->d_sb->s_dev;
/*
* A negative dentry has no backing inode, so mirror the
* guard in dump_common_audit_data() and report inode 0.
*/
__entry->ino = inode ? inode->i_ino : 0;
__assign_str(pathname);
),
TP_printk("domain=%llx same_exec=%d logged=%d blockers=%s dev=%u:%u ino=%lu path=%s",
__entry->domain_id, __entry->same_exec, __entry->logged,
__entry->blockers_type == LANDLOCK_REQUEST_FS_ACCESS ?
__print_flags(__entry->blockers_access, "|", _LANDLOCK_ACCESS_FS_NAMES) :
__print_symbolic(__entry->blockers_type,
_LANDLOCK_FS_BLOCKER_TYPE_NAMES),
MAJOR(__entry->dev), MINOR(__entry->dev), __entry->ino,
__trace_print_untrusted_str(p, __get_str(pathname),
__get_dynamic_array_len(pathname) - 1))
);
static_assert(offsetof(struct sockaddr_in, sin_port) ==
offsetof(struct sockaddr_in6, sin6_port));
static_assert(sizeof_field(struct sockaddr_in, sin_port) ==
sizeof_field(struct sockaddr_in6, sin6_port));
/**
* landlock_deny_access_net - Network access denied
*
* @hierarchy: Denying domain's hierarchy node (never NULL); its id is the
* domain field.
* @same_exec: Whether the current task entered the denying domain itself.
* @logged: The domain's audit-logging decision for this denial.
* @blockers: Request type and final missing access subset (never NULL).
* @sk: Socket object (never NULL), read without a socket lock, so its fields
* are a best-effort snapshot.
* @socket_family: Socket-family snapshot used by the verdict.
* @address: Authoritative address checked by the verdict (never NULL).
* The producer copies @addrlen bytes from the checked address and
* zeroes the remaining storage before emission. The
* &sockaddr_in.sin_port or &sockaddr_in6.sin6_port member, when
* present, remains in network endianness.
* @addrlen: Validated signed length of @address.
*
* Emitted when a Landlock domain denies a network operation. The blocker
* identifies whether the address is a bind or connect/send policy object.
* The flattened port field is converted from the checked address to host
* endianness, or is -1 when no port was checked. Zero is a valid checked
* port.
*/
TRACE_EVENT(landlock_deny_access_net,
TP_PROTO(const struct landlock_hierarchy *hierarchy, bool same_exec,
bool logged, const struct landlock_blockers *blockers,
const struct sock *sk, u16 socket_family,
const struct sockaddr_storage *address, int addrlen),
TP_ARGS(hierarchy, same_exec, logged, blockers, sk, socket_family,
address, addrlen),
TP_STRUCT__entry(
__field( u64, domain_id )
__field( bool, same_exec )
__field( bool, logged )
__field( enum landlock_request_type, blockers_type )
__field( access_mask_t, blockers_access )
__field( s64, port )
),
TP_fast_assign(
const struct sockaddr *const addr =
(const struct sockaddr *)address;
const bool has_port =
addrlen >= (int)offsetofend(struct sockaddr_in, sin_port) &&
(addr->sa_family == AF_INET ||
addr->sa_family == AF_INET6 ||
(addr->sa_family == AF_UNSPEC &&
socket_family == AF_INET));
__entry->domain_id = hierarchy->id;
__entry->same_exec = same_exec;
__entry->logged = logged;
__entry->blockers_type = blockers->type;
__entry->blockers_access = blockers->access;
__entry->port =
has_port ?
ntohs(((const struct sockaddr_in *)addr)->sin_port) :
-1;
),
TP_printk("domain=%llx same_exec=%d logged=%d blockers=%s port=%lld",
__entry->domain_id, __entry->same_exec, __entry->logged,
__entry->blockers_type == LANDLOCK_REQUEST_NET_ACCESS ?
__print_flags(__entry->blockers_access, "|", _LANDLOCK_ACCESS_NET_NAMES) :
"unknown",
__entry->port)
);
/**
* landlock_deny_ptrace - Ptrace access denied by a Landlock domain
*
* @hierarchy: Denying domain's hierarchy node (never NULL); its id is the
* domain field.
* @same_exec: Whether the tracer entered the denying domain itself.
* @logged: The domain's audit-logging decision for this denial.
* @tracee_domain_id: The tracee's Landlock domain ID, or 0 if the tracee
* is unsandboxed.
* @tracee: The target task ptrace acted on (never NULL). tracee_pid is
* the init-namespace TGID (like audit's opid).
* @tracer: The tracer or proposed tracer (never NULL); for PTRACE_TRACEME
* this is the parent, not the syscall caller.
*
* Emitted when a Landlock domain denies a ptrace operation.
*/
TRACE_EVENT(landlock_deny_ptrace,
TP_PROTO(const struct landlock_hierarchy *hierarchy, bool same_exec,
bool logged, u64 tracee_domain_id,
const struct task_struct *tracee,
const struct task_struct *tracer),
TP_ARGS(hierarchy, same_exec, logged, tracee_domain_id, tracee, tracer),
TP_STRUCT__entry(
__field( u64, domain_id )
__field( bool, same_exec )
__field( bool, logged )
__field( u64, tracee_domain_id)
__field( pid_t, tracee_pid )
__string( tracee_comm, tracee->comm )
),
TP_fast_assign(
__entry->domain_id = hierarchy->id;
__entry->same_exec = same_exec;
__entry->logged = logged;
__entry->tracee_domain_id = tracee_domain_id;
__entry->tracee_pid = task_tgid_nr((struct task_struct *)tracee);
__assign_str(tracee_comm);
),
TP_printk("domain=%llx same_exec=%d logged=%d tracee_domain=%llx tracee_pid=%d tracee_comm=%s",
__entry->domain_id, __entry->same_exec, __entry->logged,
__entry->tracee_domain_id, __entry->tracee_pid,
__trace_print_untrusted_str(p, __get_str(tracee_comm),
__get_dynamic_array_len(tracee_comm) - 1))
);
/**
* landlock_deny_scope_signal - Signal delivery denied by
* LANDLOCK_SCOPE_SIGNAL
*
* @hierarchy: Denying domain's hierarchy node (never NULL); its id is the
* domain field.
* @same_exec: Whether the policy subject entered the denying domain itself.
* @logged: The domain's audit-logging decision for this denial.
* @target_domain_id: The target's Landlock domain ID, or 0 if the target
* is unsandboxed.
* @target: The task the signal was aimed at (never NULL). target_pid is
* the init-namespace TGID (like audit's opid).
* @signal: The signal selected by the denied check. Zero is a permission
* probe, not an absent value.
*
* Emitted when a Landlock domain denies signal delivery to a scoped-out
* target.
*/
TRACE_EVENT(landlock_deny_scope_signal,
TP_PROTO(const struct landlock_hierarchy *hierarchy, bool same_exec,
bool logged, u64 target_domain_id,
const struct task_struct *target, int signal),
TP_ARGS(hierarchy, same_exec, logged, target_domain_id, target, signal),
TP_STRUCT__entry(
__field( u64, domain_id )
__field( bool, same_exec )
__field( bool, logged )
__field( u64, target_domain_id)
__field( pid_t, target_pid )
__string( target_comm, target->comm )
),
TP_fast_assign(
__entry->domain_id = hierarchy->id;
__entry->same_exec = same_exec;
__entry->logged = logged;
__entry->target_domain_id = target_domain_id;
__entry->target_pid = task_tgid_nr((struct task_struct *)target);
__assign_str(target_comm);
),
TP_printk("domain=%llx same_exec=%d logged=%d target_domain=%llx target_pid=%d target_comm=%s",
__entry->domain_id, __entry->same_exec, __entry->logged,
__entry->target_domain_id, __entry->target_pid,
__trace_print_untrusted_str(p, __get_str(target_comm),
__get_dynamic_array_len(target_comm) - 1))
);
/**
* landlock_deny_scope_abstract_unix_socket - Abstract unix socket access
* denied by LANDLOCK_SCOPE_ABSTRACT_UNIX_SOCKET
*
* @hierarchy: Denying domain's hierarchy node (never NULL); its id is the
* domain field.
* @same_exec: Whether the current task entered the denying domain itself.
* @logged: The domain's audit-logging decision for this denial.
* @peer_domain_id: The peer's Landlock domain ID, or 0 if the peer is
* unsandboxed.
* @peer: Peer socket (never NULL). peer_pid is best-effort: it is 0 for
* a datagram peer (no SO_PEERCRED), so sun_path is the reliable
* peer identifier.
*
* Emitted when a Landlock domain denies access to a scoped-out abstract
* unix socket.
*/
TRACE_EVENT(landlock_deny_scope_abstract_unix_socket,
TP_PROTO(const struct landlock_hierarchy *hierarchy, bool same_exec,
bool logged, u64 peer_domain_id, const struct sock *peer),
TP_ARGS(hierarchy, same_exec, logged, peer_domain_id, peer),
TP_STRUCT__entry(
__field( u64, domain_id )
__field( bool, same_exec )
__field( bool, logged )
__field( u64, peer_domain_id )
__field( pid_t, peer_pid )
/*
* Abstract socket names are untrusted binary data from
* user space. Use __string_len because abstract names
* are not NUL-terminated; their length is determined by
* addr->len. unix_sk(peer)->addr is stable here because
* the caller (hook_unix_stream_connect or
* hook_unix_may_send) holds unix_state_lock(peer).
*/
__string_len( sun_path,
unix_sk(peer)->addr ?
unix_sk(peer)->addr->name->sun_path + 1 :
"",
unix_sk(peer)->addr ?
unix_sk(peer)->addr->len -
offsetof(struct sockaddr_un,
sun_path) - 1 :
0)
),
TP_fast_assign(
struct pid *peer_pid;
lockdep_assert_held(&unix_sk(peer)->lock);
__entry->domain_id = hierarchy->id;
__entry->same_exec = same_exec;
__entry->logged = logged;
__entry->peer_domain_id = peer_domain_id;
/*
* Best-effort (0 for a datagram peer). The caller holds the
* peer's AF_UNIX state lock, serializing published peercred
* updates. The peer socket keeps a reference to sk_peer_pid
* through pid_nr(); sun_path is the reliable identifier.
*/
peer_pid = READ_ONCE(peer->sk_peer_pid);
__entry->peer_pid = peer_pid ? pid_nr(peer_pid) : 0;
__assign_str(sun_path);
),
TP_printk("domain=%llx same_exec=%d logged=%d peer_domain=%llx peer_pid=%d sun_path=%s",
__entry->domain_id, __entry->same_exec, __entry->logged,
__entry->peer_domain_id, __entry->peer_pid,
__trace_print_untrusted_str(p, __get_str(sun_path),
__get_dynamic_array_len(sun_path) - 1))
);
#undef _LANDLOCK_FS_BLOCKER_TYPE_NAMES
#undef _LANDLOCK_NAME_ENTRY
#endif /* _TRACE_LANDLOCK_H */
/* This part must be outside protection */
#include <trace/define_trace.h>
/* clang-format on */