1 /*
2 * Copyright (c) 2011-2022 Apple Inc. All rights reserved.
3 *
4 * @APPLE_OSREFERENCE_LICENSE_HEADER_START@
5 *
6 * This file contains Original Code and/or Modifications of Original Code
7 * as defined in and that are subject to the Apple Public Source License
8 * Version 2.0 (the 'License'). You may not use this file except in
9 * compliance with the License. The rights granted to you under the License
10 * may not be used to create, or enable the creation or redistribution of,
11 * unlawful or unlicensed copies of an Apple operating system, or to
12 * circumvent, violate, or enable the circumvention or violation of, any
13 * terms of an Apple operating system software license agreement.
14 *
15 * Please obtain a copy of the License at
16 * http://www.opensource.apple.com/apsl/ and read it before using this file.
17 *
18 * The Original Code and all software distributed under the License are
19 * distributed on an 'AS IS' basis, WITHOUT WARRANTY OF ANY KIND, EITHER
20 * EXPRESS OR IMPLIED, AND APPLE HEREBY DISCLAIMS ALL SUCH WARRANTIES,
21 * INCLUDING WITHOUT LIMITATION, ANY WARRANTIES OF MERCHANTABILITY,
22 * FITNESS FOR A PARTICULAR PURPOSE, QUIET ENJOYMENT OR NON-INFRINGEMENT.
23 * Please see the License for the specific language governing rights and
24 * limitations under the License.
25 *
26 * @APPLE_OSREFERENCE_LICENSE_HEADER_END@
27 */
28 #include <string.h>
29 #include <stdlib.h>
30 #include <mach_assert.h>
31 #include <mach_ldebug.h>
32
33 #include <mach/shared_region.h>
34 #include <mach/vm_param.h>
35 #include <mach/vm_prot.h>
36 #include <mach/vm_map.h>
37 #include <mach/machine/vm_param.h>
38 #include <mach/machine/vm_types.h>
39
40 #include <mach/boolean.h>
41 #include <kern/bits.h>
42 #include <kern/ecc.h>
43 #include <kern/thread.h>
44 #include <kern/sched.h>
45 #include <kern/zalloc.h>
46 #include <kern/zalloc_internal.h>
47 #include <kern/kalloc.h>
48 #include <kern/spl.h>
49 #include <kern/startup.h>
50 #include <kern/trustcache.h>
51
52 #include <os/overflow.h>
53
54 #include <vm/pmap.h>
55 #include <vm/pmap_cs.h>
56 #include <vm/vm_map_xnu.h>
57 #include <vm/vm_kern.h>
58 #include <vm/vm_protos.h>
59 #include <vm/vm_object_internal.h>
60 #include <vm/vm_page_internal.h>
61 #include <vm/vm_pageout.h>
62 #include <vm/cpm_internal.h>
63
64
65 #include <libkern/section_keywords.h>
66 #include <sys/errno.h>
67
68 #include <libkern/amfi/amfi.h>
69 #include <sys/trusted_execution_monitor.h>
70 #include <sys/trust_caches.h>
71 #include <sys/code_signing.h>
72
73 #include <machine/atomic.h>
74 #include <machine/thread.h>
75 #include <machine/lowglobals.h>
76
77 #include <arm/caches_internal.h>
78 #include <arm/cpu_data.h>
79 #include <arm/cpu_data_internal.h>
80 #include <arm/cpu_capabilities.h>
81 #include <arm/cpu_number.h>
82 #include <arm/machine_cpu.h>
83 #include <arm/misc_protos.h>
84 #include <arm/trap_internal.h>
85 #include <arm64/sptm/pmap/pmap_internal.h>
86 #include <arm64/sptm/sptm.h>
87
88 #include <arm64/proc_reg.h>
89 #include <pexpert/arm64/boot.h>
90 #include <arm64/ppl/uat.h>
91 #if defined(KERNEL_INTEGRITY_KTRR) || defined(KERNEL_INTEGRITY_CTRR)
92 #include <arm64/amcc_rorgn.h>
93 #endif // defined(KERNEL_INTEGRITY_KTRR) || defined(KERNEL_INTEGRITY_CTRR)
94
95 #include <pexpert/device_tree.h>
96
97 #include <san/kasan.h>
98 #include <sys/cdefs.h>
99
100 #if defined(HAS_APPLE_PAC)
101 #include <ptrauth.h>
102 #endif
103
104 #ifdef CONFIG_XNUPOST
105 #include <tests/xnupost.h>
106 #endif
107
108
109 #if HIBERNATION
110 #include <IOKit/IOHibernatePrivate.h>
111 #endif /* HIBERNATION */
112
113 #define PMAP_ROOT_ALLOC_SIZE (ARM_PGBYTES)
114
115 #define ARRAY_LEN(x) (sizeof (x) / sizeof (x[0]))
116
117
118 /**
119 * Per-CPU data used to do setup and post-processing for SPTM calls.
120 * On the setup side, this structure is used to store parameters for batched SPTM operations.
121 * These parameters may be large (upwards of 1K), and given that SPTM calls are generally
122 * issued from preemption-disabled contexts anyway, it's better to store them in per-CPU
123 * data rather than the local stack.
124 * On the post-processing side, this structure exposes a pointer to the SPTM's per-CPU array
125 * of 'prev_ptes', that is the prior value encountered in each PTE at the time of the SPTM's
126 * atomic update of that PTE.
127 */
128 pmap_sptm_percpu_data_t PERCPU_DATA(pmap_sptm_percpu);
129
130 /**
131 * Reference group for global tracking of all outstanding pmap references.
132 */
133 os_refgrp_decl(static, pmap_refgrp, "pmap", NULL);
134
135 /* Boot-arg to enable/disable the use of XNU_KERNEL_RESTRICTED type in SPTM. */
136 TUNABLE(bool, use_xnu_restricted, "xnu_restricted", true);
137
138 extern u_int32_t random(void); /* from <libkern/libkern.h> */
139
140 static bool alloc_asid(pmap_t pmap);
141 static void free_asid(pmap_t pmap);
142 static void flush_mmu_tlb_region_asid_async(vm_offset_t va, size_t length, pmap_t pmap, bool last_level_only);
143 static pt_entry_t wimg_to_pte(unsigned int wimg, pmap_paddr_t pa);
144
145 const struct page_table_ops native_pt_ops =
146 {
147 .alloc_id = alloc_asid,
148 .free_id = free_asid,
149 .flush_tlb_region_async = flush_mmu_tlb_region_asid_async,
150 .wimg_to_pte = wimg_to_pte,
151 };
152
153 const struct page_table_level_info pmap_table_level_info_16k[] =
154 {
155 [0] = {
156 .size = ARM_16K_TT_L0_SIZE,
157 .offmask = ARM_16K_TT_L0_OFFMASK,
158 .shift = ARM_16K_TT_L0_SHIFT,
159 .index_mask = ARM_16K_TT_L0_INDEX_MASK,
160 .valid_mask = ARM_TTE_VALID,
161 .type_mask = ARM_TTE_TYPE_MASK,
162 .type_block = ARM_TTE_TYPE_BLOCK
163 },
164 [1] = {
165 .size = ARM_16K_TT_L1_SIZE,
166 .offmask = ARM_16K_TT_L1_OFFMASK,
167 .shift = ARM_16K_TT_L1_SHIFT,
168 .index_mask = ARM_16K_TT_L1_INDEX_MASK,
169 .valid_mask = ARM_TTE_VALID,
170 .type_mask = ARM_TTE_TYPE_MASK,
171 .type_block = ARM_TTE_TYPE_BLOCK
172 },
173 [2] = {
174 .size = ARM_16K_TT_L2_SIZE,
175 .offmask = ARM_16K_TT_L2_OFFMASK,
176 .shift = ARM_16K_TT_L2_SHIFT,
177 .index_mask = ARM_16K_TT_L2_INDEX_MASK,
178 .valid_mask = ARM_TTE_VALID,
179 .type_mask = ARM_TTE_TYPE_MASK,
180 .type_block = ARM_TTE_TYPE_BLOCK
181 },
182 [3] = {
183 .size = ARM_16K_TT_L3_SIZE,
184 .offmask = ARM_16K_TT_L3_OFFMASK,
185 .shift = ARM_16K_TT_L3_SHIFT,
186 .index_mask = ARM_16K_TT_L3_INDEX_MASK,
187 .valid_mask = ARM_PTE_TYPE_VALID,
188 .type_mask = ARM_PTE_TYPE_MASK,
189 .type_block = ARM_TTE_TYPE_L3BLOCK
190 }
191 };
192
193 const struct page_table_level_info pmap_table_level_info_4k[] =
194 {
195 [0] = {
196 .size = ARM_4K_TT_L0_SIZE,
197 .offmask = ARM_4K_TT_L0_OFFMASK,
198 .shift = ARM_4K_TT_L0_SHIFT,
199 .index_mask = ARM_4K_TT_L0_INDEX_MASK,
200 .valid_mask = ARM_TTE_VALID,
201 .type_mask = ARM_TTE_TYPE_MASK,
202 .type_block = ARM_TTE_TYPE_BLOCK
203 },
204 [1] = {
205 .size = ARM_4K_TT_L1_SIZE,
206 .offmask = ARM_4K_TT_L1_OFFMASK,
207 .shift = ARM_4K_TT_L1_SHIFT,
208 .index_mask = ARM_4K_TT_L1_INDEX_MASK,
209 .valid_mask = ARM_TTE_VALID,
210 .type_mask = ARM_TTE_TYPE_MASK,
211 .type_block = ARM_TTE_TYPE_BLOCK
212 },
213 [2] = {
214 .size = ARM_4K_TT_L2_SIZE,
215 .offmask = ARM_4K_TT_L2_OFFMASK,
216 .shift = ARM_4K_TT_L2_SHIFT,
217 .index_mask = ARM_4K_TT_L2_INDEX_MASK,
218 .valid_mask = ARM_TTE_VALID,
219 .type_mask = ARM_TTE_TYPE_MASK,
220 .type_block = ARM_TTE_TYPE_BLOCK
221 },
222 [3] = {
223 .size = ARM_4K_TT_L3_SIZE,
224 .offmask = ARM_4K_TT_L3_OFFMASK,
225 .shift = ARM_4K_TT_L3_SHIFT,
226 .index_mask = ARM_4K_TT_L3_INDEX_MASK,
227 .valid_mask = ARM_PTE_TYPE_VALID,
228 .type_mask = ARM_PTE_TYPE_MASK,
229 .type_block = ARM_TTE_TYPE_L3BLOCK
230 }
231 };
232
233 const struct page_table_level_info pmap_table_level_info_4k_stage2[] =
234 {
235 [0] = { /* Unused */
236 .size = ARM_4K_TT_L0_SIZE,
237 .offmask = ARM_4K_TT_L0_OFFMASK,
238 .shift = ARM_4K_TT_L0_SHIFT,
239 .index_mask = ARM_4K_TT_L0_INDEX_MASK,
240 .valid_mask = ARM_TTE_VALID,
241 .type_mask = ARM_TTE_TYPE_MASK,
242 .type_block = ARM_TTE_TYPE_BLOCK
243 },
244 [1] = { /* Concatenated, so index mask is larger than normal */
245 .size = ARM_4K_TT_L1_SIZE,
246 .offmask = ARM_4K_TT_L1_OFFMASK,
247 .shift = ARM_4K_TT_L1_SHIFT,
248 #ifdef ARM_4K_TT_L1_40_BIT_CONCATENATED_INDEX_MASK
249 .index_mask = ARM_4K_TT_L1_40_BIT_CONCATENATED_INDEX_MASK,
250 #else
251 .index_mask = ARM_4K_TT_L1_INDEX_MASK,
252 #endif
253 .valid_mask = ARM_TTE_VALID,
254 .type_mask = ARM_TTE_TYPE_MASK,
255 .type_block = ARM_TTE_TYPE_BLOCK
256 },
257 [2] = {
258 .size = ARM_4K_TT_L2_SIZE,
259 .offmask = ARM_4K_TT_L2_OFFMASK,
260 .shift = ARM_4K_TT_L2_SHIFT,
261 .index_mask = ARM_4K_TT_L2_INDEX_MASK,
262 .valid_mask = ARM_TTE_VALID,
263 .type_mask = ARM_TTE_TYPE_MASK,
264 .type_block = ARM_TTE_TYPE_BLOCK
265 },
266 [3] = {
267 .size = ARM_4K_TT_L3_SIZE,
268 .offmask = ARM_4K_TT_L3_OFFMASK,
269 .shift = ARM_4K_TT_L3_SHIFT,
270 .index_mask = ARM_4K_TT_L3_INDEX_MASK,
271 .valid_mask = ARM_PTE_TYPE_VALID,
272 .type_mask = ARM_PTE_TYPE_MASK,
273 .type_block = ARM_TTE_TYPE_L3BLOCK
274 }
275 };
276
277 const struct page_table_attr pmap_pt_attr_4k = {
278 .pta_level_info = pmap_table_level_info_4k,
279 .pta_root_level = (T0SZ_BOOT - 16) / 9,
280 #if __ARM_MIXED_PAGE_SIZE__
281 .pta_commpage_level = PMAP_TT_L2_LEVEL,
282 #else /* __ARM_MIXED_PAGE_SIZE__ */
283 #if __ARM_16K_PG__
284 .pta_commpage_level = PMAP_TT_L2_LEVEL,
285 #else /* __ARM_16K_PG__ */
286 .pta_commpage_level = PMAP_TT_L1_LEVEL,
287 #endif /* __ARM_16K_PG__ */
288 #endif /* __ARM_MIXED_PAGE_SIZE__ */
289 .pta_max_level = PMAP_TT_L3_LEVEL,
290 .pta_ops = &native_pt_ops,
291 .ap_ro = ARM_PTE_AP(AP_RORO),
292 .ap_rw = ARM_PTE_AP(AP_RWRW),
293 .ap_rona = ARM_PTE_AP(AP_RONA),
294 .ap_rwna = ARM_PTE_AP(AP_RWNA),
295 .ap_xn = ARM_PTE_PNX | ARM_PTE_NX,
296 .ap_x = ARM_PTE_PNX,
297 #if __ARM_MIXED_PAGE_SIZE__
298 .pta_tcr_value = TCR_EL1_4KB,
299 #endif /* __ARM_MIXED_PAGE_SIZE__ */
300 .pta_page_size = 4096,
301 .pta_page_shift = 12,
302 .geometry_id = SPTM_PT_GEOMETRY_4K,
303 };
304
305 const struct page_table_attr pmap_pt_attr_16k = {
306 .pta_level_info = pmap_table_level_info_16k,
307 .pta_root_level = PMAP_TT_L1_LEVEL,
308 .pta_commpage_level = PMAP_TT_L2_LEVEL,
309 .pta_max_level = PMAP_TT_L3_LEVEL,
310 .pta_ops = &native_pt_ops,
311 .ap_ro = ARM_PTE_AP(AP_RORO),
312 .ap_rw = ARM_PTE_AP(AP_RWRW),
313 .ap_rona = ARM_PTE_AP(AP_RONA),
314 .ap_rwna = ARM_PTE_AP(AP_RWNA),
315 .ap_xn = ARM_PTE_PNX | ARM_PTE_NX,
316 .ap_x = ARM_PTE_PNX,
317 #if __ARM_MIXED_PAGE_SIZE__
318 .pta_tcr_value = TCR_EL1_16KB,
319 #endif /* __ARM_MIXED_PAGE_SIZE__ */
320 .pta_page_size = 16384,
321 .pta_page_shift = 14,
322 .geometry_id = SPTM_PT_GEOMETRY_16K,
323 };
324
325 #if __ARM_16K_PG__
326 const struct page_table_attr * const native_pt_attr = &pmap_pt_attr_16k;
327 #else /* !__ARM_16K_PG__ */
328 const struct page_table_attr * const native_pt_attr = &pmap_pt_attr_4k;
329 #endif /* !__ARM_16K_PG__ */
330
331
332 #if DEVELOPMENT || DEBUG
333 int vm_footprint_suspend_allowed = 1;
334
335 extern int pmap_ledgers_panic;
336 extern int pmap_ledgers_panic_leeway;
337
338 #endif /* DEVELOPMENT || DEBUG */
339
340 #if DEVELOPMENT || DEBUG
341 #define PMAP_FOOTPRINT_SUSPENDED(pmap) \
342 (current_thread()->pmap_footprint_suspended)
343 #else /* DEVELOPMENT || DEBUG */
344 #define PMAP_FOOTPRINT_SUSPENDED(pmap) (FALSE)
345 #endif /* DEVELOPMENT || DEBUG */
346
347 #define PMAP_TT_ALLOCATE_NOWAIT 0x1
348
349
350 /* Keeps track of whether the pmap has been bootstrapped */
351 SECURITY_READ_ONLY_LATE(bool) pmap_bootstrapped = false;
352
353 /*
354 * Represents a tlb range that will be flushed before returning from the pmap.
355 * Used by phys_attribute_clear_range to defer flushing pages in this range until
356 * the end of the operation, and to accumulate batched operations for submission
357 * to the SPTM as a performance optimization.
358 */
359 typedef struct pmap_tlb_flush_range {
360 /* Address space in which the flush region resides */
361 pmap_t ptfr_pmap;
362
363 /* Page-aligned beginning of the flush region */
364 vm_map_address_t ptfr_start;
365
366 /* Page-aligned non-inclusive end of the flush region */
367 vm_map_address_t ptfr_end;
368
369 /**
370 * Address of current PTE position in ptfr_pmap's [ptfr_start, ptfr_end) region.
371 * This is meant to be set up by the caller of pmap_page_protect_options_with_flush_range()
372 * or arm_force_fast_fault_with_flush_range(), and used by those functions to determine
373 * when a given mapping can be added to the SPTM's per-CPU region templates array vs.
374 * the more complex task of adding it to the disjoint ops array.
375 */
376 pt_entry_t *current_ptep;
377
378 /**
379 * Starting VA for any not-yet-submitted per-CPU region templates. This is meant to be
380 * set up by the caller of pmap_page_protect_options_with_flush_range() or
381 * arm_force_fast_fault_with_flush_range() and used by pmap_multipage_op_submit_region()
382 * when issuing the SPTM call to purge any pending region ops.
383 */
384 vm_map_address_t pending_region_start;
385
386 /**
387 * Number of entries in the per-CPU SPTM region templates array which have not
388 * yet been submitted to the SPTM.
389 */
390 unsigned int pending_region_entries;
391
392 /**
393 * Indicates whether at least one region entry was added to the per-CPU region ops
394 * array since the last time this field was checked. Intended to be cleared by the
395 * caller.
396 */
397 bool region_entry_added;
398
399 /**
400 * Marker for the current paddr "header" entry in the per-CPU SPTM disjoint ops array.
401 * This field is intended to be modified only by pmap_multipage_op_submit_disjoint()
402 * and pmap_multipage_op_add_page(), and should be treated as opaque by callers
403 * of those functions.
404 */
405 sptm_update_disjoint_multipage_op_t *current_header;
406
407 /**
408 * Position in the per-CPU SPTM ops array of the first ordinary
409 * sptm_disjoint_op_t entry following [current_header]. This is the starting
410 * point at which mappings should be inserted for the page described by
411 * [current_header].
412 */
413 unsigned int current_header_first_mapping_index;
414
415 /**
416 * Number of entries in the per-CPU SPTM disjoint ops array, including paddr headers,
417 * which have not yet been submitted to the SPTM.
418 */
419 unsigned int pending_disjoint_entries;
420
421 /**
422 * This field is used by the preemption check interval logic on the
423 * phys_attribute_clear_range() path to determine when sufficient
424 * forward progress has been made to check for and (if necessary)
425 * handle pending preemption.
426 */
427 unsigned int processed_entries;
428
429 /**
430 * Indicates whether the top-level caller needs to flush the TLB for
431 * the region in [ptfr_pmap] described by [ptfr_start, ptfr_end).
432 * This will be set if the SPTM indicates that it needed to alter
433 * any valid mapping within this region and SPTM_UPDATE_DEFER_TLBI
434 * was passed to the relevant SPTM call(s).
435 */
436 bool ptfr_flush_needed;
437 } pmap_tlb_flush_range_t;
438
439
440
441 /* Virtual memory region for early allocation */
442 #define VREGION1_HIGH_WINDOW (PE_EARLY_BOOT_VA)
443 #define VREGION1_START ((VM_MAX_KERNEL_ADDRESS & CPUWINDOWS_BASE_MASK) - VREGION1_HIGH_WINDOW)
444 #define VREGION1_SIZE (trunc_page(VM_MAX_KERNEL_ADDRESS - (VREGION1_START)))
445
446 extern uint8_t bootstrap_pagetables[];
447
448 extern unsigned int not_in_kdp;
449
450 extern vm_offset_t first_avail;
451
452 extern vm_offset_t virtual_space_start; /* Next available kernel VA */
453 extern vm_offset_t virtual_space_end; /* End of kernel address space */
454 extern vm_offset_t static_memory_end;
455
456 extern const vm_map_address_t physmap_base;
457 extern const vm_map_address_t physmap_end;
458
459 extern int maxproc, hard_maxproc;
460
461 extern bool sdsb_io_rgns_present;
462
463 vm_address_t MARK_AS_PMAP_DATA image4_slab = 0;
464 vm_address_t MARK_AS_PMAP_DATA image4_late_slab = 0;
465
466 /* The number of address bits one TTBR can cover. */
467 #define PGTABLE_ADDR_BITS (64ULL - T0SZ_BOOT)
468
469 /*
470 * The bounds on our TTBRs. These are for sanity checking that
471 * an address is accessible by a TTBR before we attempt to map it.
472 */
473
474 /* The level of the root of a page table. */
475 const uint64_t arm64_root_pgtable_level = (3 - ((PGTABLE_ADDR_BITS - 1 - ARM_PGSHIFT) / (ARM_PGSHIFT - TTE_SHIFT)));
476
477 /* The number of entries in the root TT of a page table. */
478 const uint64_t arm64_root_pgtable_num_ttes = (2 << ((PGTABLE_ADDR_BITS - 1 - ARM_PGSHIFT) % (ARM_PGSHIFT - TTE_SHIFT)));
479
480 struct pmap kernel_pmap_store MARK_AS_PMAP_DATA;
481 const pmap_t kernel_pmap = &kernel_pmap_store;
482
483 static SECURITY_READ_ONLY_LATE(zone_t) pmap_zone; /* zone of pmap structures */
484
485 MARK_AS_PMAP_DATA SIMPLE_LOCK_DECLARE(pmaps_lock, 0);
486 queue_head_t map_pmap_list MARK_AS_PMAP_DATA;
487
488 typedef struct tt_free_entry {
489 struct tt_free_entry *next;
490 } tt_free_entry_t;
491
492 unsigned int inuse_user_ttepages_count MARK_AS_PMAP_DATA = 0; /* non-root, non-leaf user pagetable pages, in units of PAGE_SIZE */
493 unsigned int inuse_user_ptepages_count MARK_AS_PMAP_DATA = 0; /* leaf user pagetable pages, in units of PAGE_SIZE */
494 unsigned int inuse_user_tteroot_count MARK_AS_PMAP_DATA = 0; /* root user pagetables, in units of PMAP_ROOT_ALLOC_SIZE */
495 unsigned int inuse_kernel_ttepages_count MARK_AS_PMAP_DATA = 0; /* non-root, non-leaf kernel pagetable pages, in units of PAGE_SIZE */
496 unsigned int inuse_kernel_ptepages_count MARK_AS_PMAP_DATA = 0; /* leaf kernel pagetable pages, in units of PAGE_SIZE */
497 unsigned int inuse_kernel_tteroot_count MARK_AS_PMAP_DATA = 0; /* root kernel pagetables, in units of PMAP_ROOT_ALLOC_SIZE */
498 _Atomic unsigned int inuse_iommu_pages_count[SPTM_IOMMUS_N_IDS] = {0}; /* number of active pages for each IOMMU class */
499
500 SECURITY_READ_ONLY_LATE(tt_entry_t *) invalid_tte = 0;
501 SECURITY_READ_ONLY_LATE(pmap_paddr_t) invalid_ttep = 0;
502
503 SECURITY_READ_ONLY_LATE(tt_entry_t *) cpu_tte = 0; /* set by arm_vm_init() - keep out of bss */
504 SECURITY_READ_ONLY_LATE(pmap_paddr_t) cpu_ttep = 0; /* set by arm_vm_init() - phys tte addr */
505
506 /* Lock group used for all pmap object locks. */
507 lck_grp_t pmap_lck_grp MARK_AS_PMAP_DATA;
508
509 #if DEVELOPMENT || DEBUG
510 int nx_enabled = 1; /* enable no-execute protection */
511 int allow_data_exec = 0; /* No apps may execute data */
512 int allow_stack_exec = 0; /* No apps may execute from the stack */
513 unsigned long pmap_asid_flushes MARK_AS_PMAP_DATA = 0;
514 unsigned long pmap_asid_hits MARK_AS_PMAP_DATA = 0;
515 unsigned long pmap_asid_misses MARK_AS_PMAP_DATA = 0;
516 unsigned long pmap_speculation_restrictions MARK_AS_PMAP_DATA = 0;
517 #else /* DEVELOPMENT || DEBUG */
518 const int nx_enabled = 1; /* enable no-execute protection */
519 const int allow_data_exec = 0; /* No apps may execute data */
520 const int allow_stack_exec = 0; /* No apps may execute from the stack */
521 #endif /* DEVELOPMENT || DEBUG */
522
523
524 #if MACH_ASSERT
525 static void pmap_check_ledgers(pmap_t pmap);
526 #else
527 static inline void
pmap_check_ledgers(__unused pmap_t pmap)528 pmap_check_ledgers(__unused pmap_t pmap)
529 {
530 }
531 #endif /* MACH_ASSERT */
532
533 SIMPLE_LOCK_DECLARE(phys_backup_lock, 0);
534
535 SECURITY_READ_ONLY_LATE(pmap_paddr_t) vm_first_phys = (pmap_paddr_t) 0;
536 SECURITY_READ_ONLY_LATE(pmap_paddr_t) vm_last_phys = (pmap_paddr_t) 0;
537
538 SECURITY_READ_ONLY_LATE(boolean_t) pmap_initialized = FALSE; /* Has pmap_init completed? */
539
540 SECURITY_READ_ONLY_LATE(vm_map_offset_t) arm_pmap_max_offset_default = 0x0;
541
542 /* end of shared region + 512MB for various purposes */
543 #define ARM64_MIN_MAX_ADDRESS (SHARED_REGION_BASE_ARM64 + SHARED_REGION_SIZE_ARM64 + 0x20000000)
544 _Static_assert((ARM64_MIN_MAX_ADDRESS > SHARED_REGION_BASE_ARM64) && (ARM64_MIN_MAX_ADDRESS <= MACH_VM_MAX_ADDRESS),
545 "Minimum address space size outside allowable range");
546
547 // Max offset is 15.375GB for devices with "large" memory config
548 #define ARM64_MAX_OFFSET_DEVICE_LARGE (ARM64_MIN_MAX_ADDRESS + 0x138000000)
549 // Max offset is 11.375GB for devices with "small" memory config
550 #define ARM64_MAX_OFFSET_DEVICE_SMALL (ARM64_MIN_MAX_ADDRESS + 0x38000000)
551
552
553 _Static_assert((ARM64_MAX_OFFSET_DEVICE_LARGE > ARM64_MIN_MAX_ADDRESS) && (ARM64_MAX_OFFSET_DEVICE_LARGE <= MACH_VM_MAX_ADDRESS),
554 "Large device address space size outside allowable range");
555 _Static_assert((ARM64_MAX_OFFSET_DEVICE_SMALL > ARM64_MIN_MAX_ADDRESS) && (ARM64_MAX_OFFSET_DEVICE_SMALL <= MACH_VM_MAX_ADDRESS),
556 "Small device address space size outside allowable range");
557
558 # ifdef XNU_TARGET_OS_OSX
559 SECURITY_READ_ONLY_LATE(vm_map_offset_t) arm64_pmap_max_offset_default = MACH_VM_MAX_ADDRESS;
560 # else
561 SECURITY_READ_ONLY_LATE(vm_map_offset_t) arm64_pmap_max_offset_default = 0x0;
562 # endif
563
564 #if PMAP_PANIC_DEV_WIMG_ON_MANAGED && (DEVELOPMENT || DEBUG)
565 SECURITY_READ_ONLY_LATE(boolean_t) pmap_panic_dev_wimg_on_managed = TRUE;
566 #else
567 SECURITY_READ_ONLY_LATE(boolean_t) pmap_panic_dev_wimg_on_managed = FALSE;
568 #endif
569
570 MARK_AS_PMAP_DATA SIMPLE_LOCK_DECLARE(asid_lock, 0);
571 SECURITY_READ_ONLY_LATE(uint32_t) pmap_max_asids = 0;
572 SECURITY_READ_ONLY_LATE(static bitmap_t*) asid_bitmap;
573 #if !HAS_16BIT_ASID
574 static bitmap_t asid_plru_bitmap[BITMAP_LEN(MAX_HW_ASIDS)] MARK_AS_PMAP_DATA;
575 static uint64_t asid_plru_generation[BITMAP_LEN(MAX_HW_ASIDS)] MARK_AS_PMAP_DATA = {0};
576 static uint64_t asid_plru_gencount MARK_AS_PMAP_DATA = 0;
577 SECURITY_READ_ONLY_LATE(int) pmap_asid_plru = 1;
578 #else
579 static uint16_t last_allocated_asid = 0;
580 #endif /* !HAS_16BIT_ASID */
581
582
583 SECURITY_READ_ONLY_LATE(static pmap_paddr_t) commpage_default_table;
584 //SECURITY_READ_ONLY_LATE(static pmap_paddr_t) commpage32_default_table;
585 #if __ARM_MIXED_PAGE_SIZE__
586 SECURITY_READ_ONLY_LATE(static pmap_paddr_t) commpage_4k_table;
587 //SECURITY_READ_ONLY_LATE(static pmap_paddr_t) commpage32_4k_table;
588 #endif
589 SECURITY_READ_ONLY_LATE(static pmap_paddr_t) commpage_data_pa = 0;
590 SECURITY_READ_ONLY_LATE(static pmap_paddr_t) commpage_text_pa = 0;
591 SECURITY_READ_ONLY_LATE(static vm_map_address_t) commpage_text_user_va = 0;
592 SECURITY_READ_ONLY_LATE(static pmap_paddr_t) commpage_ro_data_pa = 0;
593
594
595 #if (DEVELOPMENT || DEBUG)
596 /* Caches whether the SPTM sysreg API has been enabled by the SPTM */
597 SECURITY_READ_ONLY_LATE(static bool) sptm_sysreg_available = false;
598 #endif /* (DEVELOPMENT || DEBUG) */
599
600 /* PTE Define Macros */
601
602 #ifndef SPTM_PTE_IN_FLIGHT_MARKER
603 /* SPTM TODO: Get rid of this once we export SPTM_PTE_IN_FLIGHT_MARKER from the SPTM. */
604 #define SPTM_PTE_IN_FLIGHT_MARKER 0x80U
605 #endif /* SPTM_PTE_IN_FLIGHT_MARKER */
606
607 /**
608 * Determine whether a PTE has been marked as compressed. This function also panics if
609 * the PTE contains bits that shouldn't be present in a compressed PTE, which is most of them.
610 *
611 * @param pte the PTE contents to check
612 * @param ptep the address of the PTE contents, for diagnostic purposes only
613 *
614 * @return true if the PTE is compressed, false otherwise
615 */
616 static inline bool
pte_is_compressed(pt_entry_t pte,pt_entry_t * ptep)617 pte_is_compressed(pt_entry_t pte, pt_entry_t *ptep)
618 {
619 const bool compressed = (((pte & ARM_PTE_TYPE_VALID) == ARM_PTE_TYPE_FAULT) && (pte & ARM_PTE_COMPRESSED));
620 /**
621 * Check for bits that shouldn't be present in a compressed PTE. This is everything except the
622 * compressed/compressed-alt bits, as well as the SPTM's in-flight marker which may be set while
623 * the SPTM is in the process of flushing the TLBs after marking a previously-valid PTE as
624 * compressed.
625 */
626 if (__improbable(compressed && (pte & ~(ARM_PTE_COMPRESSED_MASK | SPTM_PTE_IN_FLIGHT_MARKER)))) {
627 panic("compressed PTE %p 0x%llx has extra bits 0x%llx: corrupted?",
628 ptep, pte, pte & ~(ARM_PTE_COMPRESSED_MASK | SPTM_PTE_IN_FLIGHT_MARKER));
629 }
630 return compressed;
631 }
632
633 #define pte_is_wired(pte) \
634 (((pte) & ARM_PTE_WIRED_MASK) == ARM_PTE_WIRED)
635
636 #define pte_was_writeable(pte) \
637 (((pte) & ARM_PTE_WRITEABLE) == ARM_PTE_WRITEABLE)
638
639 #define pte_set_was_writeable(pte, was_writeable) \
640 do { \
641 if ((was_writeable)) { \
642 (pte) |= ARM_PTE_WRITEABLE; \
643 } else { \
644 (pte) &= ~ARM_PTE_WRITEABLE; \
645 } \
646 } while(0)
647
648
649 /**
650 * Updated wired-mapping accountings in the PTD and ledger.
651 *
652 * @param pmap The pmap against which to update accounting
653 * @param pte_p The PTE whose wired state is being changed
654 * @param wired Indicates whether the PTE is being wired or unwired.
655 */
656 static inline void
pte_update_wiredcnt(pmap_t pmap,pt_entry_t * pte_p,boolean_t wired)657 pte_update_wiredcnt(pmap_t pmap, pt_entry_t *pte_p, boolean_t wired)
658 {
659 const pt_attr_t * const pt_attr = pmap_get_pt_attr(pmap);
660 unsigned short *ptd_wiredcnt_ptr = &(ptep_get_info(pte_p)->wiredcnt);
661 if (wired) {
662 if (__improbable(os_atomic_inc_orig(ptd_wiredcnt_ptr, relaxed) == UINT16_MAX)) {
663 panic("pmap %p (pte %p): wired count overflow", pmap, pte_p);
664 }
665 pmap_ledger_credit(pmap, task_ledgers.wired_mem, pt_attr_page_size(pt_attr) * PAGE_RATIO);
666 } else {
667 if (__improbable(os_atomic_dec_orig(ptd_wiredcnt_ptr, relaxed) == 0)) {
668 panic("pmap %p (pte %p): wired count underflow", pmap, pte_p);
669 }
670 pmap_ledger_debit(pmap, task_ledgers.wired_mem, pt_attr_page_size(pt_attr) * PAGE_RATIO);
671 }
672 }
673
674 /*
675 * Synchronize updates to PTEs that were previously invalid or had the AF bit cleared,
676 * therefore not requiring TLBI. Use a store-load barrier to ensure subsequent loads
677 * will observe the updated PTE.
678 */
679 #define FLUSH_PTE() \
680 __builtin_arm_dmb(DMB_ISH);
681
682 /*
683 * Synchronize updates to PTEs that were previously valid and thus may be cached in
684 * TLBs. DSB is required to ensure the PTE stores have completed prior to the ensuing
685 * TLBI. This should only require a store-store barrier, as subsequent accesses in
686 * program order will not issue until the DSB completes. Prior loads may be reordered
687 * after the barrier, but their behavior should not be materially affected by the
688 * reordering. For fault-driven PTE updates such as COW, PTE contents should not
689 * matter for loads until the access is re-driven well after the TLB update is
690 * synchronized. For "involuntary" PTE access restriction due to paging lifecycle,
691 * we should be in a position to handle access faults. For "voluntary" PTE access
692 * restriction due to unmapping or protection, the decision to restrict access should
693 * have a data dependency on prior loads in order to avoid a data race.
694 */
695 #define FLUSH_PTE_STRONG() \
696 __builtin_arm_dsb(DSB_ISHST);
697
698 /**
699 * Write enough page table entries to map a single VM page. On systems where the
700 * VM page size does not match the hardware page size, multiple page table
701 * entries will need to be written.
702 *
703 * @note This function does not emit a barrier to ensure these page table writes
704 * have completed before continuing. This is commonly needed. In the case
705 * where a DMB or DSB barrier is needed, then use the write_pte() and
706 * write_pte_strong() functions respectively instead of this one.
707 *
708 * @param ptep Pointer to the first page table entry to update.
709 * @param pte The value to write into each page table entry. In the case that
710 * multiple PTEs are updated to a non-empty value, then the address
711 * in this value will automatically be incremented for each PTE
712 * write.
713 */
714 static void
write_pte_fast(pt_entry_t * ptep,pt_entry_t pte)715 write_pte_fast(pt_entry_t *ptep, pt_entry_t pte)
716 {
717 /**
718 * The PAGE_SHIFT (and in turn, the PAGE_RATIO) can be a variable on some
719 * systems, which is why it's checked at runtime instead of compile time.
720 * The "unreachable" warning needs to be suppressed because it still is a
721 * compile time constant on some systems.
722 */
723 __unreachable_ok_push
724 if (TEST_PAGE_RATIO_4) {
725 if (((uintptr_t)ptep) & 0x1f) {
726 panic("%s: PTE write is unaligned, ptep=%p, pte=%p",
727 __func__, ptep, (void*)pte);
728 }
729
730 if ((pte & ~ARM_PTE_COMPRESSED_MASK) == ARM_PTE_EMPTY) {
731 /**
732 * If we're writing an empty/compressed PTE value, then don't
733 * auto-increment the address for each PTE write.
734 */
735 *ptep = pte;
736 *(ptep + 1) = pte;
737 *(ptep + 2) = pte;
738 *(ptep + 3) = pte;
739 } else {
740 *ptep = pte;
741 *(ptep + 1) = pte | 0x1000;
742 *(ptep + 2) = pte | 0x2000;
743 *(ptep + 3) = pte | 0x3000;
744 }
745 } else {
746 *ptep = pte;
747 }
748 __unreachable_ok_pop
749 }
750
751 /**
752 * Writes enough page table entries to map a single VM page and then ensures
753 * those writes complete by executing a Data Memory Barrier.
754 *
755 * @note The DMB issued by this function is not strong enough to protect against
756 * TLB invalidates from being reordered above the PTE writes. If a TLBI
757 * instruction is going to immediately be called after this write, it's
758 * recommended to call write_pte_strong() instead of this function.
759 *
760 * See the function header for write_pte_fast() for more details on the
761 * parameters.
762 */
763 void
write_pte(pt_entry_t * ptep,pt_entry_t pte)764 write_pte(pt_entry_t *ptep, pt_entry_t pte)
765 {
766 write_pte_fast(ptep, pte);
767 FLUSH_PTE();
768 }
769
770 /**
771 * Retrieve the pmap structure for the thread running on the current CPU.
772 */
773 pmap_t
current_pmap()774 current_pmap()
775 {
776 const pmap_t current = vm_map_pmap(current_thread()->map);
777 assert(current != NULL);
778 return current;
779 }
780
781 #if DEVELOPMENT || DEBUG
782
783 /*
784 * Trace levels are controlled by a bitmask in which each
785 * level can be enabled/disabled by the (1<<level) position
786 * in the boot arg
787 * Level 0: PPL extension functionality
788 * Level 1: pmap lifecycle (create/destroy/switch)
789 * Level 2: mapping lifecycle (enter/remove/protect/nest/unnest)
790 * Level 3: internal state management (attributes/fast-fault)
791 * Level 4-7: TTE traces for paging levels 0-3. TTBs are traced at level 4.
792 */
793
794 SECURITY_READ_ONLY_LATE(unsigned int) pmap_trace_mask = 0;
795
796 #define PMAP_TRACE(level, ...) \
797 if (__improbable((1 << (level)) & pmap_trace_mask)) { \
798 KDBG_RELEASE(__VA_ARGS__); \
799 }
800 #else /* DEVELOPMENT || DEBUG */
801
802 #define PMAP_TRACE(level, ...)
803
804 #endif /* DEVELOPMENT || DEBUG */
805
806
807 /*
808 * Internal function prototypes (forward declarations).
809 */
810
811 static vm_map_size_t pmap_user_va_size(pmap_t pmap);
812
813 static void pmap_set_reference(ppnum_t pn);
814
815 pmap_paddr_t pmap_vtophys(pmap_t pmap, addr64_t va);
816
817 static kern_return_t pmap_expand(
818 pmap_t, vm_map_address_t, unsigned int options, unsigned int level);
819
820 static void pmap_remove_range(pmap_t, vm_map_address_t, vm_map_address_t);
821
822 static tt_entry_t *pmap_tt1_allocate(pmap_t, uint8_t);
823
824 static void pmap_tt1_deallocate(pmap_t, tt_entry_t *);
825
826 static kern_return_t pmap_tt_allocate(
827 pmap_t, tt_entry_t **, unsigned int, unsigned int);
828
829 const unsigned int arm_hardware_page_size = ARM_PGBYTES;
830 const unsigned int arm_pt_desc_size = sizeof(pt_desc_t);
831 const unsigned int arm_pt_root_size = PMAP_ROOT_ALLOC_SIZE;
832
833 static void pmap_unmap_commpage(
834 pmap_t pmap);
835
836 static boolean_t
837 pmap_is_64bit(pmap_t);
838
839
840 static void pmap_flush_tlb_for_paddr_async(pmap_paddr_t);
841
842 static void pmap_update_pp_attr_wimg_bits_locked(unsigned int, unsigned int);
843
844 static boolean_t arm_clear_fast_fault(
845 ppnum_t ppnum,
846 vm_prot_t fault_type,
847 uintptr_t pvh,
848 pt_entry_t *pte_p,
849 pp_attr_t attrs_to_clear);
850
851 static void pmap_trim_self(pmap_t pmap);
852 static void pmap_trim_subord(pmap_t subord);
853
854
855 /*
856 * Temporary prototypes, while we wait for pmap_enter to move to taking an
857 * address instead of a page number.
858 */
859 kern_return_t
860 pmap_enter(
861 pmap_t pmap,
862 vm_map_address_t v,
863 ppnum_t pn,
864 vm_prot_t prot,
865 vm_prot_t fault_type,
866 unsigned int flags,
867 boolean_t wired,
868 pmap_mapping_type_t mapping_type);
869
870 static kern_return_t
871 pmap_enter_addr(
872 pmap_t pmap,
873 vm_map_address_t v,
874 pmap_paddr_t pa,
875 vm_prot_t prot,
876 vm_prot_t fault_type,
877 unsigned int flags,
878 boolean_t wired,
879 pmap_mapping_type_t mapping_type);
880
881 kern_return_t
882 pmap_enter_options_addr(
883 pmap_t pmap,
884 vm_map_address_t v,
885 pmap_paddr_t pa,
886 vm_prot_t prot,
887 vm_prot_t fault_type,
888 unsigned int flags,
889 boolean_t wired,
890 unsigned int options,
891 __unused void *arg,
892 pmap_mapping_type_t mapping_type);
893
894 #ifdef CONFIG_XNUPOST
895 kern_return_t pmap_test(void);
896 #endif /* CONFIG_XNUPOST */
897
898 PMAP_SUPPORT_PROTOTYPES(
899 kern_return_t,
900 arm_fast_fault, (pmap_t pmap,
901 vm_map_address_t va,
902 vm_prot_t fault_type,
903 bool was_af_fault,
904 bool from_user), ARM_FAST_FAULT_INDEX);
905
906 PMAP_SUPPORT_PROTOTYPES(
907 boolean_t,
908 arm_force_fast_fault, (ppnum_t ppnum,
909 vm_prot_t allow_mode,
910 int options), ARM_FORCE_FAST_FAULT_INDEX);
911
912 MARK_AS_PMAP_TEXT static boolean_t
913 arm_force_fast_fault_with_flush_range(
914 ppnum_t ppnum,
915 vm_prot_t allow_mode,
916 int options,
917 locked_pvh_t *locked_pvh,
918 pp_attr_t bits_to_clear,
919 pmap_tlb_flush_range_t *flush_range);
920
921 PMAP_SUPPORT_PROTOTYPES(
922 void,
923 pmap_batch_set_cache_attributes, (
924 const unified_page_list_t * page_list,
925 unsigned int cacheattr,
926 bool update_attr_table), PMAP_BATCH_SET_CACHE_ATTRIBUTES_INDEX);
927
928 PMAP_SUPPORT_PROTOTYPES(
929 void,
930 pmap_change_wiring, (pmap_t pmap,
931 vm_map_address_t v,
932 boolean_t wired), PMAP_CHANGE_WIRING_INDEX);
933
934 PMAP_SUPPORT_PROTOTYPES(
935 pmap_t,
936 pmap_create_options, (ledger_t ledger,
937 vm_map_size_t size,
938 unsigned int flags,
939 kern_return_t * kr), PMAP_CREATE_INDEX);
940
941 PMAP_SUPPORT_PROTOTYPES(
942 void,
943 pmap_destroy, (pmap_t pmap), PMAP_DESTROY_INDEX);
944
945 PMAP_SUPPORT_PROTOTYPES(
946 kern_return_t,
947 pmap_enter_options, (pmap_t pmap,
948 vm_map_address_t v,
949 pmap_paddr_t pa,
950 vm_prot_t prot,
951 vm_prot_t fault_type,
952 unsigned int flags,
953 boolean_t wired,
954 unsigned int options,
955 pmap_mapping_type_t mapping_type), PMAP_ENTER_OPTIONS_INDEX);
956
957 PMAP_SUPPORT_PROTOTYPES(
958 pmap_paddr_t,
959 pmap_find_pa, (pmap_t pmap,
960 addr64_t va), PMAP_FIND_PA_INDEX);
961
962 PMAP_SUPPORT_PROTOTYPES(
963 kern_return_t,
964 pmap_insert_commpage, (pmap_t pmap), PMAP_INSERT_COMMPAGE_INDEX);
965
966
967 PMAP_SUPPORT_PROTOTYPES(
968 boolean_t,
969 pmap_is_empty, (pmap_t pmap,
970 vm_map_offset_t va_start,
971 vm_map_offset_t va_end), PMAP_IS_EMPTY_INDEX);
972
973
974 PMAP_SUPPORT_PROTOTYPES(
975 unsigned int,
976 pmap_map_cpu_windows_copy, (ppnum_t pn,
977 vm_prot_t prot,
978 unsigned int wimg_bits), PMAP_MAP_CPU_WINDOWS_COPY_INDEX);
979
980 PMAP_SUPPORT_PROTOTYPES(
981 void,
982 pmap_ro_zone_memcpy, (zone_id_t zid,
983 vm_offset_t va,
984 vm_offset_t offset,
985 const vm_offset_t new_data,
986 vm_size_t new_data_size), PMAP_RO_ZONE_MEMCPY_INDEX);
987
988 PMAP_SUPPORT_PROTOTYPES(
989 uint64_t,
990 pmap_ro_zone_atomic_op, (zone_id_t zid,
991 vm_offset_t va,
992 vm_offset_t offset,
993 zro_atomic_op_t op,
994 uint64_t value), PMAP_RO_ZONE_ATOMIC_OP_INDEX);
995
996 PMAP_SUPPORT_PROTOTYPES(
997 void,
998 pmap_ro_zone_bzero, (zone_id_t zid,
999 vm_offset_t va,
1000 vm_offset_t offset,
1001 vm_size_t size), PMAP_RO_ZONE_BZERO_INDEX);
1002
1003 PMAP_SUPPORT_PROTOTYPES(
1004 kern_return_t,
1005 pmap_nest, (pmap_t grand,
1006 pmap_t subord,
1007 addr64_t vstart,
1008 uint64_t size), PMAP_NEST_INDEX);
1009
1010 PMAP_SUPPORT_PROTOTYPES(
1011 void,
1012 pmap_page_protect_options, (ppnum_t ppnum,
1013 vm_prot_t prot,
1014 unsigned int options,
1015 void *arg), PMAP_PAGE_PROTECT_OPTIONS_INDEX);
1016
1017 PMAP_SUPPORT_PROTOTYPES(
1018 vm_map_address_t,
1019 pmap_protect_options, (pmap_t pmap,
1020 vm_map_address_t start,
1021 vm_map_address_t end,
1022 vm_prot_t prot,
1023 unsigned int options,
1024 void *args), PMAP_PROTECT_OPTIONS_INDEX);
1025
1026 PMAP_SUPPORT_PROTOTYPES(
1027 kern_return_t,
1028 pmap_query_page_info, (pmap_t pmap,
1029 vm_map_offset_t va,
1030 int *disp_p), PMAP_QUERY_PAGE_INFO_INDEX);
1031
1032 PMAP_SUPPORT_PROTOTYPES(
1033 mach_vm_size_t,
1034 pmap_query_resident, (pmap_t pmap,
1035 vm_map_address_t start,
1036 vm_map_address_t end,
1037 mach_vm_size_t * compressed_bytes_p), PMAP_QUERY_RESIDENT_INDEX);
1038
1039 PMAP_SUPPORT_PROTOTYPES(
1040 void,
1041 pmap_reference, (pmap_t pmap), PMAP_REFERENCE_INDEX);
1042
1043 PMAP_SUPPORT_PROTOTYPES(
1044 vm_map_address_t,
1045 pmap_remove_options, (pmap_t pmap,
1046 vm_map_address_t start,
1047 vm_map_address_t end,
1048 int options), PMAP_REMOVE_OPTIONS_INDEX);
1049
1050
1051 PMAP_SUPPORT_PROTOTYPES(
1052 void,
1053 pmap_set_cache_attributes, (ppnum_t pn,
1054 unsigned int cacheattr,
1055 bool update_attr_table), PMAP_SET_CACHE_ATTRIBUTES_INDEX);
1056
1057 PMAP_SUPPORT_PROTOTYPES(
1058 void,
1059 pmap_update_compressor_page, (ppnum_t pn,
1060 unsigned int prev_cacheattr, unsigned int new_cacheattr), PMAP_UPDATE_COMPRESSOR_PAGE_INDEX);
1061
1062 PMAP_SUPPORT_PROTOTYPES(
1063 void,
1064 pmap_set_nested, (pmap_t pmap), PMAP_SET_NESTED_INDEX);
1065
1066 #if MACH_ASSERT
1067 PMAP_SUPPORT_PROTOTYPES(
1068 void,
1069 pmap_set_process, (pmap_t pmap,
1070 int pid,
1071 char *procname), PMAP_SET_PROCESS_INDEX);
1072 #endif
1073
1074 PMAP_SUPPORT_PROTOTYPES(
1075 void,
1076 pmap_unmap_cpu_windows_copy, (unsigned int index), PMAP_UNMAP_CPU_WINDOWS_COPY_INDEX);
1077
1078 PMAP_SUPPORT_PROTOTYPES(
1079 void,
1080 pmap_unnest_options, (pmap_t grand,
1081 addr64_t vaddr,
1082 uint64_t size,
1083 unsigned int option), PMAP_UNNEST_OPTIONS_INDEX);
1084
1085 PMAP_SUPPORT_PROTOTYPES(
1086 void,
1087 phys_attribute_set, (ppnum_t pn,
1088 unsigned int bits), PHYS_ATTRIBUTE_SET_INDEX);
1089
1090 PMAP_SUPPORT_PROTOTYPES(
1091 void,
1092 phys_attribute_clear, (ppnum_t pn,
1093 unsigned int bits,
1094 int options,
1095 void *arg), PHYS_ATTRIBUTE_CLEAR_INDEX);
1096
1097 #if __ARM_RANGE_TLBI__
1098 PMAP_SUPPORT_PROTOTYPES(
1099 vm_map_address_t,
1100 phys_attribute_clear_range, (pmap_t pmap,
1101 vm_map_address_t start,
1102 vm_map_address_t end,
1103 unsigned int bits,
1104 unsigned int options), PHYS_ATTRIBUTE_CLEAR_RANGE_INDEX);
1105 #endif /* __ARM_RANGE_TLBI__ */
1106
1107
1108 PMAP_SUPPORT_PROTOTYPES(
1109 void,
1110 pmap_switch, (pmap_t pmap), PMAP_SWITCH_INDEX);
1111
1112 PMAP_SUPPORT_PROTOTYPES(
1113 void,
1114 pmap_clear_user_ttb, (void), PMAP_CLEAR_USER_TTB_INDEX);
1115
1116 PMAP_SUPPORT_PROTOTYPES(
1117 void,
1118 pmap_set_vm_map_cs_enforced, (pmap_t pmap, bool new_value), PMAP_SET_VM_MAP_CS_ENFORCED_INDEX);
1119
1120 PMAP_SUPPORT_PROTOTYPES(
1121 void,
1122 pmap_set_tpro, (pmap_t pmap), PMAP_SET_TPRO_INDEX);
1123
1124 PMAP_SUPPORT_PROTOTYPES(
1125 void,
1126 pmap_set_jit_entitled, (pmap_t pmap), PMAP_SET_JIT_ENTITLED_INDEX);
1127
1128 #if __has_feature(ptrauth_calls) && (defined(XNU_TARGET_OS_OSX) || (DEVELOPMENT || DEBUG))
1129 PMAP_SUPPORT_PROTOTYPES(
1130 void,
1131 pmap_disable_user_jop, (pmap_t pmap), PMAP_DISABLE_USER_JOP_INDEX);
1132 #endif /* __has_feature(ptrauth_calls) && (defined(XNU_TARGET_OS_OSX) || (DEVELOPMENT || DEBUG)) */
1133
1134 PMAP_SUPPORT_PROTOTYPES(
1135 void,
1136 pmap_trim, (pmap_t grand,
1137 pmap_t subord,
1138 addr64_t vstart,
1139 uint64_t size), PMAP_TRIM_INDEX);
1140
1141 #if HAS_APPLE_PAC
1142 PMAP_SUPPORT_PROTOTYPES(
1143 void *,
1144 pmap_sign_user_ptr, (void *value, ptrauth_key key, uint64_t discriminator, uint64_t jop_key), PMAP_SIGN_USER_PTR);
1145 PMAP_SUPPORT_PROTOTYPES(
1146 void *,
1147 pmap_auth_user_ptr, (void *value, ptrauth_key key, uint64_t discriminator, uint64_t jop_key), PMAP_AUTH_USER_PTR);
1148 #endif /* HAS_APPLE_PAC */
1149
1150
1151 void pmap_footprint_suspend(vm_map_t map,
1152 boolean_t suspend);
1153 PMAP_SUPPORT_PROTOTYPES(
1154 void,
1155 pmap_footprint_suspend, (vm_map_t map,
1156 boolean_t suspend),
1157 PMAP_FOOTPRINT_SUSPEND_INDEX);
1158
1159
1160
1161
1162
1163 /*
1164 * The low global vector page is mapped at a fixed alias.
1165 * Since the page size is 16k for H8 and newer we map the globals to a 16k
1166 * aligned address. Readers of the globals (e.g. lldb, panic server) need
1167 * to check both addresses anyway for backward compatibility. So for now
1168 * we leave H6 and H7 where they were.
1169 */
1170 #if (ARM_PGSHIFT == 14)
1171 #define LOWGLOBAL_ALIAS (LOW_GLOBAL_BASE_ADDRESS + 0x4000)
1172 #else
1173 #define LOWGLOBAL_ALIAS (LOW_GLOBAL_BASE_ADDRESS + 0x2000)
1174 #endif
1175
1176 static inline void
PMAP_ZINFO_PALLOC(pmap_t pmap,int bytes)1177 PMAP_ZINFO_PALLOC(
1178 pmap_t pmap, int bytes)
1179 {
1180 pmap_ledger_credit(pmap, task_ledgers.tkm_private, bytes);
1181 }
1182
1183 static inline void
PMAP_ZINFO_PFREE(pmap_t pmap,int bytes)1184 PMAP_ZINFO_PFREE(
1185 pmap_t pmap,
1186 int bytes)
1187 {
1188 pmap_ledger_debit(pmap, task_ledgers.tkm_private, bytes);
1189 }
1190
1191 void
pmap_tt_ledger_credit(pmap_t pmap,vm_size_t size)1192 pmap_tt_ledger_credit(
1193 pmap_t pmap,
1194 vm_size_t size)
1195 {
1196 if (pmap != kernel_pmap) {
1197 pmap_ledger_credit(pmap, task_ledgers.phys_footprint, size);
1198 pmap_ledger_credit(pmap, task_ledgers.page_table, size);
1199 }
1200 }
1201
1202 void
pmap_tt_ledger_debit(pmap_t pmap,vm_size_t size)1203 pmap_tt_ledger_debit(
1204 pmap_t pmap,
1205 vm_size_t size)
1206 {
1207 if (pmap != kernel_pmap) {
1208 pmap_ledger_debit(pmap, task_ledgers.phys_footprint, size);
1209 pmap_ledger_debit(pmap, task_ledgers.page_table, size);
1210 }
1211 }
1212
1213 static inline void
pmap_update_plru(uint16_t asid_index __unused)1214 pmap_update_plru(uint16_t asid_index __unused)
1215 {
1216 #if !HAS_16BIT_ASID
1217 if (__probable(pmap_asid_plru)) {
1218 unsigned plru_index = asid_index >> 6;
1219 if (__improbable(os_atomic_andnot(&asid_plru_bitmap[plru_index], (1ULL << (asid_index & 63)), relaxed) == 0)) {
1220 asid_plru_generation[plru_index] = ++asid_plru_gencount;
1221 asid_plru_bitmap[plru_index] = ((plru_index == 0) ? ~1ULL : UINT64_MAX);
1222 }
1223 }
1224 #endif /* !HAS_16BIT_ASID */
1225 }
1226
1227 static bool
alloc_asid(pmap_t pmap)1228 alloc_asid(pmap_t pmap)
1229 {
1230 int vasid = -1;
1231
1232 pmap_simple_lock(&asid_lock);
1233
1234 #if !HAS_16BIT_ASID
1235 if (__probable(pmap_asid_plru)) {
1236 unsigned plru_index = 0;
1237 uint64_t lowest_gen = asid_plru_generation[0];
1238 uint64_t lowest_gen_bitmap = asid_plru_bitmap[0];
1239 for (unsigned i = 1; i < (sizeof(asid_plru_generation) / sizeof(asid_plru_generation[0])); ++i) {
1240 if (asid_plru_generation[i] < lowest_gen) {
1241 plru_index = i;
1242 lowest_gen = asid_plru_generation[i];
1243 lowest_gen_bitmap = asid_plru_bitmap[i];
1244 }
1245 }
1246
1247 for (; plru_index < BITMAP_LEN(pmap_max_asids); plru_index += (MAX_HW_ASIDS >> 6)) {
1248 uint64_t temp_plru = lowest_gen_bitmap & asid_bitmap[plru_index];
1249 if (temp_plru) {
1250 vasid = (plru_index << 6) + lsb_first(temp_plru);
1251 #if DEVELOPMENT || DEBUG
1252 ++pmap_asid_hits;
1253 #endif
1254 break;
1255 }
1256 }
1257 }
1258 #else
1259 /**
1260 * For 16-bit ASID targets, we assume a 1:1 correspondence between ASIDs and active tasks and
1261 * therefore allocate directly from the ASID bitmap instead of using the pLRU allocator.
1262 * However, we first try to allocate starting from the position of the most-recently allocated
1263 * ASID. This is done both as an allocator performance optimization (as it avoids crowding the
1264 * lower bit positions and then re-checking those same lower positions every time we allocate
1265 * an ASID) as well as a security mitigation to increase the temporal distance between ASID
1266 * reuse. This increases the difficulty of leveraging ASID reuse to train branch predictor
1267 * logic, without requiring prohibitively expensive RCTX instructions.
1268 */
1269 vasid = bitmap_lsb_next(&asid_bitmap[0], pmap_max_asids, last_allocated_asid);
1270 #endif /* !HAS_16BIT_ASID */
1271 if (__improbable(vasid < 0)) {
1272 // bitmap_first() returns highest-order bits first, but a 0-based scheme works
1273 // slightly better with the collision detection scheme used by pmap_switch_internal().
1274 vasid = bitmap_lsb_first(&asid_bitmap[0], pmap_max_asids);
1275 #if DEVELOPMENT || DEBUG
1276 ++pmap_asid_misses;
1277 #endif
1278 }
1279 if (__improbable(vasid < 0)) {
1280 pmap_simple_unlock(&asid_lock);
1281 return false;
1282 }
1283 assert((uint32_t)vasid < pmap_max_asids);
1284 assert(bitmap_test(&asid_bitmap[0], (unsigned int)vasid));
1285 bitmap_clear(&asid_bitmap[0], (unsigned int)vasid);
1286 const uint16_t hw_asid = (uint16_t)(vasid & (MAX_HW_ASIDS - 1));
1287 #if HAS_16BIT_ASID
1288 last_allocated_asid = hw_asid;
1289 #endif /* HAS_16BIT_ASID */
1290 pmap_simple_unlock(&asid_lock);
1291 assert(hw_asid != 0); // Should never alias kernel ASID
1292 pmap->asid = (uint16_t)vasid;
1293 pmap_update_plru(hw_asid);
1294 return true;
1295 }
1296
1297 static void
free_asid(pmap_t pmap)1298 free_asid(pmap_t pmap)
1299 {
1300 const uint16_t vasid = os_atomic_xchg(&pmap->asid, 0, relaxed);
1301 if (__improbable(vasid == 0)) {
1302 return;
1303 }
1304
1305 #if !HAS_16BIT_ASID
1306 if (pmap_asid_plru) {
1307 const uint16_t hw_asid = vasid & (MAX_HW_ASIDS - 1);
1308 os_atomic_or(&asid_plru_bitmap[hw_asid >> 6], (1ULL << (hw_asid & 63)), relaxed);
1309 }
1310 #endif /* !HAS_16BIT_ASID */
1311 pmap_simple_lock(&asid_lock);
1312 assert(!bitmap_test(&asid_bitmap[0], vasid));
1313 bitmap_set(&asid_bitmap[0], vasid);
1314 pmap_simple_unlock(&asid_lock);
1315 }
1316
1317
1318 boolean_t
pmap_valid_address(pmap_paddr_t addr)1319 pmap_valid_address(
1320 pmap_paddr_t addr)
1321 {
1322 return pa_valid(addr);
1323 }
1324
1325
1326
1327
1328
1329
1330 /*
1331 * Map memory at initialization. The physical addresses being
1332 * mapped are not managed and are never unmapped.
1333 *
1334 * For now, VM is already on, we only need to map the
1335 * specified memory.
1336 */
1337 vm_map_address_t
pmap_map(vm_map_address_t virt,vm_offset_t start,vm_offset_t end,vm_prot_t prot,unsigned int flags)1338 pmap_map(
1339 vm_map_address_t virt,
1340 vm_offset_t start,
1341 vm_offset_t end,
1342 vm_prot_t prot,
1343 unsigned int flags)
1344 {
1345 kern_return_t kr;
1346 vm_size_t ps;
1347
1348 ps = PAGE_SIZE;
1349 while (start < end) {
1350 kr = pmap_enter(kernel_pmap, virt, (ppnum_t)atop(start),
1351 prot, VM_PROT_NONE, flags, FALSE, PMAP_MAPPING_TYPE_INFER);
1352
1353 if (kr != KERN_SUCCESS) {
1354 panic("%s: failed pmap_enter, "
1355 "virt=%p, start_addr=%p, end_addr=%p, prot=%#x, flags=%#x",
1356 __FUNCTION__,
1357 (void *) virt, (void *) start, (void *) end, prot, flags);
1358 }
1359
1360 virt += ps;
1361 start += ps;
1362 }
1363 return virt;
1364 }
1365
1366 /**
1367 * Force the permission of a PTE to be kernel RO if a page has XNU_PROTECTED_IO type.
1368 *
1369 * @param paddr The physical address of the page.
1370 * @param tmplate The PTE value to be evaluated.
1371 *
1372 * @return A new PTE value with permission bits modified.
1373 */
1374 static inline
1375 pt_entry_t
pmap_force_pte_kernel_ro_if_protected_io(pmap_paddr_t paddr,pt_entry_t tmplate)1376 pmap_force_pte_kernel_ro_if_protected_io(pmap_paddr_t paddr, pt_entry_t tmplate)
1377 {
1378 /**
1379 * When requesting RW mappings to an XNU_PROTECTED_IO frame, downgrade
1380 * the mapping to RO. This is required because IOKit relies on this
1381 * behavior currently in the PPL.
1382 */
1383 const sptm_frame_type_t frame_type = sptm_get_frame_type(paddr);
1384 if (frame_type == XNU_PROTECTED_IO) {
1385 /* SPTM to own the page by converting KERN_RW to PPL_RW. */
1386 const uint64_t xprr_perm = pte_to_xprr_perm(tmplate);
1387 switch (xprr_perm) {
1388 case XPRR_KERN_RO_PERM:
1389 break;
1390 case XPRR_KERN_RW_PERM:
1391 tmplate &= ~ARM_PTE_XPRR_MASK;
1392 tmplate |= xprr_perm_to_pte(XPRR_KERN_RO_PERM);
1393 break;
1394 default:
1395 panic("%s: Unsupported xPRR perm %llu for pte 0x%llx", __func__, xprr_perm, (uint64_t)tmplate);
1396 }
1397 }
1398
1399 return tmplate;
1400 }
1401
1402 vm_map_address_t
pmap_map_bd_with_options(vm_map_address_t virt,vm_offset_t start,vm_offset_t end,vm_prot_t prot,int32_t options)1403 pmap_map_bd_with_options(
1404 vm_map_address_t virt,
1405 vm_offset_t start,
1406 vm_offset_t end,
1407 vm_prot_t prot,
1408 int32_t options)
1409 {
1410 pt_entry_t tmplate;
1411 vm_map_address_t vaddr;
1412 vm_offset_t paddr;
1413 pt_entry_t mem_attr;
1414
1415 switch (options & PMAP_MAP_BD_MASK) {
1416 case PMAP_MAP_BD_WCOMB:
1417 mem_attr = ARM_PTE_ATTRINDX(CACHE_ATTRINDX_WRITECOMB);
1418 mem_attr |= ARM_PTE_SH(SH_OUTER_MEMORY);
1419 break;
1420 case PMAP_MAP_BD_POSTED:
1421 mem_attr = ARM_PTE_ATTRINDX(CACHE_ATTRINDX_POSTED);
1422 break;
1423 case PMAP_MAP_BD_POSTED_REORDERED:
1424 mem_attr = ARM_PTE_ATTRINDX(CACHE_ATTRINDX_POSTED_REORDERED);
1425 break;
1426 case PMAP_MAP_BD_POSTED_COMBINED_REORDERED:
1427 mem_attr = ARM_PTE_ATTRINDX(CACHE_ATTRINDX_POSTED_COMBINED_REORDERED);
1428 break;
1429 default:
1430 mem_attr = ARM_PTE_ATTRINDX(CACHE_ATTRINDX_DISABLE);
1431 break;
1432 }
1433
1434 tmplate = ARM_PTE_AP((prot & VM_PROT_WRITE) ? AP_RWNA : AP_RONA) |
1435 mem_attr | ARM_PTE_TYPE | ARM_PTE_NX | ARM_PTE_PNX | ARM_PTE_AF;
1436
1437 #if __ARM_KERNEL_PROTECT__
1438 tmplate |= ARM_PTE_NG;
1439 #endif /* __ARM_KERNEL_PROTECT__ */
1440
1441 vaddr = virt;
1442 paddr = start;
1443 while (paddr < end) {
1444 __assert_only sptm_return_t ret = sptm_map_page(kernel_pmap->ttep, vaddr, pmap_force_pte_kernel_ro_if_protected_io(paddr, tmplate) | pa_to_pte(paddr));
1445 assert((ret == SPTM_SUCCESS) || (ret == SPTM_MAP_VALID));
1446
1447 vaddr += PAGE_SIZE;
1448 paddr += PAGE_SIZE;
1449 }
1450
1451 return vaddr;
1452 }
1453
1454 /*
1455 * Back-door routine for mapping kernel VM at initialization.
1456 * Useful for mapping memory outside the range
1457 * [vm_first_phys, vm_last_phys] (i.e., devices).
1458 * Otherwise like pmap_map.
1459 */
1460 vm_map_address_t
pmap_map_bd(vm_map_address_t virt,vm_offset_t start,vm_offset_t end,vm_prot_t prot)1461 pmap_map_bd(
1462 vm_map_address_t virt,
1463 vm_offset_t start,
1464 vm_offset_t end,
1465 vm_prot_t prot)
1466 {
1467 return pmap_map_bd_with_options(virt, start, end, prot, 0);
1468 }
1469
1470 /*
1471 * Back-door routine for mapping kernel VM at initialization.
1472 * Useful for mapping memory specific physical addresses in early
1473 * boot (i.e., before kernel_map is initialized).
1474 *
1475 * Maps are in the VM_HIGH_KERNEL_WINDOW area.
1476 */
1477
1478 vm_map_address_t
pmap_map_high_window_bd(vm_offset_t pa_start,vm_size_t len,vm_prot_t prot)1479 pmap_map_high_window_bd(
1480 vm_offset_t pa_start,
1481 vm_size_t len,
1482 vm_prot_t prot)
1483 {
1484 pt_entry_t *ptep, pte;
1485 vm_map_address_t va_start = VREGION1_START;
1486 vm_map_address_t va_max = VREGION1_START + VREGION1_SIZE;
1487 vm_map_address_t va_end;
1488 vm_map_address_t va;
1489 vm_size_t offset;
1490
1491 offset = pa_start & PAGE_MASK;
1492 pa_start -= offset;
1493 len += offset;
1494
1495 if (len > (va_max - va_start)) {
1496 panic("%s: area too large, "
1497 "pa_start=%p, len=%p, prot=0x%x",
1498 __FUNCTION__,
1499 (void*)pa_start, (void*)len, prot);
1500 }
1501
1502 scan:
1503 for (; va_start < va_max; va_start += PAGE_SIZE) {
1504 ptep = pmap_pte(kernel_pmap, va_start);
1505 assert(!pte_is_compressed(*ptep, ptep));
1506 if (*ptep == ARM_PTE_TYPE_FAULT) {
1507 break;
1508 }
1509 }
1510 if (va_start > va_max) {
1511 panic("%s: insufficient pages, "
1512 "pa_start=%p, len=%p, prot=0x%x",
1513 __FUNCTION__,
1514 (void*)pa_start, (void*)len, prot);
1515 }
1516
1517 for (va_end = va_start + PAGE_SIZE; va_end < va_start + len; va_end += PAGE_SIZE) {
1518 ptep = pmap_pte(kernel_pmap, va_end);
1519 assert(!pte_is_compressed(*ptep, ptep));
1520 if (*ptep != ARM_PTE_TYPE_FAULT) {
1521 va_start = va_end + PAGE_SIZE;
1522 goto scan;
1523 }
1524 }
1525
1526 for (va = va_start; va < va_end; va += PAGE_SIZE, pa_start += PAGE_SIZE) {
1527 ptep = pmap_pte(kernel_pmap, va);
1528 pte = pa_to_pte(pa_start)
1529 | ARM_PTE_TYPE | ARM_PTE_AF | ARM_PTE_NX | ARM_PTE_PNX
1530 | ARM_PTE_AP((prot & VM_PROT_WRITE) ? AP_RWNA : AP_RONA)
1531 | ARM_PTE_ATTRINDX(CACHE_ATTRINDX_DEFAULT)
1532 | ARM_PTE_SH(SH_OUTER_MEMORY);
1533 #if __ARM_KERNEL_PROTECT__
1534 pte |= ARM_PTE_NG;
1535 #endif /* __ARM_KERNEL_PROTECT__ */
1536 __assert_only sptm_return_t ret = sptm_map_page(kernel_pmap->ttep, va, pte);
1537 assert((ret == SPTM_SUCCESS) || (ret == SPTM_MAP_VALID));
1538 }
1539 #if KASAN
1540 kasan_notify_address(va_start, len);
1541 #endif
1542 return va_start;
1543 }
1544
1545 /*
1546 * pmap_get_arm64_prot
1547 *
1548 * return effective armv8 VMSA block protections including
1549 * table AP/PXN/XN overrides of a pmap entry
1550 *
1551 */
1552
1553 uint64_t
pmap_get_arm64_prot(pmap_t pmap,vm_offset_t addr)1554 pmap_get_arm64_prot(
1555 pmap_t pmap,
1556 vm_offset_t addr)
1557 {
1558 tt_entry_t tte = 0;
1559 unsigned int level = 0;
1560 uint64_t tte_type = 0;
1561 uint64_t effective_prot_bits = 0;
1562 uint64_t aggregate_tte = 0;
1563 uint64_t table_ap_bits = 0, table_xn = 0, table_pxn = 0;
1564 const pt_attr_t * const pt_attr = pmap_get_pt_attr(pmap);
1565
1566 for (level = pt_attr->pta_root_level; level <= pt_attr->pta_max_level; level++) {
1567 tte = *pmap_ttne(pmap, level, addr);
1568
1569 if (!(tte & ARM_TTE_VALID)) {
1570 return 0;
1571 }
1572
1573 tte_type = tte & ARM_TTE_TYPE_MASK;
1574
1575 if ((tte_type == ARM_TTE_TYPE_BLOCK) ||
1576 (level == pt_attr->pta_max_level)) {
1577 /* Block or page mapping; both have the same protection bit layout. */
1578 break;
1579 } else if (tte_type == ARM_TTE_TYPE_TABLE) {
1580 /* All of the table bits we care about are overrides, so just OR them together. */
1581 aggregate_tte |= tte;
1582 }
1583 }
1584
1585 table_ap_bits = ((aggregate_tte >> ARM_TTE_TABLE_APSHIFT) & AP_MASK);
1586 table_xn = (aggregate_tte & ARM_TTE_TABLE_XN);
1587 table_pxn = (aggregate_tte & ARM_TTE_TABLE_PXN);
1588
1589 /* Start with the PTE bits. */
1590 effective_prot_bits = tte & (ARM_PTE_APMASK | ARM_PTE_NX | ARM_PTE_PNX);
1591
1592 /* Table AP bits mask out block/page AP bits */
1593 effective_prot_bits &= ~(ARM_PTE_AP(table_ap_bits));
1594
1595 /* XN/PXN bits can be OR'd in. */
1596 effective_prot_bits |= (table_xn ? ARM_PTE_NX : 0);
1597 effective_prot_bits |= (table_pxn ? ARM_PTE_PNX : 0);
1598
1599 return effective_prot_bits;
1600 }
1601
1602 /*
1603 * Bootstrap the system enough to run with virtual memory.
1604 *
1605 * The early VM initialization code has already allocated
1606 * the first CPU's translation table and made entries for
1607 * all the one-to-one mappings to be found there.
1608 *
1609 * We must set up the kernel pmap structures, the
1610 * physical-to-virtual translation lookup tables for the
1611 * physical memory to be managed (between avail_start and
1612 * avail_end).
1613 *
1614 * Map the kernel's code and data, and allocate the system page table.
1615 * Page_size must already be set.
1616 *
1617 * Parameters:
1618 * first_avail first available physical page -
1619 * after kernel page tables
1620 * avail_start PA of first managed physical page
1621 * avail_end PA of last managed physical page
1622 */
1623
1624 void
pmap_bootstrap(vm_offset_t vstart)1625 pmap_bootstrap(
1626 vm_offset_t vstart)
1627 {
1628 vm_map_offset_t maxoffset;
1629
1630 lck_grp_init(&pmap_lck_grp, "pmap", LCK_GRP_ATTR_NULL);
1631
1632 #if DEVELOPMENT || DEBUG
1633 if (PE_parse_boot_argn("pmap_trace", &pmap_trace_mask, sizeof(pmap_trace_mask))) {
1634 kprintf("Kernel traces for pmap operations enabled\n");
1635 }
1636 #endif
1637
1638 /*
1639 * Initialize the kernel pmap.
1640 */
1641 #if ARM_PARAMETERIZED_PMAP
1642 kernel_pmap->pmap_pt_attr = native_pt_attr;
1643 #endif /* ARM_PARAMETERIZED_PMAP */
1644 #if HAS_APPLE_PAC
1645 kernel_pmap->disable_jop = 0;
1646 #endif /* HAS_APPLE_PAC */
1647 kernel_pmap->tte = cpu_tte;
1648 kernel_pmap->ttep = cpu_ttep;
1649 kernel_pmap->min = UINT64_MAX - (1ULL << (64 - T1SZ_BOOT)) + 1;
1650 kernel_pmap->max = UINTPTR_MAX;
1651 os_ref_init_count_raw(&kernel_pmap->ref_count, &pmap_refgrp, 1);
1652 kernel_pmap->nx_enabled = TRUE;
1653 kernel_pmap->is_64bit = TRUE;
1654 #if CONFIG_ROSETTA
1655 kernel_pmap->is_rosetta = FALSE;
1656 #endif
1657
1658 #if ARM_PARAMETERIZED_PMAP
1659 kernel_pmap->pmap_pt_attr = native_pt_attr;
1660 #endif /* ARM_PARAMETERIZED_PMAP */
1661
1662 kernel_pmap->nested_region_addr = 0x0ULL;
1663 kernel_pmap->nested_region_size = 0x0ULL;
1664 kernel_pmap->nested_region_unnested_table_bitmap = NULL;
1665 kernel_pmap->type = PMAP_TYPE_KERNEL;
1666
1667 kernel_pmap->asid = 0;
1668
1669 pmap_lock_init(kernel_pmap);
1670
1671 pmap_max_asids = SPTMArgs->num_asids;
1672
1673 const vm_size_t asid_table_size = sizeof(*asid_bitmap) * BITMAP_LEN(pmap_max_asids);
1674
1675 /**
1676 * Bootstrap the core pmap data structures (e.g., pv_head_table,
1677 * pp_attr_table, etc). This function will use `avail_start` to allocate
1678 * space for these data structures.
1679 * */
1680 pmap_data_bootstrap();
1681
1682 /**
1683 * Bootstrap any necessary UAT data structures and values needed from the device tree.
1684 */
1685 uat_bootstrap();
1686
1687 /**
1688 * Don't make any assumptions about the alignment of avail_start before this
1689 * point (i.e., pmap_data_bootstrap() performs allocations).
1690 */
1691 avail_start = PMAP_ALIGN(avail_start, __alignof(bitmap_t));
1692
1693 const pmap_paddr_t pmap_struct_start = avail_start;
1694
1695 asid_bitmap = (bitmap_t*)phystokv(avail_start);
1696 avail_start = round_page(avail_start + asid_table_size);
1697
1698 memset((char *)phystokv(pmap_struct_start), 0, avail_start - pmap_struct_start);
1699
1700 queue_init(&map_pmap_list);
1701 queue_enter(&map_pmap_list, kernel_pmap, pmap_t, pmaps);
1702
1703 virtual_space_start = vstart;
1704 virtual_space_end = VM_MAX_KERNEL_ADDRESS;
1705
1706 bitmap_full(&asid_bitmap[0], pmap_max_asids);
1707 // Clear the ASIDs which will alias the reserved kernel ASID of 0
1708 for (unsigned int i = 0; i < pmap_max_asids; i += MAX_HW_ASIDS) {
1709 bitmap_clear(&asid_bitmap[0], i);
1710 }
1711
1712 #if !HAS_16BIT_ASID
1713 /**
1714 * Align the range of available hardware ASIDs to a multiple of 64 to enable the
1715 * masking used by the PLRU scheme. This means we must handle the case in which
1716 * the returned hardware ASID is 0, which we do by clearing all vASIDs that will
1717 * alias the kernel ASID.
1718 */
1719 pmap_max_asids = pmap_max_asids & ~63ul;
1720 if (__improbable(pmap_max_asids == 0)) {
1721 panic("%s: insufficient number of ASIDs (%u) supplied by SPTM", __func__, (unsigned int)pmap_max_asids);
1722 }
1723 pmap_asid_plru = (pmap_max_asids > MAX_HW_ASIDS);
1724 PE_parse_boot_argn("pmap_asid_plru", &pmap_asid_plru, sizeof(pmap_asid_plru));
1725 _Static_assert(sizeof(asid_plru_bitmap[0] == sizeof(uint64_t)), "bitmap_t is not a 64-bit integer");
1726 _Static_assert((MAX_HW_ASIDS % 64) == 0, "MAX_HW_ASIDS is not divisible by 64");
1727 bitmap_full(&asid_plru_bitmap[0], MAX_HW_ASIDS);
1728 bitmap_clear(&asid_plru_bitmap[0], 0);
1729 #endif /* !HAS_16BIT_ASID */
1730
1731
1732 if (PE_parse_boot_argn("arm_maxoffset", &maxoffset, sizeof(maxoffset))) {
1733 maxoffset = trunc_page(maxoffset);
1734 if ((maxoffset >= pmap_max_offset(FALSE, ARM_PMAP_MAX_OFFSET_MIN))
1735 && (maxoffset <= pmap_max_offset(FALSE, ARM_PMAP_MAX_OFFSET_MAX))) {
1736 arm_pmap_max_offset_default = maxoffset;
1737 }
1738 }
1739 if (PE_parse_boot_argn("arm64_maxoffset", &maxoffset, sizeof(maxoffset))) {
1740 maxoffset = trunc_page(maxoffset);
1741 if ((maxoffset >= pmap_max_offset(TRUE, ARM_PMAP_MAX_OFFSET_MIN))
1742 && (maxoffset <= pmap_max_offset(TRUE, ARM_PMAP_MAX_OFFSET_MAX))) {
1743 arm64_pmap_max_offset_default = maxoffset;
1744 }
1745 }
1746
1747 PE_parse_boot_argn("pmap_panic_dev_wimg_on_managed", &pmap_panic_dev_wimg_on_managed, sizeof(pmap_panic_dev_wimg_on_managed));
1748
1749
1750 #if DEVELOPMENT || DEBUG
1751 PE_parse_boot_argn("vm_footprint_suspend_allowed",
1752 &vm_footprint_suspend_allowed,
1753 sizeof(vm_footprint_suspend_allowed));
1754 #endif /* DEVELOPMENT || DEBUG */
1755
1756 #if KASAN
1757 /* Shadow the CPU copy windows, as they fall outside of the physical aperture */
1758 kasan_map_shadow(CPUWINDOWS_BASE, CPUWINDOWS_TOP - CPUWINDOWS_BASE, true);
1759 #endif /* KASAN */
1760
1761 /**
1762 * Ensure that avail_start is always left on a page boundary. The calling
1763 * code might not perform any alignment before allocating page tables so
1764 * this is important.
1765 */
1766 avail_start = round_page(avail_start);
1767
1768
1769 #if (DEVELOPMENT || DEBUG)
1770 sptm_features_available(SPTM_FEATURE_SYSREG, &sptm_sysreg_available);
1771 #endif /* (DEVELOPMENT || DEBUG) */
1772
1773 /* Signal that the pmap has been bootstrapped */
1774 pmap_bootstrapped = true;
1775 }
1776
1777 /**
1778 * Helper for creating a populated commpage table
1779 *
1780 * In order to avoid burning extra pages on mapping the commpage, we create a
1781 * dedicated table hierarchy for the commpage. We forcibly nest the translation tables from
1782 * this pmap into other pmaps. The level we will nest at depends on the MMU configuration (page
1783 * size, TTBR range, etc). Typically, this is at L1 for 4K tasks and L2 for 16K tasks.
1784 *
1785 * @note that this is NOT "the nested pmap" (which is used to nest the shared cache).
1786 *
1787 * @param rw_va Virtual address at which to insert a mapping to the kernel R/W commpage
1788 * @param ro_va Virtual address at which to insert a mapping to the kernel R/O commpage
1789 * @param rw_pa Physical address of kernel R/W commpage
1790 * @param ro_pa Physical address of kernel R/O commpage, may be 0 if not supported in this
1791 * configuration
1792 * @param rx_pa Physical address of user executable (and kernel R/O) commpage, may be 0 if
1793 * not supported in this configuration
1794 * @param pmap_create_flags Control flags for the temporary pmap created by this function
1795 *
1796 * @return the physical address of the created commpage table, typed as
1797 * XNU_PAGE_TABLE_COMMPAGE and containing all relevant commpage mappings.
1798 */
1799 static pmap_paddr_t
pmap_create_commpage_table(vm_map_address_t rw_va,vm_map_address_t ro_va,pmap_paddr_t rw_pa,pmap_paddr_t ro_pa,pmap_paddr_t rx_pa,unsigned int pmap_create_flags)1800 pmap_create_commpage_table(vm_map_address_t rw_va, vm_map_address_t ro_va,
1801 pmap_paddr_t rw_pa, pmap_paddr_t ro_pa, pmap_paddr_t rx_pa, unsigned int pmap_create_flags)
1802 {
1803 pmap_t temp_commpage_pmap = pmap_create_options(NULL, 0, pmap_create_flags);
1804 assert(temp_commpage_pmap != NULL);
1805 assert(rw_pa != 0);
1806 const pt_attr_t *pt_attr = pmap_get_pt_attr(temp_commpage_pmap);
1807
1808 /*
1809 * We only use pmap_expand to expand the pmap up to the commpage nesting level. At that level
1810 * and beyond, all the newly created tables will be nested directly into the userspace region
1811 * for each process, and as such they must be of the dedicated SPTM commpage table type so that
1812 * the SPTM can enforce the commpage security model which forbids random replacement of commpage
1813 * mappings.
1814 */
1815 kern_return_t kr = pmap_expand(temp_commpage_pmap, rw_va, 0, pt_attr_commpage_level(pt_attr));
1816 assert(kr == KERN_SUCCESS);
1817
1818 pmap_paddr_t commpage_table_pa = 0;
1819 for (unsigned int i = pt_attr_commpage_level(pt_attr); i < pt_attr_leaf_level(pt_attr); i++) {
1820 pmap_paddr_t new_table = 0;
1821 kr = pmap_page_alloc(&new_table, 0);
1822 assert((kr == KERN_SUCCESS) && (new_table != 0));
1823 if (commpage_table_pa == 0) {
1824 commpage_table_pa = new_table;
1825 }
1826
1827 sptm_retype_params_t retype_params = {.raw = SPTM_RETYPE_PARAMS_NULL};
1828 retype_params.level = (sptm_pt_level_t)pt_attr_leaf_level(pt_attr);
1829 sptm_retype(new_table, XNU_DEFAULT, XNU_PAGE_TABLE_COMMPAGE, retype_params);
1830
1831 const sptm_tte_t table_tte = (new_table & ARM_TTE_TABLE_MASK) | ARM_TTE_TYPE_TABLE | ARM_TTE_VALID;
1832
1833 sptm_map_table(temp_commpage_pmap->ttep, pt_attr_align_va(pt_attr, i, rw_va),
1834 (sptm_pt_level_t)i, table_tte);
1835 }
1836
1837 /*
1838 * Note the lack of ARM_PTE_NG here: commpage mappings are at fixed addresses and
1839 * frequently accessed, so we map them global to avoid unnecessary TLB pressure.
1840 */
1841 static const sptm_pte_t commpage_pte_template = ARM_PTE_TYPE_VALID
1842 | ARM_PTE_ATTRINDX(CACHE_ATTRINDX_WRITEBACK)
1843 | ARM_PTE_SH(SH_INNER_MEMORY) | ARM_PTE_PNX
1844 | ARM_PTE_AP(AP_RORO) | ARM_PTE_AF;
1845
1846 sptm_return_t sptm_ret = sptm_map_page(temp_commpage_pmap->ttep, rw_va,
1847 commpage_pte_template | ARM_PTE_NX | pa_to_pte(rw_pa));
1848 assert(sptm_ret == SPTM_SUCCESS);
1849
1850 if (ro_pa != 0) {
1851 assert((ro_va & ~pt_attr_twig_offmask(pt_attr)) == (rw_va & ~pt_attr_twig_offmask(pt_attr)));
1852 sptm_ret = sptm_map_page(temp_commpage_pmap->ttep, ro_va,
1853 commpage_pte_template | ARM_PTE_NX | pa_to_pte(ro_pa));
1854 assert(sptm_ret == SPTM_SUCCESS);
1855 }
1856
1857 if (rx_pa != 0) {
1858 assert((commpage_text_user_va & ~pt_attr_twig_offmask(pt_attr)) == (rw_va & ~pt_attr_twig_offmask(pt_attr)));
1859 assert((commpage_text_user_va != rw_va) && (commpage_text_user_va != ro_va));
1860 sptm_ret = sptm_map_page(temp_commpage_pmap->ttep, commpage_text_user_va, commpage_pte_template | pa_to_pte(rx_pa));
1861 assert(sptm_ret == SPTM_SUCCESS);
1862 }
1863
1864 sptm_unmap_table(temp_commpage_pmap->ttep, pt_attr_align_va(pt_attr, pt_attr_commpage_level(pt_attr), rw_va),
1865 (sptm_pt_level_t)pt_attr_commpage_level(pt_attr));
1866 pmap_destroy(temp_commpage_pmap);
1867
1868 return commpage_table_pa;
1869 }
1870
1871 /**
1872 * Helper for creating all commpage tables applicable to the current configuration.
1873 *
1874 * @note This function is intended to be called during bootstrap.
1875 * @note This function assumes that pmap_create_commpages has already executed, and therefore
1876 * the commpage_*_pa variables have been assigned to their final values. commpage_data_pa
1877 * is the kernel RW commpage and is assumed to be present on all configurations, so it
1878 * therefore must be non-zero at this point. The other variables are considered optional
1879 * depending upon configuration and may be zero.
1880 */
1881 void pmap_prepare_commpages(void);
1882 void
pmap_prepare_commpages(void)1883 pmap_prepare_commpages(void)
1884 {
1885 sptm_retype_params_t retype_params = {.raw = SPTM_RETYPE_PARAMS_NULL};
1886 assert(commpage_data_pa != 0);
1887 sptm_retype(commpage_data_pa, XNU_DEFAULT, XNU_COMMPAGE_RW, retype_params);
1888 if (commpage_ro_data_pa != 0) {
1889 sptm_retype(commpage_ro_data_pa, XNU_DEFAULT, XNU_COMMPAGE_RO, retype_params);
1890 }
1891 if (commpage_text_pa != 0) {
1892 sptm_retype(commpage_text_pa, XNU_DEFAULT, XNU_COMMPAGE_RX, retype_params);
1893 }
1894
1895 /*
1896 * User mapping of comm page text section for 64 bit mapping only
1897 *
1898 * We don't insert the text commpage into the 32 bit mapping because we don't want
1899 * 32-bit user processes to get this page mapped in, they should never call into
1900 * this page.
1901 */
1902 commpage_default_table = pmap_create_commpage_table(_COMM_PAGE64_BASE_ADDRESS, _COMM_PAGE64_RO_ADDRESS,
1903 commpage_data_pa, commpage_ro_data_pa, commpage_text_pa, 0);
1904
1905 /*
1906 * SPTM TODO: Enable this, along with the appropriate 32-bit commpage address checks and flushes in the
1907 * SPTM, if we ever need to support arm64_32 processes in the SPTM.
1908 *
1909 * commpage32_default_table = pmap_create_commpage_table(_COMM_PAGE32_BASE_ADDRESS, _COMM_PAGE32_RO_ADDRESS,
1910 * commpage_data_pa, commpage_ro_data_pa, 0, 0);
1911 */
1912 #if __ARM_MIXED_PAGE_SIZE__
1913 commpage_4k_table = pmap_create_commpage_table(_COMM_PAGE64_BASE_ADDRESS, _COMM_PAGE64_RO_ADDRESS,
1914 commpage_data_pa, commpage_ro_data_pa, 0, PMAP_CREATE_FORCE_4K_PAGES);
1915
1916 /*
1917 * SPTM TODO: Enable this, along with the appropriate 32-bit commpage address checks and flushes in the
1918 * SPTM, if we ever need to support arm64_32 processes in the SPTM.
1919 * commpage32_4k_table = pmap_create_commpage_table(_COMM_PAGE32_BASE_ADDRESS, _COMM_PAGE32_RO_ADDRESS,
1920 * commpage_data_pa, commpage_ro_data_pa, 0, PMAP_CREATE_FORCE_4K_PAGES);
1921 */
1922 #endif /* __ARM_MIXED_PAGE_SIZE__ */
1923
1924 }
1925
1926 void
pmap_virtual_space(vm_offset_t * startp,vm_offset_t * endp)1927 pmap_virtual_space(
1928 vm_offset_t *startp,
1929 vm_offset_t *endp
1930 )
1931 {
1932 *startp = virtual_space_start;
1933 *endp = virtual_space_end;
1934 }
1935
1936
1937 boolean_t
pmap_virtual_region(unsigned int region_select,vm_map_offset_t * startp,vm_map_size_t * size)1938 pmap_virtual_region(
1939 unsigned int region_select,
1940 vm_map_offset_t *startp,
1941 vm_map_size_t *size
1942 )
1943 {
1944 boolean_t ret = FALSE;
1945 #if defined(KERNEL_INTEGRITY_KTRR) || defined(KERNEL_INTEGRITY_CTRR)
1946 if (region_select == 0) {
1947 /*
1948 * In this config, the bootstrap mappings should occupy their own L2
1949 * TTs, as they should be immutable after boot. Having the associated
1950 * TTEs and PTEs in their own pages allows us to lock down those pages,
1951 * while allowing the rest of the kernel address range to be remapped.
1952 */
1953 *startp = LOW_GLOBAL_BASE_ADDRESS & ~ARM_TT_L2_OFFMASK;
1954 #if defined(ARM_LARGE_MEMORY)
1955 *size = ((KERNEL_PMAP_HEAP_RANGE_START - *startp) & ~PAGE_MASK);
1956 #else
1957 *size = ((VM_MAX_KERNEL_ADDRESS - *startp) & ~PAGE_MASK);
1958 #endif
1959 ret = TRUE;
1960 }
1961
1962 #if defined(ARM_LARGE_MEMORY)
1963 if (region_select == 1) {
1964 *startp = VREGION1_START;
1965 *size = VREGION1_SIZE;
1966 ret = TRUE;
1967 }
1968 #endif
1969 #else /* !(defined(KERNEL_INTEGRITY_KTRR) || defined(KERNEL_INTEGRITY_CTRR)) */
1970 #if defined(ARM_LARGE_MEMORY)
1971 /* For large memory systems with no KTRR/CTRR such as virtual machines */
1972 if (region_select == 0) {
1973 *startp = LOW_GLOBAL_BASE_ADDRESS & ~ARM_TT_L2_OFFMASK;
1974 *size = ((KERNEL_PMAP_HEAP_RANGE_START - *startp) & ~PAGE_MASK);
1975 ret = TRUE;
1976 }
1977
1978 if (region_select == 1) {
1979 *startp = VREGION1_START;
1980 *size = VREGION1_SIZE;
1981 ret = TRUE;
1982 }
1983 #else /* !defined(ARM_LARGE_MEMORY) */
1984 unsigned long low_global_vr_mask = 0;
1985 vm_map_size_t low_global_vr_size = 0;
1986
1987 if (region_select == 0) {
1988 /* Round to avoid overlapping with the V=P area; round to at least the L2 block size. */
1989 if (!TEST_PAGE_SIZE_4K) {
1990 *startp = gVirtBase & 0xFFFFFFFFFE000000;
1991 *size = ((virtual_space_start - (gVirtBase & 0xFFFFFFFFFE000000)) + ~0xFFFFFFFFFE000000) & 0xFFFFFFFFFE000000;
1992 } else {
1993 *startp = gVirtBase & 0xFFFFFFFFFF800000;
1994 *size = ((virtual_space_start - (gVirtBase & 0xFFFFFFFFFF800000)) + ~0xFFFFFFFFFF800000) & 0xFFFFFFFFFF800000;
1995 }
1996 ret = TRUE;
1997 }
1998 if (region_select == 1) {
1999 *startp = VREGION1_START;
2000 *size = VREGION1_SIZE;
2001 ret = TRUE;
2002 }
2003 /* We need to reserve a range that is at least the size of an L2 block mapping for the low globals */
2004 if (!TEST_PAGE_SIZE_4K) {
2005 low_global_vr_mask = 0xFFFFFFFFFE000000;
2006 low_global_vr_size = 0x2000000;
2007 } else {
2008 low_global_vr_mask = 0xFFFFFFFFFF800000;
2009 low_global_vr_size = 0x800000;
2010 }
2011
2012 if (((gVirtBase & low_global_vr_mask) != LOW_GLOBAL_BASE_ADDRESS) && (region_select == 2)) {
2013 *startp = LOW_GLOBAL_BASE_ADDRESS;
2014 *size = low_global_vr_size;
2015 ret = TRUE;
2016 }
2017
2018 if (region_select == 3) {
2019 /* In this config, we allow the bootstrap mappings to occupy the same
2020 * page table pages as the heap.
2021 */
2022 *startp = VM_MIN_KERNEL_ADDRESS;
2023 *size = LOW_GLOBAL_BASE_ADDRESS - *startp;
2024 ret = TRUE;
2025 }
2026 #endif /* defined(ARM_LARGE_MEMORY) */
2027 #endif /* defined(KERNEL_INTEGRITY_KTRR) || defined(KERNEL_INTEGRITY_CTRR) */
2028 return ret;
2029 }
2030
2031 /*
2032 * Routines to track and allocate physical pages during early boot.
2033 * On most systems that memory runs from first_avail through to avail_end
2034 * with no gaps.
2035 *
2036 * If the system supports ECC and ecc_bad_pages_count > 0, we
2037 * need to skip those pages.
2038 */
2039
2040 static unsigned int avail_page_count = 0;
2041 static bool need_ram_ranges_init = true;
2042
2043
2044 /**
2045 * Checks to see if a given page is in
2046 * the array of known bad pages
2047 *
2048 * @param ppn page number to check
2049 */
2050 bool
pmap_is_bad_ram(__unused ppnum_t ppn)2051 pmap_is_bad_ram(__unused ppnum_t ppn)
2052 {
2053 return false;
2054 }
2055
2056 /**
2057 * Prepare bad ram pages to be skipped.
2058 */
2059
2060
2061 /*
2062 * Initialize the count of available pages. No lock needed here,
2063 * as this code is called while kernel boot up is single threaded.
2064 */
2065 static void
initialize_ram_ranges(void)2066 initialize_ram_ranges(void)
2067 {
2068 pmap_paddr_t first = first_avail;
2069 pmap_paddr_t end = avail_end;
2070
2071 assert(first <= end);
2072 assert(first == (first & ~PAGE_MASK));
2073 assert(end == (end & ~PAGE_MASK));
2074 avail_page_count = atop(end - first);
2075
2076 need_ram_ranges_init = false;
2077
2078 }
2079
2080 unsigned int
pmap_free_pages(void)2081 pmap_free_pages(
2082 void)
2083 {
2084 if (need_ram_ranges_init) {
2085 initialize_ram_ranges();
2086 }
2087 return avail_page_count;
2088 }
2089
2090 unsigned int
pmap_free_pages_span(void)2091 pmap_free_pages_span(
2092 void)
2093 {
2094 if (need_ram_ranges_init) {
2095 initialize_ram_ranges();
2096 }
2097 return (unsigned int)atop(avail_end - first_avail);
2098 }
2099
2100
2101 boolean_t
pmap_next_page_hi(ppnum_t * pnum,__unused boolean_t might_free)2102 pmap_next_page_hi(
2103 ppnum_t * pnum,
2104 __unused boolean_t might_free)
2105 {
2106 return pmap_next_page(pnum);
2107 }
2108
2109
2110 boolean_t
pmap_next_page(ppnum_t * pnum)2111 pmap_next_page(
2112 ppnum_t *pnum)
2113 {
2114 if (need_ram_ranges_init) {
2115 initialize_ram_ranges();
2116 }
2117
2118
2119 if (first_avail != avail_end) {
2120 *pnum = (ppnum_t)atop(first_avail);
2121 first_avail += PAGE_SIZE;
2122 assert(avail_page_count > 0);
2123 --avail_page_count;
2124 return TRUE;
2125 }
2126 assert(avail_page_count == 0);
2127 return FALSE;
2128 }
2129
2130
2131
2132
2133 /*
2134 * Initialize the pmap module.
2135 * Called by vm_init, to initialize any structures that the pmap
2136 * system needs to map virtual memory.
2137 */
2138 void
pmap_init(void)2139 pmap_init(
2140 void)
2141 {
2142 /*
2143 * Protect page zero in the kernel map.
2144 * (can be overruled by permanent transltion
2145 * table entries at page zero - see arm_vm_init).
2146 */
2147 vm_protect(kernel_map, 0, PAGE_SIZE, TRUE, VM_PROT_NONE);
2148
2149 pmap_initialized = TRUE;
2150
2151 /*
2152 * Create the zone of physical maps
2153 * and the physical-to-virtual entries.
2154 */
2155 pmap_zone = zone_create_ext("pmap", sizeof(struct pmap),
2156 ZC_ZFREE_CLEARMEM, ZONE_ID_PMAP, NULL);
2157
2158
2159 /*
2160 * Initialize the pmap object (for tracking the vm_page_t
2161 * structures for pages we allocate to be page tables in
2162 * pmap_expand().
2163 */
2164 _vm_object_allocate(mem_size, pmap_object);
2165 pmap_object->copy_strategy = MEMORY_OBJECT_COPY_NONE;
2166
2167 /*
2168 * Initialize the TXM VM object in the same way as the
2169 * PMAP VM object.
2170 */
2171 _vm_object_allocate(mem_size, txm_vm_object);
2172 txm_vm_object->copy_strategy = MEMORY_OBJECT_COPY_NONE;
2173
2174 /*
2175 * The values of [hard_]maxproc may have been scaled, make sure
2176 * they are still less than the value of pmap_max_asids.
2177 */
2178 if ((uint32_t)maxproc > pmap_max_asids) {
2179 maxproc = pmap_max_asids;
2180 }
2181 if ((uint32_t)hard_maxproc > pmap_max_asids) {
2182 hard_maxproc = pmap_max_asids;
2183 }
2184 }
2185
2186 /**
2187 * Verify that a given physical page contains no mappings (outside of the
2188 * default physical aperture mapping).
2189 *
2190 * @param ppnum Physical page number to check there are no mappings to.
2191 *
2192 * @return True if there are no mappings, false otherwise or if the page is not
2193 * kernel-managed.
2194 */
2195 bool
pmap_verify_free(ppnum_t ppnum)2196 pmap_verify_free(ppnum_t ppnum)
2197 {
2198 const pmap_paddr_t pa = ptoa(ppnum);
2199
2200 assert(pa != vm_page_fictitious_addr);
2201
2202 /* Only mappings to kernel-managed physical memory are tracked. */
2203 if (!pa_valid(pa)) {
2204 return false;
2205 }
2206
2207 const unsigned int pai = pa_index(pa);
2208
2209 return pvh_test_type(pai_to_pvh(pai), PVH_TYPE_NULL);
2210 }
2211
2212 #if MACH_ASSERT
2213 /**
2214 * Verify that a given physical page contains no mappings (outside of the
2215 * default physical aperture mapping) and if it does, then panic.
2216 *
2217 * @note It's recommended to use pmap_verify_free() directly when operating in
2218 * the PPL since the PVH lock isn't getting grabbed here (due to this code
2219 * normally being called from outside of the PPL, and the pv_head_table
2220 * can't be modified outside of the PPL).
2221 *
2222 * @param ppnum Physical page number to check there are no mappings to.
2223 */
2224 void
pmap_assert_free(ppnum_t ppnum)2225 pmap_assert_free(ppnum_t ppnum)
2226 {
2227 const pmap_paddr_t pa = ptoa(ppnum);
2228
2229 /* Only mappings to kernel-managed physical memory are tracked. */
2230 if (__probable(!pa_valid(pa) || pmap_verify_free(ppnum))) {
2231 return;
2232 }
2233
2234 const unsigned int pai = pa_index(pa);
2235 const uintptr_t pvh = pai_to_pvh(pai);
2236
2237 /**
2238 * This function is always called from outside of the PPL. Because of this,
2239 * the PVH entry can't be locked. This function is generally only called
2240 * before the VM reclaims a physical page and shouldn't be creating new
2241 * mappings. Even if a new mapping is created while parsing the hierarchy,
2242 * the worst case is that the system will panic in another way, and we were
2243 * already about to panic anyway.
2244 */
2245
2246 /**
2247 * Since pmap_verify_free() returned false, that means there is at least one
2248 * mapping left. Let's get some extra info on the first mapping we find to
2249 * dump in the panic string (the common case is that there is one spare
2250 * mapping that was never unmapped).
2251 */
2252 pt_entry_t *first_ptep = PT_ENTRY_NULL;
2253
2254 if (pvh_test_type(pvh, PVH_TYPE_PTEP)) {
2255 first_ptep = pvh_ptep(pvh);
2256 } else if (pvh_test_type(pvh, PVH_TYPE_PVEP)) {
2257 pv_entry_t *pvep = pvh_pve_list(pvh);
2258
2259 /* Each PVE can contain multiple PTEs. Let's find the first one. */
2260 for (int pve_ptep_idx = 0; pve_ptep_idx < PTE_PER_PVE; pve_ptep_idx++) {
2261 first_ptep = pve_get_ptep(pvep, pve_ptep_idx);
2262 if (first_ptep != PT_ENTRY_NULL) {
2263 break;
2264 }
2265 }
2266
2267 /* The PVE should have at least one valid PTE. */
2268 assert(first_ptep != PT_ENTRY_NULL);
2269 } else if (pvh_test_type(pvh, PVH_TYPE_PTDP)) {
2270 panic("%s: Physical page is being used as a page table at PVH %p (pai: %d)",
2271 __func__, (void*)pvh, pai);
2272 } else {
2273 /**
2274 * The mapping disappeared between here and the pmap_verify_free() call.
2275 * The only way that can happen is if the VM was racing this call with
2276 * a call that unmaps PTEs. Operations on this page should not be
2277 * occurring at the same time as this check, and unfortunately we can't
2278 * lock the PVH entry to prevent it, so just panic instead.
2279 */
2280 panic("%s: Mapping was detected but is now gone. Is the VM racing this "
2281 "call with an operation that unmaps PTEs? PVH %p (pai: %d)",
2282 __func__, (void*)pvh, pai);
2283 }
2284
2285 /* Panic with a unique string identifying the first bad mapping and owner. */
2286 {
2287 /* First PTE is mapped by the main CPUs. */
2288 pmap_t pmap = ptep_get_pmap(first_ptep);
2289 const char *type = (pmap == kernel_pmap) ? "Kernel" : "User";
2290
2291 panic("%s: Found at least one mapping to %#llx. First PTEP (%p) is a "
2292 "%s CPU mapping (pmap: %p)",
2293 __func__, (uint64_t)pa, first_ptep, type, pmap);
2294 }
2295 }
2296 #endif
2297
2298
2299 static vm_size_t
pmap_root_alloc_size(pmap_t pmap)2300 pmap_root_alloc_size(pmap_t pmap)
2301 {
2302 #pragma unused(pmap)
2303 const pt_attr_t * const pt_attr = pmap_get_pt_attr(pmap);
2304 unsigned int root_level = pt_attr_root_level(pt_attr);
2305 return ((pt_attr_ln_index_mask(pt_attr, root_level) >> pt_attr_ln_shift(pt_attr, root_level)) + 1) * sizeof(tt_entry_t);
2306 }
2307
2308 /*
2309 * Create and return a physical map.
2310 *
2311 * If the size specified for the map
2312 * is zero, the map is an actual physical
2313 * map, and may be referenced by the
2314 * hardware.
2315 *
2316 * If the size specified is non-zero,
2317 * the map will be used in software only, and
2318 * is bounded by that size.
2319 */
2320 MARK_AS_PMAP_TEXT pmap_t
pmap_create_options_internal(ledger_t ledger,vm_map_size_t size,unsigned int flags,kern_return_t * kr)2321 pmap_create_options_internal(
2322 ledger_t ledger,
2323 vm_map_size_t size,
2324 unsigned int flags,
2325 kern_return_t *kr)
2326 {
2327 pmap_t p;
2328 bool is_64bit = flags & PMAP_CREATE_64BIT;
2329 #if defined(HAS_APPLE_PAC)
2330 bool disable_jop = flags & PMAP_CREATE_DISABLE_JOP;
2331 #endif /* defined(HAS_APPLE_PAC) */
2332 kern_return_t local_kr = KERN_SUCCESS;
2333 __unused uint8_t sptm_root_flags = SPTM_ROOT_PT_FLAGS_DEFAULT;
2334 TXMAddressSpaceFlags_t txm_flags = kTXMAddressSpaceFlagInit;
2335 const bool is_stage2 = false;
2336
2337 if (size != 0) {
2338 {
2339 // Size parameter should only be set for stage 2.
2340 return PMAP_NULL;
2341 }
2342 }
2343
2344 if (0 != (flags & ~PMAP_CREATE_KNOWN_FLAGS)) {
2345 return PMAP_NULL;
2346 }
2347
2348 /*
2349 * Allocate a pmap struct from the pmap_zone. Then allocate
2350 * the translation table of the right size for the pmap.
2351 */
2352 if ((p = (pmap_t) zalloc(pmap_zone)) == PMAP_NULL) {
2353 local_kr = KERN_RESOURCE_SHORTAGE;
2354 goto pmap_create_fail;
2355 }
2356
2357 p->ledger = ledger;
2358
2359
2360 p->pmap_vm_map_cs_enforced = false;
2361 p->min = 0;
2362
2363
2364 #if CONFIG_ROSETTA
2365 if (flags & PMAP_CREATE_ROSETTA) {
2366 p->is_rosetta = TRUE;
2367 } else {
2368 p->is_rosetta = FALSE;
2369 }
2370 #endif /* CONFIG_ROSETTA */
2371 #if defined(HAS_APPLE_PAC)
2372 p->disable_jop = disable_jop;
2373
2374 if (p->disable_jop) {
2375 sptm_root_flags &= ~SPTM_ROOT_PT_FLAG_JOP;
2376 }
2377 #endif /* defined(HAS_APPLE_PAC) */
2378
2379 p->nested_region_true_start = 0;
2380 p->nested_region_true_end = ~0;
2381
2382 p->nx_enabled = true;
2383 p->is_64bit = is_64bit;
2384 p->nested_pmap = PMAP_NULL;
2385 p->type = PMAP_TYPE_USER;
2386
2387 #if ARM_PARAMETERIZED_PMAP
2388 /* Default to the native pt_attr */
2389 p->pmap_pt_attr = native_pt_attr;
2390 #endif /* ARM_PARAMETERIZED_PMAP */
2391 #if __ARM_MIXED_PAGE_SIZE__
2392 if (flags & PMAP_CREATE_FORCE_4K_PAGES) {
2393 p->pmap_pt_attr = &pmap_pt_attr_4k;
2394 }
2395 #endif /* __ARM_MIXED_PAGE_SIZE__ */
2396 p->max = pmap_user_va_size(p);
2397
2398 if (!pmap_get_pt_ops(p)->alloc_id(p)) {
2399 local_kr = KERN_NO_SPACE;
2400 goto id_alloc_fail;
2401 }
2402
2403 /**
2404 * We expect top level translation tables to always fit into a single
2405 * physical page. This would also catch a misconfiguration if 4K
2406 * concatenated page tables needed more than one physical tt1 page.
2407 */
2408 vm_size_t pmap_root_size = pmap_root_alloc_size(p);
2409 if (__improbable(pmap_root_size > PAGE_SIZE)) {
2410 panic("%s: translation tables do not fit into a single physical page %u", __FUNCTION__, (unsigned)pmap_root_size);
2411 }
2412
2413 pmap_lock_init(p);
2414
2415 p->tte = pmap_tt1_allocate(p, sptm_root_flags);
2416 if (!(p->tte)) {
2417 local_kr = KERN_RESOURCE_SHORTAGE;
2418 goto tt1_alloc_fail;
2419 }
2420
2421 p->ttep = kvtophys_nofail((vm_offset_t)p->tte);
2422 PMAP_TRACE(4, PMAP_CODE(PMAP__TTE), VM_KERNEL_ADDRHIDE(p), VM_KERNEL_ADDRHIDE(p->min), VM_KERNEL_ADDRHIDE(p->max), p->ttep);
2423
2424 /*
2425 * initialize the rest of the structure
2426 */
2427 p->nested_region_addr = 0x0ULL;
2428 p->nested_region_size = 0x0ULL;
2429 p->nested_region_unnested_table_bitmap = NULL;
2430
2431 p->nested_has_no_bounds_ref = false;
2432 p->nested_no_bounds_refcnt = 0;
2433 p->nested_bounds_set = false;
2434
2435
2436 #if MACH_ASSERT
2437 p->pmap_pid = 0;
2438 strlcpy(p->pmap_procname, "<nil>", sizeof(p->pmap_procname));
2439 #endif /* MACH_ASSERT */
2440 #if DEVELOPMENT || DEBUG
2441 p->footprint_was_suspended = FALSE;
2442 #endif /* DEVELOPMENT || DEBUG */
2443
2444 os_ref_init_count_raw(&p->ref_count, &pmap_refgrp, 1);
2445 pmap_simple_lock(&pmaps_lock);
2446 queue_enter(&map_pmap_list, p, pmap_t, pmaps);
2447 pmap_simple_unlock(&pmaps_lock);
2448
2449 /**
2450 * The SPTM pmap's concurrency model can sometimes allow ledger balances to transiently
2451 * go negative. Note that we still check overall ledger balance on pmap destruction.
2452 */
2453 ledger_disable_panic_on_negative(p->ledger, task_ledgers.phys_footprint);
2454 ledger_disable_panic_on_negative(p->ledger, task_ledgers.internal);
2455 ledger_disable_panic_on_negative(p->ledger, task_ledgers.internal_compressed);
2456 ledger_disable_panic_on_negative(p->ledger, task_ledgers.iokit_mapped);
2457 ledger_disable_panic_on_negative(p->ledger, task_ledgers.alternate_accounting);
2458 ledger_disable_panic_on_negative(p->ledger, task_ledgers.alternate_accounting_compressed);
2459 ledger_disable_panic_on_negative(p->ledger, task_ledgers.external);
2460 ledger_disable_panic_on_negative(p->ledger, task_ledgers.reusable);
2461 ledger_disable_panic_on_negative(p->ledger, task_ledgers.wired_mem);
2462
2463 if (!is_stage2) {
2464 /*
2465 * Complete initialization for the TXM address space. This needs to be done
2466 * after the SW ASID has been registered with the SPTM.
2467 * TXM enforcement does not apply to virtual machines.
2468 */
2469 if (flags & PMAP_CREATE_TEST) {
2470 txm_flags |= kTXMAddressSpaceFlagTest;
2471 }
2472
2473 pmap_txmlock_init(p);
2474 txm_register_address_space(p, p->asid, txm_flags);
2475 p->txm_trust_level = kCSTrustUntrusted;
2476 }
2477
2478 return p;
2479
2480 tt1_alloc_fail:
2481 pmap_get_pt_ops(p)->free_id(p);
2482 id_alloc_fail:
2483 zfree(pmap_zone, p);
2484 pmap_create_fail:
2485 *kr = local_kr;
2486 return PMAP_NULL;
2487 }
2488
2489 pmap_t
pmap_create_options(ledger_t ledger,vm_map_size_t size,unsigned int flags)2490 pmap_create_options(
2491 ledger_t ledger,
2492 vm_map_size_t size,
2493 unsigned int flags)
2494 {
2495 pmap_t pmap;
2496 kern_return_t kr = KERN_SUCCESS;
2497
2498 PMAP_TRACE(1, PMAP_CODE(PMAP__CREATE) | DBG_FUNC_START, size, flags);
2499
2500 ledger_reference(ledger);
2501
2502 pmap = pmap_create_options_internal(ledger, size, flags, &kr);
2503
2504 if (pmap == PMAP_NULL) {
2505 ledger_dereference(ledger);
2506 }
2507
2508 PMAP_TRACE(1, PMAP_CODE(PMAP__CREATE) | DBG_FUNC_END, VM_KERNEL_ADDRHIDE(pmap), PMAP_VASID(pmap), PMAP_HWASID(pmap));
2509
2510 return pmap;
2511 }
2512
2513 #if MACH_ASSERT
2514 MARK_AS_PMAP_TEXT void
pmap_set_process_internal(__unused pmap_t pmap,__unused int pid,__unused char * procname)2515 pmap_set_process_internal(
2516 __unused pmap_t pmap,
2517 __unused int pid,
2518 __unused char *procname)
2519 {
2520 if (pmap == NULL || pmap->pmap_pid == -1) {
2521 return;
2522 }
2523
2524 validate_pmap_mutable(pmap);
2525
2526 pmap->pmap_pid = pid;
2527 strlcpy(pmap->pmap_procname, procname, sizeof(pmap->pmap_procname));
2528 }
2529 #endif /* MACH_ASSERT */
2530
2531 #if MACH_ASSERT
2532 void
pmap_set_process(pmap_t pmap,int pid,char * procname)2533 pmap_set_process(
2534 pmap_t pmap,
2535 int pid,
2536 char *procname)
2537 {
2538 pmap_set_process_internal(pmap, pid, procname);
2539 }
2540 #endif /* MACH_ASSERT */
2541
2542 /*
2543 * pmap_deallocate_all_leaf_tts:
2544 *
2545 * Recursive function for deallocating all leaf TTEs. Walks the given TT,
2546 * removing and deallocating all TTEs.
2547 */
2548 MARK_AS_PMAP_TEXT static void
pmap_deallocate_all_leaf_tts(pmap_t pmap,tt_entry_t * first_ttep,vm_map_address_t start_va,unsigned level)2549 pmap_deallocate_all_leaf_tts(pmap_t pmap, tt_entry_t * first_ttep, vm_map_address_t start_va, unsigned level)
2550 {
2551 tt_entry_t tte = ARM_TTE_EMPTY;
2552 tt_entry_t * ttep = NULL;
2553 tt_entry_t * last_ttep = NULL;
2554
2555 const pt_attr_t * const pt_attr = pmap_get_pt_attr(pmap);
2556 const uint64_t size = pt_attr->pta_level_info[level].size;
2557
2558 assert(level < pt_attr_leaf_level(pt_attr));
2559
2560 last_ttep = &first_ttep[ttn_index(pt_attr, ~0, level)];
2561
2562 const uint64_t page_ratio = PAGE_SIZE / pt_attr_page_size(pt_attr);
2563 vm_map_address_t va = start_va;
2564 for (ttep = first_ttep; ttep <= last_ttep; ttep += page_ratio, va += (size * page_ratio)) {
2565 if (!(*ttep & ARM_TTE_VALID)) {
2566 continue;
2567 }
2568
2569 for (unsigned i = 0; i < page_ratio; i++) {
2570 tte = ttep[i];
2571
2572 if (!(tte & ARM_TTE_VALID)) {
2573 panic("%s: found unexpectedly invalid tte, ttep=%p, tte=%p, "
2574 "pmap=%p, first_ttep=%p, level=%u",
2575 __FUNCTION__, ttep + i, (void *)tte,
2576 pmap, first_ttep, level);
2577 }
2578
2579 if ((tte & ARM_TTE_TYPE_MASK) == ARM_TTE_TYPE_BLOCK) {
2580 panic("%s: found block mapping, ttep=%p, tte=%p, "
2581 "pmap=%p, first_ttep=%p, level=%u",
2582 __FUNCTION__, ttep + i, (void *)tte,
2583 pmap, first_ttep, level);
2584 }
2585
2586 /* Must be valid, type table */
2587 if (level < pt_attr_twig_level(pt_attr)) {
2588 /* If we haven't reached the twig level, recurse to the next level. */
2589 pmap_deallocate_all_leaf_tts(pmap, (tt_entry_t *)phystokv((tte) & ARM_TTE_TABLE_MASK),
2590 va + (size * i), level + 1);
2591 }
2592 }
2593
2594 /* Remove the TTE. */
2595 pmap_lock(pmap, PMAP_LOCK_EXCLUSIVE);
2596 pmap_tte_deallocate(pmap, va, ttep, level);
2597 }
2598 }
2599
2600 /*
2601 * We maintain stats and ledgers so that a task's physical footprint is:
2602 * phys_footprint = ((internal - alternate_accounting)
2603 * + (internal_compressed - alternate_accounting_compressed)
2604 * + iokit_mapped
2605 * + purgeable_nonvolatile
2606 * + purgeable_nonvolatile_compressed
2607 * + page_table)
2608 * where "alternate_accounting" includes "iokit" and "purgeable" memory.
2609 */
2610
2611 /*
2612 * Retire the given physical map from service.
2613 * Should only be called if the map contains
2614 * no valid mappings.
2615 */
2616 MARK_AS_PMAP_TEXT void
pmap_destroy_internal(pmap_t pmap)2617 pmap_destroy_internal(
2618 pmap_t pmap)
2619 {
2620 if (pmap == PMAP_NULL) {
2621 return;
2622 }
2623
2624 validate_pmap(pmap);
2625
2626 __unused const pt_attr_t * const pt_attr = pmap_get_pt_attr(pmap);
2627 const bool is_stage2_pmap = false;
2628
2629 if (os_ref_release_raw(&pmap->ref_count, &pmap_refgrp) > 0) {
2630 return;
2631 }
2632
2633 if (!is_stage2_pmap) {
2634 /*
2635 * Complete all clean up required for TXM. This needs to happen before the
2636 * SW ASID has been unregistered with the SPTM.
2637 */
2638 txm_unregister_address_space(pmap);
2639 pmap_txmlock_destroy(pmap);
2640 }
2641
2642 /**
2643 * Drain any concurrent retype-sensitive SPTM operations. This is needed to
2644 * ensure that we don't unmap and retype the page tables while those operations
2645 * are still finishing on other CPUs, leading to an SPTM violation. In particular,
2646 * the multipage batched cacheability/attribute update code may issue SPTM calls
2647 * without holding the relevant PVH or pmap locks, so we can't guarantee those
2648 * calls have actually completed despite observing refcnt == 0.
2649 *
2650 * At this point, we CAN guarantee that:
2651 * 1) All prior PTE removals required to empty the pmap have completed and
2652 * been synchronized with DSB, *except* the commpage removal which doesn't
2653 * involve pages that can ever be retyped. Subsequent calls not already
2654 * in the retype epoch will no longer observe these mappings.
2655 * 2) The pmap now has a zero refcount, so in a correctly functioning system
2656 * no further mappings will be requested for it.
2657 */
2658 pmap_retype_epoch_prepare_drain();
2659
2660 if (!is_stage2_pmap) {
2661 pmap_unmap_commpage(pmap);
2662 }
2663
2664 pmap_simple_lock(&pmaps_lock);
2665 queue_remove(&map_pmap_list, pmap, pmap_t, pmaps);
2666 pmap_simple_unlock(&pmaps_lock);
2667
2668 pmap_retype_epoch_drain();
2669
2670 pmap_trim_self(pmap);
2671
2672 /*
2673 * Free the memory maps, then the
2674 * pmap structure.
2675 */
2676 pmap_deallocate_all_leaf_tts(pmap, pmap->tte, pmap->min, pt_attr_root_level(pt_attr));
2677
2678 if (pmap->tte) {
2679 pmap_tt1_deallocate(pmap, pmap->tte);
2680 pmap->tte = (tt_entry_t *) NULL;
2681 pmap->ttep = 0;
2682 }
2683
2684 if (pmap->type != PMAP_TYPE_NESTED) {
2685 /* return its asid to the pool */
2686 pmap_get_pt_ops(pmap)->free_id(pmap);
2687 if (pmap->nested_pmap != NULL) {
2688 /* release the reference we hold on the nested pmap */
2689 pmap_destroy_internal(pmap->nested_pmap);
2690 }
2691 }
2692
2693 pmap_check_ledgers(pmap);
2694
2695 if (pmap->nested_region_unnested_table_bitmap) {
2696 bitmap_free(pmap->nested_region_unnested_table_bitmap, pmap->nested_region_size >> pt_attr_twig_shift(pt_attr));
2697 }
2698
2699 pmap_lock_destroy(pmap);
2700 zfree(pmap_zone, pmap);
2701 }
2702
2703 void
pmap_destroy(pmap_t pmap)2704 pmap_destroy(
2705 pmap_t pmap)
2706 {
2707 PMAP_TRACE(1, PMAP_CODE(PMAP__DESTROY) | DBG_FUNC_START, VM_KERNEL_ADDRHIDE(pmap), PMAP_VASID(pmap), PMAP_HWASID(pmap));
2708
2709 ledger_t ledger = pmap->ledger;
2710
2711 pmap_destroy_internal(pmap);
2712
2713 ledger_dereference(ledger);
2714
2715 PMAP_TRACE(1, PMAP_CODE(PMAP__DESTROY) | DBG_FUNC_END);
2716 }
2717
2718
2719 /*
2720 * Add a reference to the specified pmap.
2721 */
2722 MARK_AS_PMAP_TEXT void
pmap_reference_internal(pmap_t pmap)2723 pmap_reference_internal(
2724 pmap_t pmap)
2725 {
2726 if (pmap != PMAP_NULL) {
2727 validate_pmap_mutable(pmap);
2728 os_ref_retain_raw(&pmap->ref_count, &pmap_refgrp);
2729 }
2730 }
2731
2732 void
pmap_reference(pmap_t pmap)2733 pmap_reference(
2734 pmap_t pmap)
2735 {
2736 pmap_reference_internal(pmap);
2737 }
2738
2739 static sptm_frame_type_t
get_sptm_pt_type(pmap_t pmap)2740 get_sptm_pt_type(pmap_t pmap)
2741 {
2742 const bool is_stage2_pmap = false;
2743 if (is_stage2_pmap) {
2744 assert(pmap->type != PMAP_TYPE_NESTED);
2745 return XNU_STAGE2_PAGE_TABLE;
2746 } else {
2747 return pmap->type == PMAP_TYPE_NESTED ? XNU_PAGE_TABLE_SHARED : XNU_PAGE_TABLE;
2748 }
2749 }
2750
2751 static tt_entry_t *
pmap_tt1_allocate(pmap_t pmap,uint8_t sptm_root_flags)2752 pmap_tt1_allocate(pmap_t pmap, uint8_t sptm_root_flags)
2753 {
2754 pmap_paddr_t pa = 0;
2755 const pt_attr_t * const pt_attr = pmap_get_pt_attr(pmap);
2756 const bool is_stage2_pmap = false;
2757
2758 const kern_return_t ret = pmap_page_alloc(&pa, PMAP_PAGE_NOZEROFILL);
2759
2760 if (ret != KERN_SUCCESS) {
2761 return (tt_entry_t *)0;
2762 }
2763
2764 /**
2765 * Drain the epochs to ensure any lingering batched operations that may have taken
2766 * an in-flight reference to this page are complete.
2767 */
2768 pmap_retype_epoch_prepare_drain();
2769
2770 assert(pa);
2771
2772 /* Always report root allocations in units of PMAP_ROOT_ALLOC_SIZE, which can be obtained by sysctl arm_pt_root_size.
2773 * Depending on the device, this can vary between 512b and 16K. */
2774 OSAddAtomic(1, (pmap == kernel_pmap ? &inuse_kernel_tteroot_count : &inuse_user_tteroot_count));
2775 pmap_tt_ledger_credit(pmap, PAGE_SIZE);
2776
2777 sptm_retype_params_t retype_params = {.raw = SPTM_RETYPE_PARAMS_NULL};
2778 retype_params.attr_idx = pt_attr->geometry_id;
2779 retype_params.flags = sptm_root_flags;
2780 if (is_stage2_pmap) {
2781 retype_params.vmid = pmap->vmid;
2782 } else {
2783 retype_params.asid = pmap->asid;
2784 }
2785
2786 pmap_retype_epoch_drain();
2787
2788 sptm_retype(pa, XNU_DEFAULT, is_stage2_pmap ? XNU_STAGE2_ROOT_TABLE : XNU_USER_ROOT_TABLE,
2789 retype_params);
2790
2791 return (tt_entry_t *) phystokv(pa);
2792 }
2793
2794 static void
pmap_tt1_deallocate(pmap_t pmap,tt_entry_t * tt)2795 pmap_tt1_deallocate(
2796 pmap_t pmap,
2797 tt_entry_t *tt)
2798 {
2799 pmap_paddr_t pa = kvtophys_nofail((vm_offset_t)tt);
2800 const bool is_stage2_pmap = false;
2801 const sptm_frame_type_t page_type = is_stage2_pmap ? XNU_STAGE2_ROOT_TABLE :
2802 pmap->type == PMAP_TYPE_NESTED ? XNU_SHARED_ROOT_TABLE : XNU_USER_ROOT_TABLE;
2803
2804 sptm_retype_params_t retype_params = {.raw = SPTM_RETYPE_PARAMS_NULL};
2805 sptm_retype(pa, page_type, XNU_DEFAULT, retype_params);
2806 pmap_page_free(pa);
2807
2808 OSAddAtomic(-1, (pmap == kernel_pmap ? &inuse_kernel_tteroot_count : &inuse_user_tteroot_count));
2809 pmap_tt_ledger_debit(pmap, PAGE_SIZE);
2810 }
2811
2812 MARK_AS_PMAP_TEXT static kern_return_t
pmap_tt_allocate(pmap_t pmap,tt_entry_t ** ttp,unsigned int level,unsigned int options)2813 pmap_tt_allocate(
2814 pmap_t pmap,
2815 tt_entry_t **ttp,
2816 unsigned int level,
2817 unsigned int options)
2818 {
2819 pmap_paddr_t pa;
2820 *ttp = NULL;
2821
2822 if (*ttp == NULL) {
2823 const unsigned int alloc_flags =
2824 (options & PMAP_TT_ALLOCATE_NOWAIT) ? PMAP_PAGE_ALLOCATE_NOWAIT : 0;
2825
2826 /* Allocate a VM page to be used as the page table. */
2827 if (pmap_page_alloc(&pa, alloc_flags) != KERN_SUCCESS) {
2828 return KERN_RESOURCE_SHORTAGE;
2829 }
2830
2831 pt_desc_t *ptdp = ptd_alloc(pmap, alloc_flags);
2832 if (ptdp == NULL) {
2833 pmap_page_free(pa);
2834 return KERN_RESOURCE_SHORTAGE;
2835 }
2836
2837 unsigned int pai = pa_index(pa);
2838 locked_pvh_t locked_pvh = pvh_lock(pai);
2839 assertf(pvh_test_type(locked_pvh.pvh, PVH_TYPE_NULL), "%s: non-empty PVH %p",
2840 __func__, (void*)locked_pvh.pvh);
2841
2842 /**
2843 * Drain the epochs to ensure any lingering batched operations that may have taken
2844 * an in-flight reference to this page are complete.
2845 */
2846 pmap_retype_epoch_prepare_drain();
2847
2848 if (level < pt_attr_leaf_level(pmap_get_pt_attr(pmap))) {
2849 OSAddAtomic(1, (pmap == kernel_pmap ? &inuse_kernel_ttepages_count : &inuse_user_ttepages_count));
2850 } else {
2851 OSAddAtomic(1, (pmap == kernel_pmap ? &inuse_kernel_ptepages_count : &inuse_user_ptepages_count));
2852 }
2853
2854 pmap_tt_ledger_credit(pmap, PAGE_SIZE);
2855
2856 PMAP_ZINFO_PALLOC(pmap, PAGE_SIZE);
2857
2858 pvh_update_head(&locked_pvh, ptdp, PVH_TYPE_PTDP);
2859 pvh_unlock(&locked_pvh);
2860
2861 sptm_retype_params_t retype_params = {.raw = SPTM_RETYPE_PARAMS_NULL};
2862 retype_params.level = (sptm_pt_level_t)level;
2863
2864 /**
2865 * SPTM TODO: To reduce the cost of draining and retyping, consider caching freed page table pages
2866 * in a small per-CPU bucket and reusing them in preference to calling pmap_page_alloc() above.
2867 */
2868 pmap_retype_epoch_drain();
2869
2870 sptm_retype(pa, XNU_DEFAULT, get_sptm_pt_type(pmap), retype_params);
2871
2872 *ttp = (tt_entry_t *)phystokv(pa);
2873 }
2874
2875 assert(*ttp);
2876
2877 return KERN_SUCCESS;
2878 }
2879
2880 static void
pmap_tt_deallocate(pmap_t pmap,tt_entry_t * ttp,unsigned int level)2881 pmap_tt_deallocate(
2882 pmap_t pmap,
2883 tt_entry_t *ttp,
2884 unsigned int level)
2885 {
2886 pt_desc_t *ptdp;
2887 vm_offset_t free_page = 0;
2888 const pt_attr_t * const pt_attr = pmap_get_pt_attr(pmap);
2889
2890 ptdp = ptep_get_ptd(ttp);
2891 ptdp->va = (vm_offset_t)-1;
2892
2893 const uint16_t refcnt = sptm_get_page_table_refcnt(kvtophys_nofail((vm_offset_t)ttp));
2894
2895 if (__improbable(refcnt != 0)) {
2896 panic("pmap_tt_deallocate(): ptdp %p, count %d", ptdp, refcnt);
2897 }
2898
2899 free_page = (vm_offset_t)ttp & ~PAGE_MASK;
2900 if (free_page != 0) {
2901 pmap_paddr_t pa = kvtophys_nofail(free_page);
2902 sptm_retype_params_t retype_params = {.raw = SPTM_RETYPE_PARAMS_NULL};
2903 sptm_retype(pa, get_sptm_pt_type(pmap), XNU_DEFAULT, retype_params);
2904 ptd_deallocate(ptep_get_ptd((pt_entry_t*)free_page));
2905
2906 unsigned int pai = pa_index(pa);
2907 locked_pvh_t locked_pvh = pvh_lock(pai);
2908 assertf(pvh_test_type(locked_pvh.pvh, PVH_TYPE_PTDP), "%s: non-PTD PVH %p",
2909 __func__, (void*)locked_pvh.pvh);
2910 pvh_update_head(&locked_pvh, NULL, PVH_TYPE_NULL);
2911 pvh_unlock(&locked_pvh);
2912 pmap_page_free(pa);
2913 if (level < pt_attr_leaf_level(pt_attr)) {
2914 OSAddAtomic(-1, (pmap == kernel_pmap ? &inuse_kernel_ttepages_count : &inuse_user_ttepages_count));
2915 } else {
2916 OSAddAtomic(-1, (pmap == kernel_pmap ? &inuse_kernel_ptepages_count : &inuse_user_ptepages_count));
2917 }
2918 PMAP_ZINFO_PFREE(pmap, PAGE_SIZE);
2919 pmap_tt_ledger_debit(pmap, PAGE_SIZE);
2920 }
2921 }
2922
2923 /**
2924 * Check table refcounts after clearing a translation table entry pointing to that table
2925 *
2926 * @note If the cleared TTE points to a leaf table, then that leaf table
2927 * must have a refcnt of zero before the TTE can be removed.
2928 *
2929 * @param pmap The pmap containing the page table whose TTE is being removed.
2930 * @param tte Value stored in the TTE prior to clearing it
2931 * @param level The level of the page table that contains the TTE being removed
2932 */
2933 static void
pmap_tte_check_refcounts(pmap_t pmap,tt_entry_t tte,unsigned int level)2934 pmap_tte_check_refcounts(
2935 pmap_t pmap,
2936 tt_entry_t tte,
2937 unsigned int level)
2938 {
2939 const pt_attr_t * const pt_attr = pmap_get_pt_attr(pmap);
2940
2941 /**
2942 * Remember, the passed in "level" parameter refers to the level above the
2943 * table that's getting removed (e.g., removing an L2 TTE will unmap an L3
2944 * page table).
2945 */
2946 const bool remove_leaf_table = (level == pt_attr_twig_level(pt_attr));
2947
2948 unsigned short refcnt = 0;
2949
2950 /**
2951 * It's possible that a concurrent pmap_disconnect() operation may need to reference
2952 * a PTE on the pagetable page to be removed. A full disconnect() may have cleared
2953 * one or more PTEs on this page but not yet dropped the refcount, which would cause
2954 * us to panic in this function on a non-zero refcount. Moreover, it's possible for
2955 * a disconnect-to-compress operation to set the compressed marker on a PTE, and
2956 * for pmap_remove_range_options() to concurrently observe that marker, clear it, and
2957 * drop the pagetable refcount accordingly, without taking any PVH locks that could
2958 * synchronize it against the disconnect operation. If that removal caused the
2959 * refcount to reach zero, the pagetable page could be freed before the disconnect
2960 * operation is finished using the relevant pagetable descriptor.
2961 * Address these cases by waiting until all CPUs have been observed to not be
2962 * executing pmap_disconnect().
2963 */
2964 if (remove_leaf_table) {
2965 bitmap_t active_disconnects[BITMAP_LEN(MAX_CPUS)];
2966 const int max_cpu = ml_get_max_cpu_number();
2967 bitmap_full(&active_disconnects[0], max_cpu + 1);
2968 bool inflight_disconnect;
2969
2970 /*
2971 * Ensure the ensuing load of per-CPU inflight_disconnect is not speculated
2972 * ahead of any prior PTE load which may have observed the effect of a
2973 * concurrent disconnect operation. An acquire fence is required for this;
2974 * a load-acquire operation is insufficient.
2975 */
2976 os_atomic_thread_fence(acquire);
2977 do {
2978 inflight_disconnect = false;
2979 for (int i = bitmap_first(&active_disconnects[0], max_cpu + 1);
2980 i >= 0;
2981 i = bitmap_next(&active_disconnects[0], i)) {
2982 const pmap_cpu_data_t *cpu_data = pmap_get_remote_cpu_data(i);
2983 if (cpu_data == NULL) {
2984 continue;
2985 }
2986 if (os_atomic_load_exclusive(&cpu_data->inflight_disconnect, relaxed)) {
2987 __builtin_arm_wfe();
2988 inflight_disconnect = true;
2989 continue;
2990 }
2991 os_atomic_clear_exclusive();
2992 bitmap_clear(&active_disconnects[0], (unsigned int)i);
2993 }
2994 } while (inflight_disconnect);
2995 /* Ensure the refcount is observed after any observation of inflight_disconnect */
2996 os_atomic_thread_fence(acquire);
2997 refcnt = sptm_get_page_table_refcnt(tte_to_pa(tte));
2998 }
2999
3000 #if MACH_ASSERT
3001 /**
3002 * On internal devices, always do the page table consistency check
3003 * regardless of page table level or the actual refcnt value.
3004 */
3005 {
3006 #else /* MACH_ASSERT */
3007 /**
3008 * Only perform the page table consistency check when deleting leaf page
3009 * tables and it seems like there might be valid/compressed mappings
3010 * leftover.
3011 */
3012 if (__improbable(remove_leaf_table && refcnt != 0)) {
3013 #endif /* MACH_ASSERT */
3014
3015 /**
3016 * There are multiple problems that can arise as a non-zero refcnt:
3017 * 1. A bug in the refcnt management logic.
3018 * 2. A memory stomper or hardware failure.
3019 * 3. The VM forgetting to unmap all of the valid mappings in an address
3020 * space before destroying a pmap.
3021 *
3022 * By looping over the page table and determining how many valid or
3023 * compressed entries there actually are, we can narrow down which of
3024 * these three cases is causing this panic. If the expected refcnt
3025 * (valid + compressed) and the actual refcnt don't match then the
3026 * problem is probably either a memory corruption issue (if the
3027 * non-empty entries don't match valid+compressed, that could also be a
3028 * sign of corruption) or refcnt management bug. Otherwise, there
3029 * actually are leftover mappings and the higher layers of xnu are
3030 * probably at fault.
3031 *
3032 * Note that we use PAGE_SIZE to govern the range of the table check,
3033 * because even for 4K processes we still allocate a 16K page for each
3034 * page table; we simply map it using 4 adjacent TTEs for the 4K case.
3035 */
3036 pt_entry_t *bpte = ((pt_entry_t *) (ttetokv(tte) & ~(PAGE_SIZE - 1)));
3037
3038 pt_entry_t *ptep = bpte;
3039 unsigned short wiredcnt = ptep_get_info((pt_entry_t*)ttetokv(tte))->wiredcnt;
3040 unsigned short non_empty = 0, valid = 0, comp = 0;
3041 for (unsigned int i = 0; i < (PAGE_SIZE / sizeof(*ptep)); i++, ptep++) {
3042 /* Keep track of all non-empty entries to detect memory corruption. */
3043 if (__improbable(*ptep != ARM_PTE_EMPTY)) {
3044 non_empty++;
3045 }
3046
3047 if (__improbable(pte_is_compressed(*ptep, ptep))) {
3048 comp++;
3049 } else if (__improbable((*ptep & ARM_PTE_TYPE_VALID) == ARM_PTE_TYPE)) {
3050 valid++;
3051 }
3052 }
3053
3054 #if MACH_ASSERT
3055 /**
3056 * On internal machines, panic whenever a page table getting deleted has
3057 * leftover mappings (valid or otherwise) or a leaf page table has a
3058 * non-zero refcnt.
3059 */
3060 if (__improbable((non_empty != 0) || (remove_leaf_table && ((refcnt != 0) || (wiredcnt != 0))))) {
3061 #else /* MACH_ASSERT */
3062 /* We already know the leaf page-table has a non-zero refcnt, so panic. */
3063 {
3064 #endif /* MACH_ASSERT */
3065 panic("%s: Found inconsistent state in soon to be deleted L%d table: %d valid, "
3066 "%d compressed, %d non-empty, refcnt=%d, wiredcnt=%d, L%d tte=%#llx, pmap=%p, bpte=%p", __func__,
3067 level + 1, valid, comp, non_empty, refcnt, wiredcnt, level, (uint64_t)tte, pmap, bpte);
3068 }
3069 }
3070 }
3071
3072 /**
3073 * Remove translation table entry pointing to a nested shared region table
3074 *
3075 * @note The TTE to clear out is expected to point to a leaf table with a refcnt
3076 * of zero.
3077 *
3078 * @param pmap The user pmap containing the nested page table whose TTE is being removed.
3079 * @param va_start Beginning of the VA range mapped by the table being removed, for TLB maintenance.
3080 * @param ttep Pointer to the TTE that should be cleared out.
3081 */
3082 static void
3083 pmap_tte_trim(
3084 pmap_t pmap,
3085 vm_offset_t va_start,
3086 tt_entry_t *ttep)
3087 {
3088 pmap_assert_locked(pmap, PMAP_LOCK_EXCLUSIVE);
3089 assert(ttep != NULL);
3090 const tt_entry_t tte = *ttep;
3091 const pt_attr_t * const pt_attr = pmap_get_pt_attr(pmap);
3092
3093 if (__improbable(tte == ARM_TTE_EMPTY)) {
3094 panic("%s: L%d TTE is already empty. Potential double unmap or memory "
3095 "stomper? pmap=%p ttep=%p", __func__, pt_attr_twig_level(pt_attr), pmap, ttep);
3096 }
3097
3098 const uint64_t page_ratio = PAGE_SIZE / pt_attr_page_size(pt_attr);
3099 sptm_unnest_region(pmap->ttep, pmap->nested_pmap->ttep, va_start, (pt_attr_twig_size(pt_attr) * page_ratio) >> pt_attr->pta_page_shift);
3100
3101 pmap_unlock(pmap, PMAP_LOCK_EXCLUSIVE);
3102
3103 pmap_tte_check_refcounts(pmap, tte, pt_attr_twig_level(pt_attr));
3104 }
3105
3106 /**
3107 * Remove a translation table entry.
3108 *
3109 * @note If the TTE to clear out points to a leaf table, then that leaf table
3110 * must have a mapping refcount of zero before the TTE can be removed.
3111 * @note This function expects to be called with pmap locked exclusive, and will
3112 * return with pmap unlocked.
3113 *
3114 * @param pmap The pmap containing the page table whose TTE is being removed.
3115 * @param va_start Beginning of the VA range mapped by the table being removed, for TLB maintenance.
3116 * @param ttep Pointer to the TTE that should be cleared out.
3117 * @param level The level of the page table that contains the TTE to be removed.
3118 */
3119 static void
3120 pmap_tte_remove(
3121 pmap_t pmap,
3122 vm_offset_t va_start,
3123 tt_entry_t *ttep,
3124 unsigned int level)
3125 {
3126 pmap_assert_locked(pmap, PMAP_LOCK_EXCLUSIVE);
3127 assert(ttep != NULL);
3128 const tt_entry_t tte = *ttep;
3129
3130 if (__improbable(tte == ARM_TTE_EMPTY)) {
3131 panic("%s: L%d TTE is already empty. Potential double unmap or memory "
3132 "stomper? pmap=%p ttep=%p", __func__, level, pmap, ttep);
3133 }
3134
3135 sptm_unmap_table(pmap->ttep, pt_attr_align_va(pmap_get_pt_attr(pmap), level, va_start), (sptm_pt_level_t)level);
3136
3137 pmap_unlock(pmap, PMAP_LOCK_EXCLUSIVE);
3138
3139 pmap_tte_check_refcounts(pmap, tte, level);
3140 }
3141
3142 /**
3143 * Given a pointer to an entry within a `level` page table, delete the
3144 * page table at `level` + 1 that is represented by that entry. For instance,
3145 * to delete an unused L3 table, `ttep` would be a pointer to the L2 entry that
3146 * contains the PA of the L3 table, and `level` would be "2".
3147 *
3148 * @note If the table getting deallocated is a leaf table, then that leaf table
3149 * must have a mapping refcount of zero before getting deallocated.
3150 * @note This function expects to be called with pmap locked exclusive and will
3151 * return with pmap unlocked.
3152 *
3153 * @param pmap The pmap that owns the page table to be deallocated.
3154 * @param va_start Beginning of the VA range mapped by the table being removed, for TLB maintenance.
3155 * @param ttep Pointer to the `level` TTE to remove.
3156 * @param level The level of the table that contains an entry pointing to the
3157 * table to be removed. The deallocated page table will be a
3158 * `level` + 1 table (so if `level` is 2, then an L3 table will be
3159 * deleted).
3160 */
3161 void
3162 pmap_tte_deallocate(
3163 pmap_t pmap,
3164 vm_offset_t va_start,
3165 tt_entry_t *ttep,
3166 unsigned int level)
3167 {
3168 tt_entry_t tte;
3169
3170 pmap_assert_locked(pmap, PMAP_LOCK_EXCLUSIVE);
3171
3172 tte = *ttep;
3173
3174 if (tte_get_ptd(tte)->pmap != pmap) {
3175 panic("%s: Passed in pmap doesn't own the page table to be deleted ptd=%p ptd->pmap=%p pmap=%p",
3176 __func__, tte_get_ptd(tte), tte_get_ptd(tte)->pmap, pmap);
3177 }
3178
3179 assertf((tte & ARM_TTE_TYPE_MASK) == ARM_TTE_TYPE_TABLE, "%s: invalid TTE %p (0x%llx)",
3180 __func__, ttep, (unsigned long long)tte);
3181
3182 /* pmap_tte_remove() will drop the pmap lock */
3183 pmap_tte_remove(pmap, va_start, ttep, level);
3184
3185 pmap_tt_deallocate(pmap, (tt_entry_t *) phystokv(tte_to_pa(tte)), level + 1);
3186 }
3187
3188 /*
3189 * Remove a range of hardware page-table entries.
3190 * The range is given as the first (inclusive)
3191 * and last (exclusive) virtual addresses mapped by
3192 * the PTE region to be removed.
3193 *
3194 * The pmap must be locked shared.
3195 * If the pmap is not the kernel pmap, the range must lie
3196 * entirely within one pte-page. Assumes that the pte-page exists.
3197 *
3198 * Returns the number of PTE changed
3199 */
3200 MARK_AS_PMAP_TEXT static void
3201 pmap_remove_range(
3202 pmap_t pmap,
3203 vm_map_address_t va,
3204 vm_map_address_t end)
3205 {
3206 pmap_remove_range_options(pmap, va, end, PMAP_OPTIONS_REMOVE);
3207 }
3208
3209 MARK_AS_PMAP_TEXT void
3210 pmap_remove_range_options(
3211 pmap_t pmap,
3212 vm_map_address_t start,
3213 vm_map_address_t end,
3214 int options)
3215 {
3216 const unsigned int sptm_flags = ((options & PMAP_OPTIONS_REMOVE) ? SPTM_REMOVE_COMPRESSED : 0);
3217 unsigned int num_removed = 0;
3218 unsigned int num_external = 0, num_internal = 0, num_reusable = 0;
3219 unsigned int num_alt_internal = 0;
3220 unsigned int num_compressed = 0, num_alt_compressed = 0;
3221 unsigned short num_unwired = 0;
3222 bool need_strong_sync = false;
3223
3224 /*
3225 * The pmap lock should be held here. It will only be held shared in most if not all cases.
3226 */
3227 pmap_assert_locked(pmap, PMAP_LOCK_HELD);
3228
3229 const pt_attr_t * const pt_attr = pmap_get_pt_attr(pmap);
3230 const uint64_t pmap_page_size = PAGE_RATIO * pt_attr_page_size(pt_attr);
3231 const uint64_t pmap_page_shift = pt_attr_leaf_shift(pt_attr);
3232 vm_map_address_t va = start;
3233 pt_entry_t *cpte = pmap_pte(pmap, va);
3234 assert(cpte != NULL);
3235
3236 while (va < end) {
3237 /**
3238 * We may need to sleep when taking the PVH lock below, and our pmap_pv_remove()
3239 * call below may also place the lock in sleep mode if processing a large PV list.
3240 * We therefore can't leave preemption disabled across that code, which means we
3241 * can't directly use the per-CPU prev_ptes array in that code. Since that code
3242 * only cares about the physical address stored in each prev_ptes entry, we'll
3243 * use a local array to stash off only the 4-byte physical address index in order
3244 * to reduce stack usage.
3245 */
3246 unsigned int pai_list[SPTM_MAPPING_LIMIT];
3247 _Static_assert(SPTM_MAPPING_LIMIT <= 64,
3248 "SPTM_MAPPING_LIMIT value causes excessive stack usage for pai_list");
3249
3250 unsigned int num_mappings = (end - va) >> pmap_page_shift;
3251 if (num_mappings > SPTM_MAPPING_LIMIT) {
3252 num_mappings = SPTM_MAPPING_LIMIT;
3253 }
3254
3255 /**
3256 * Disable preemption to ensure that we can safely access per-CPU mapping data after
3257 * issuing the SPTM call.
3258 */
3259 disable_preemption();
3260 /**
3261 * Enter the retype epoch for the batched unmap operation. This is necessary because we
3262 * cannot reasonably hold the PVH locks for all pages mapped by the region during this
3263 * call, so a concurrent pmap_page_protect() operation against one of those pages may
3264 * race this call. That should be perfectly fine as far as the PTE updates are concerned,
3265 * but if pmap_page_protect() then needs to retype the page, an SPTM violation may result
3266 * if it does not first drain our epoch.
3267 */
3268 pmap_retype_epoch_enter();
3269 sptm_unmap_region(pmap->ttep, va, num_mappings, sptm_flags);
3270 pmap_retype_epoch_exit();
3271
3272 sptm_pte_t *prev_ptes = PERCPU_GET(pmap_sptm_percpu)->sptm_prev_ptes;
3273 for (unsigned int i = 0; i < num_mappings; ++i, ++cpte) {
3274 const pt_entry_t prev_pte = prev_ptes[i];
3275
3276 if (pte_is_compressed(prev_pte, cpte)) {
3277 if (options & PMAP_OPTIONS_REMOVE) {
3278 ++num_compressed;
3279 if (prev_pte & ARM_PTE_COMPRESSED_ALT) {
3280 ++num_alt_compressed;
3281 }
3282 }
3283 pai_list[i] = INVALID_PAI;
3284 continue;
3285 } else if ((prev_pte & ARM_PTE_TYPE_MASK) == ARM_PTE_TYPE_FAULT) {
3286 pai_list[i] = INVALID_PAI;
3287 continue;
3288 }
3289
3290 if (pte_is_wired(prev_pte)) {
3291 num_unwired++;
3292 }
3293
3294 const pmap_paddr_t pa = pte_to_pa(prev_pte);
3295
3296 if (__improbable(!pa_valid(pa))) {
3297 pai_list[i] = INVALID_PAI;
3298 continue;
3299 }
3300 pai_list[i] = pa_index(pa);
3301 }
3302
3303 enable_preemption();
3304 cpte -= num_mappings;
3305
3306 for (unsigned int i = 0; i < num_mappings; ++i, ++cpte) {
3307 if (pai_list[i] == INVALID_PAI) {
3308 continue;
3309 }
3310 locked_pvh_t locked_pvh;
3311 if (__improbable(options & PMAP_OPTIONS_NOPREEMPT)) {
3312 locked_pvh = pvh_lock_nopreempt(pai_list[i]);
3313 } else {
3314 locked_pvh = pvh_lock(pai_list[i]);
3315 }
3316
3317 bool is_internal, is_altacct;
3318 pv_remove_return_t remove_status = pmap_remove_pv(pmap, cpte, &locked_pvh, &is_internal, &is_altacct);
3319
3320 switch (remove_status) {
3321 case PV_REMOVE_SUCCESS:
3322 ++num_removed;
3323 if (is_altacct) {
3324 assert(is_internal);
3325 num_internal++;
3326 num_alt_internal++;
3327 } else if (is_internal) {
3328 if (ppattr_test_reusable(pai_list[i])) {
3329 num_reusable++;
3330 } else {
3331 num_internal++;
3332 }
3333 } else {
3334 num_external++;
3335 }
3336 break;
3337 default:
3338 /*
3339 * PVE already removed; this can happen due to a concurrent pmap_disconnect()
3340 * executing before we grabbed the PVH lock.
3341 */
3342 break;
3343 }
3344
3345 pvh_unlock(&locked_pvh);
3346 }
3347
3348 va += (num_mappings << pmap_page_shift);
3349 }
3350
3351 if (__improbable(need_strong_sync)) {
3352 arm64_sync_tlb(true);
3353 }
3354
3355 /*
3356 * Update the counts
3357 */
3358 pmap_ledger_debit(pmap, task_ledgers.phys_mem, num_removed * pmap_page_size);
3359
3360 if (pmap != kernel_pmap) {
3361 if (num_unwired != 0) {
3362 ptd_info_t * const ptd_info = ptep_get_info(cpte - 1);
3363 if (__improbable(os_atomic_sub_orig(&ptd_info->wiredcnt, num_unwired, relaxed) < num_unwired)) {
3364 panic("%s: pmap %p VA [0x%llx, 0x%llx) (ptd info %p) wired count underflow", __func__, pmap,
3365 (unsigned long long)start, (unsigned long long)end, ptd_info);
3366 }
3367 }
3368
3369 /* update ledgers */
3370 pmap_ledger_debit(pmap, task_ledgers.external, (num_external) * pmap_page_size);
3371 pmap_ledger_debit(pmap, task_ledgers.reusable, (num_reusable) * pmap_page_size);
3372 pmap_ledger_debit(pmap, task_ledgers.wired_mem, (num_unwired) * pmap_page_size);
3373 pmap_ledger_debit(pmap, task_ledgers.internal, (num_internal) * pmap_page_size);
3374 pmap_ledger_debit(pmap, task_ledgers.alternate_accounting, (num_alt_internal) * pmap_page_size);
3375 pmap_ledger_debit(pmap, task_ledgers.alternate_accounting_compressed, (num_alt_compressed) * pmap_page_size);
3376 pmap_ledger_debit(pmap, task_ledgers.internal_compressed, (num_compressed) * pmap_page_size);
3377 /* make needed adjustments to phys_footprint */
3378 pmap_ledger_debit(pmap, task_ledgers.phys_footprint,
3379 ((num_internal -
3380 num_alt_internal) +
3381 (num_compressed -
3382 num_alt_compressed)) * pmap_page_size);
3383 }
3384 }
3385
3386
3387 /*
3388 * Remove the given range of addresses
3389 * from the specified map.
3390 *
3391 * It is assumed that the start and end are properly
3392 * rounded to the hardware page size.
3393 */
3394 void
3395 pmap_remove(
3396 pmap_t pmap,
3397 vm_map_address_t start,
3398 vm_map_address_t end)
3399 {
3400 pmap_remove_options(pmap, start, end, PMAP_OPTIONS_REMOVE);
3401 }
3402
3403 MARK_AS_PMAP_TEXT vm_map_address_t
3404 pmap_remove_options_internal(
3405 pmap_t pmap,
3406 vm_map_address_t start,
3407 vm_map_address_t end,
3408 int options)
3409 {
3410 vm_map_address_t eva = end;
3411 tt_entry_t *tte_p;
3412 bool unlock = true;
3413
3414 if (__improbable(end < start)) {
3415 panic("%s: invalid address range %p, %p", __func__, (void*)start, (void*)end);
3416 }
3417 if (__improbable(pmap->type == PMAP_TYPE_COMMPAGE)) {
3418 panic("%s: attempt to remove mappings from commpage pmap %p", __func__, pmap);
3419 }
3420
3421 validate_pmap_mutable(pmap);
3422
3423 __unused const pt_attr_t * const pt_attr = pmap_get_pt_attr(pmap);
3424
3425 pmap_lock_mode_t lock_mode = PMAP_LOCK_SHARED;
3426 pmap_lock(pmap, lock_mode);
3427
3428 tte_p = pmap_tte(pmap, start);
3429
3430 if ((tte_p == NULL) || ((*tte_p & ARM_TTE_TYPE_MASK) == ARM_TTE_TYPE_FAULT)) {
3431 goto done;
3432 }
3433
3434 assertf((*tte_p & ARM_TTE_TYPE_MASK) == ARM_TTE_TYPE_TABLE, "%s: invalid TTE %p (0x%llx) for pmap %p va 0x%llx",
3435 __func__, tte_p, (unsigned long long)*tte_p, pmap, (unsigned long long)start);
3436
3437 pmap_remove_range_options(pmap, start, end, options);
3438
3439 if (pmap->type != PMAP_TYPE_USER) {
3440 goto done;
3441 }
3442
3443 uint16_t refcnt = sptm_get_page_table_refcnt(tte_to_pa(*tte_p));
3444 if (__improbable(refcnt == 0)) {
3445 ptd_info_t *ptd_info = ptep_get_info((pt_entry_t*)ttetokv(*tte_p));
3446 os_atomic_inc(&ptd_info->wiredcnt, relaxed); // Prevent someone else from freeing the table if we need to drop the lock
3447 if (!pmap_lock_shared_to_exclusive(pmap)) {
3448 pmap_lock(pmap, PMAP_LOCK_EXCLUSIVE);
3449 }
3450 lock_mode = PMAP_LOCK_EXCLUSIVE;
3451 refcnt = sptm_get_page_table_refcnt(tte_to_pa(*tte_p));
3452 if ((os_atomic_dec(&ptd_info->wiredcnt, relaxed) == 0) && (refcnt == 0)) {
3453 /**
3454 * Drain any concurrent retype-sensitive SPTM operations. This is needed to
3455 * ensure that we don't unmap the page table and retype it while those operations
3456 * are still finishing on other CPUs, leading to an SPTM violation. In particular,
3457 * the multipage batched cacheability/attribute update code may issue SPTM calls
3458 * without holding the relevant PVH or pmap locks, so we can't guarantee those
3459 * calls have actually completed despite observing refcnt == 0.
3460 *
3461 * At this point, we CAN guarantee that:
3462 * 1) All prior PTE removals required to produce refcnt == 0 have
3463 * completed and been synchronized for all observers by DSB, and the
3464 * relevant PV list entries removed. Subsequent calls not already in the
3465 * retype epoch will no longer observe these mappings.
3466 * 2) We now hold the pmap lock exclusive, so there will be no further attempt
3467 * to enter mappings in this page table before it is unmapped.
3468 */
3469 pmap_retype_epoch_prepare_drain();
3470 pmap_retype_epoch_drain();
3471 pmap_tte_deallocate(pmap, start, tte_p, pt_attr_twig_level(pt_attr));
3472 unlock = false; // pmap_tte_deallocate() has dropped the lock
3473 }
3474 }
3475 done:
3476 if (unlock) {
3477 pmap_unlock(pmap, lock_mode);
3478 }
3479
3480 return eva;
3481 }
3482
3483 void
3484 pmap_remove_options(
3485 pmap_t pmap,
3486 vm_map_address_t start,
3487 vm_map_address_t end,
3488 int options)
3489 {
3490 vm_map_address_t va;
3491
3492 if (pmap == PMAP_NULL) {
3493 return;
3494 }
3495
3496 __unused const pt_attr_t * const pt_attr = pmap_get_pt_attr(pmap);
3497
3498 PMAP_TRACE(2, PMAP_CODE(PMAP__REMOVE) | DBG_FUNC_START,
3499 VM_KERNEL_ADDRHIDE(pmap), VM_KERNEL_ADDRHIDE(start),
3500 VM_KERNEL_ADDRHIDE(end));
3501
3502 #if MACH_ASSERT
3503 if ((start | end) & pt_attr_leaf_offmask(pt_attr)) {
3504 panic("pmap_remove_options() pmap %p start 0x%llx end 0x%llx",
3505 pmap, (uint64_t)start, (uint64_t)end);
3506 }
3507 if ((end < start) || (start < pmap->min) || (end > pmap->max)) {
3508 panic("pmap_remove_options(): invalid address range, pmap=%p, start=0x%llx, end=0x%llx",
3509 pmap, (uint64_t)start, (uint64_t)end);
3510 }
3511 #endif
3512
3513 /*
3514 * We allow single-page requests to execute non-preemptibly,
3515 * as it doesn't make sense to sample AST_URGENT for a single-page
3516 * operation, and there are a couple of special use cases that
3517 * require a non-preemptible single-page operation.
3518 */
3519 if ((end - start) > (pt_attr_page_size(pt_attr) * PAGE_RATIO)) {
3520 pmap_verify_preemptible();
3521 }
3522
3523 /*
3524 * Invalidate the translation buffer first
3525 */
3526 va = start;
3527 while (va < end) {
3528 vm_map_address_t l;
3529
3530 l = ((va + pt_attr_twig_size(pt_attr)) & ~pt_attr_twig_offmask(pt_attr));
3531 if (l > end) {
3532 l = end;
3533 }
3534
3535 va = pmap_remove_options_internal(pmap, va, l, options);
3536 }
3537
3538 PMAP_TRACE(2, PMAP_CODE(PMAP__REMOVE) | DBG_FUNC_END);
3539 }
3540
3541
3542 /*
3543 * Remove phys addr if mapped in specified map
3544 */
3545 void
3546 pmap_remove_some_phys(
3547 __unused pmap_t map,
3548 __unused ppnum_t pn)
3549 {
3550 /* Implement to support working set code */
3551 }
3552
3553 /*
3554 * Implementation of PMAP_SWITCH_USER that Mach VM uses to
3555 * switch a thread onto a new vm_map.
3556 */
3557 void
3558 pmap_switch_user(thread_t thread, vm_map_t new_map)
3559 {
3560 pmap_t new_pmap = new_map->pmap;
3561
3562
3563 thread->map = new_map;
3564 pmap_set_pmap(new_pmap, thread);
3565
3566 }
3567 void
3568 pmap_set_pmap(
3569 pmap_t pmap,
3570 __unused thread_t thread)
3571 {
3572 pmap_switch(pmap);
3573 }
3574
3575 MARK_AS_PMAP_TEXT void
3576 pmap_switch_internal(
3577 pmap_t pmap)
3578 {
3579 validate_pmap_mutable(pmap);
3580 __unused const pt_attr_t * const pt_attr = pmap_get_pt_attr(pmap);
3581 const uint16_t asid_index = PMAP_HWASID(pmap);
3582 if (__improbable((asid_index == 0) && (pmap != kernel_pmap))) {
3583 panic("%s: attempt to activate pmap with invalid ASID %p", __func__, pmap);
3584 }
3585
3586 #if __ARM_KERNEL_PROTECT__
3587 asid_index >>= 1;
3588 #endif
3589
3590 if (asid_index > 0) {
3591 pmap_update_plru(asid_index);
3592 }
3593
3594 __unused const sptm_return_t sptm_return = sptm_switch_root(pmap->ttep);
3595
3596 #if DEVELOPMENT || DEBUG
3597 if (__improbable(sptm_return & SPTM_SWITCH_ASID_TLBI_FLUSH)) {
3598 os_atomic_inc(&pmap_asid_flushes, relaxed);
3599 }
3600
3601 if (__improbable(sptm_return & SPTM_SWITCH_RCTX_FLUSH)) {
3602 os_atomic_inc(&pmap_speculation_restrictions, relaxed);
3603 }
3604 #endif /* DEVELOPMENT || DEBUG */
3605 }
3606
3607 void
3608 pmap_switch(
3609 pmap_t pmap)
3610 {
3611 PMAP_TRACE(1, PMAP_CODE(PMAP__SWITCH) | DBG_FUNC_START, VM_KERNEL_ADDRHIDE(pmap), PMAP_VASID(pmap), PMAP_HWASID(pmap));
3612 pmap_switch_internal(pmap);
3613 PMAP_TRACE(1, PMAP_CODE(PMAP__SWITCH) | DBG_FUNC_END);
3614 }
3615
3616 void
3617 pmap_page_protect(
3618 ppnum_t ppnum,
3619 vm_prot_t prot)
3620 {
3621 pmap_page_protect_options(ppnum, prot, 0, NULL);
3622 }
3623
3624 /**
3625 * Helper function for performing per-mapping accounting following an SPTM disjoint unmap request.
3626 *
3627 * @note [pmap] cannot be the kernel pmap. This is because we do not maintain a ledger in the
3628 * kernel pmap.
3629 *
3630 * @param pmap The pmap that contained the mapping
3631 * @param pai The physical page index mapped by the mapping
3632 * @param is_compressed Indicates whether the operation was an unmap-to-compress vs. a full unmap
3633 * @param is_internal Indicates whether the mapping was for an internal (aka anonymous) VM page
3634 * @param is_altacct Indicates whether the mapping was subject to alternate accounting.
3635 */
3636 static void
3637 pmap_disjoint_unmap_accounting(pmap_t pmap, unsigned int pai, bool is_compressed, bool is_internal, bool is_altacct)
3638 {
3639 const pt_attr_t *const pt_attr = pmap_get_pt_attr(pmap);
3640 pvh_assert_locked(pai);
3641
3642 assert(pmap != kernel_pmap);
3643
3644 if (is_internal &&
3645 !is_altacct &&
3646 ppattr_test_reusable(pai)) {
3647 pmap_ledger_debit(pmap, task_ledgers.reusable, pt_attr_page_size(pt_attr) * PAGE_RATIO);
3648 } else if (!is_internal) {
3649 pmap_ledger_debit(pmap, task_ledgers.external, pt_attr_page_size(pt_attr) * PAGE_RATIO);
3650 }
3651
3652 if (is_altacct) {
3653 assert(is_internal);
3654 pmap_ledger_debit(pmap, task_ledgers.internal, pt_attr_page_size(pt_attr) * PAGE_RATIO);
3655 pmap_ledger_debit(pmap, task_ledgers.alternate_accounting, pt_attr_page_size(pt_attr) * PAGE_RATIO);
3656 if (is_compressed) {
3657 pmap_ledger_credit(pmap, task_ledgers.internal_compressed, pt_attr_page_size(pt_attr) * PAGE_RATIO);
3658 pmap_ledger_credit(pmap, task_ledgers.alternate_accounting_compressed, pt_attr_page_size(pt_attr) * PAGE_RATIO);
3659 }
3660 } else if (ppattr_test_reusable(pai)) {
3661 assert(is_internal);
3662 if (is_compressed) {
3663 pmap_ledger_credit(pmap, task_ledgers.internal_compressed, pt_attr_page_size(pt_attr) * PAGE_RATIO);
3664 /* was not in footprint, but is now */
3665 pmap_ledger_credit(pmap, task_ledgers.phys_footprint, pt_attr_page_size(pt_attr) * PAGE_RATIO);
3666 }
3667 } else if (is_internal) {
3668 pmap_ledger_debit(pmap, task_ledgers.internal, pt_attr_page_size(pt_attr) * PAGE_RATIO);
3669
3670 /*
3671 * Update all stats related to physical footprint, which only
3672 * deals with internal pages.
3673 */
3674 if (is_compressed) {
3675 /*
3676 * This removal is only being done so we can send this page to
3677 * the compressor; therefore it mustn't affect total task footprint.
3678 */
3679 pmap_ledger_credit(pmap, task_ledgers.internal_compressed, pt_attr_page_size(pt_attr) * PAGE_RATIO);
3680 } else {
3681 /*
3682 * This internal page isn't going to the compressor, so adjust stats to keep
3683 * phys_footprint up to date.
3684 */
3685 pmap_ledger_debit(pmap, task_ledgers.phys_footprint, pt_attr_page_size(pt_attr) * PAGE_RATIO);
3686 }
3687 } else {
3688 /* external page: no impact on ledgers */
3689 }
3690 }
3691
3692 /**
3693 * Helper function for issuing a disjoint unmap request to the SPTM and performing
3694 * related accounting. This function uses the 'prev_ptes' list generated by
3695 * the sptm_unmap_disjoint() call to determine whether said call altered the
3696 * relevant PTEs in a manner that would require accounting updates.
3697 *
3698 * @param pa The physical address against which the disjoint unmap will be issued.
3699 * @param num_mappings The number of disjoint mappings for the SPTM to update.
3700 * The per-CPU sptm_ops array should contain the same number
3701 * of individual disjoint requests.
3702 */
3703 static void
3704 pmap_disjoint_unmap(pmap_paddr_t pa, unsigned int num_mappings)
3705 {
3706 const unsigned int pai = pa_index(pa);
3707
3708 pvh_assert_locked(pai);
3709
3710 assert(num_mappings <= SPTM_MAPPING_LIMIT);
3711
3712 assert(get_preemption_level() > 0);
3713 pmap_sptm_percpu_data_t *sptm_pcpu = PERCPU_GET(pmap_sptm_percpu);
3714
3715 sptm_unmap_disjoint(pa, sptm_pcpu->sptm_ops_pa, num_mappings);
3716
3717 for (unsigned int cur_mapping = 0; cur_mapping < num_mappings; ++cur_mapping) {
3718 pt_entry_t prev_pte = sptm_pcpu->sptm_prev_ptes[cur_mapping];
3719
3720 pt_desc_t * const ptdp = sptm_pcpu->sptm_ptds[cur_mapping];
3721 const pmap_t pmap = ptdp->pmap;
3722
3723 assertf(((prev_pte & ARM_PTE_TYPE_MASK) == ARM_PTE_TYPE_FAULT) ||
3724 ((pte_to_pa(prev_pte) & ~PAGE_MASK) == pa), "%s: prev_pte 0x%llx does not map pa 0x%llx",
3725 __func__, (unsigned long long)prev_pte, (unsigned long long)pa);
3726
3727 const pt_attr_t *const pt_attr = pmap_get_pt_attr(pmap);
3728 pmap_ledger_debit(pmap, task_ledgers.phys_mem, pt_attr_page_size(pt_attr) * PAGE_RATIO);
3729
3730 if (pmap != kernel_pmap) {
3731 /*
3732 * If the prior PTE is invalid (which may happen due to a concurrent remove operation),
3733 * the compressed marker won't be written so we shouldn't account the mapping as compressed.
3734 */
3735 const bool is_compressed = (((prev_pte & ARM_PTE_TYPE_MASK) != ARM_PTE_TYPE_FAULT) &&
3736 ((sptm_pcpu->sptm_ops[cur_mapping].pte_template & ARM_PTE_COMPRESSED_MASK) != 0));
3737 const bool is_internal = (sptm_pcpu->sptm_acct_flags[cur_mapping] & PMAP_SPTM_FLAG_INTERNAL) != 0;
3738 const bool is_altacct = (sptm_pcpu->sptm_acct_flags[cur_mapping] & PMAP_SPTM_FLAG_ALTACCT) != 0;
3739
3740 /*
3741 * The rule is that accounting related to PTE contents (wired, PTD refcount)
3742 * must be updated by whoever clears the PTE, while accounting related to physical page
3743 * attributes must be updated by whoever clears the PVE. We therefore always call
3744 * pmap_disjoint_unmap_accounting() here since we're removing the PVE, but only update
3745 * wired/PTD accounting if the prior PTE was valid.
3746 */
3747 pmap_disjoint_unmap_accounting(pmap, pai, is_compressed, is_internal, is_altacct);
3748
3749 if ((prev_pte & ARM_PTE_TYPE_MASK) == ARM_PTE_TYPE_FAULT) {
3750 continue;
3751 }
3752
3753 if (pte_is_wired(prev_pte)) {
3754 pmap_ledger_debit(pmap, task_ledgers.wired_mem, pt_attr_page_size(pt_attr) * PAGE_RATIO);
3755 if (__improbable(os_atomic_dec_orig(&sptm_pcpu->sptm_ptd_info[cur_mapping]->wiredcnt, relaxed) == 0)) {
3756 panic("%s: over-unwire of ptdp %p, ptd info %p", __func__,
3757 ptdp, sptm_pcpu->sptm_ptd_info[cur_mapping]);
3758 }
3759 }
3760 }
3761 }
3762 }
3763
3764 /**
3765 * The following two functions, pmap_multipage_op_submit_disjoint() and
3766 * pmap_multipage_op_add_page(), are intended to allow callers to manage batched SPTM
3767 * operations that may span multiple physical pages. They are intended to operate in
3768 * a way that allows callers such as pmap_page_protect_options_with_flush_range() to
3769 * insert mappings into the per-CPU SPTM disjoint ops array in the same manner that
3770 * they would for an ordinary single-page operation.
3771 * Functions such as pmap_page_protect_options_with_flush_range() operate on a single
3772 * physical page but may be passed a non-NULL flush_range object to indicate that the
3773 * call is part of a larger batched operation which may span multiple physical pages.
3774 * In that scenario, these functions are intended to be used as follows:
3775 * 1) Call pmap_multipage_op_add_page() to insert a "header" for the page into the per-
3776 * CPU SPTM ops array. Use the return value from this call as the starting index
3777 * at which to add ordinary mapping entries into the same array.
3778 * 2) Insert sptm_disjoint_op_t entries into the ops array in the normal manner until
3779 * the array is full, the SPTM options required for the upcoming sequence of pages
3780 * need to change, or the current mapping matches flush_range->current_ptep.
3781 * In the latter case, pmap_insert_flush_range_template() may instead be used
3782 * to insert the mapping into the per-CPU SPTM region templates array. See the
3783 * documentation for pmap_insert_flush_range_template() below.
3784 * 3) If the array is full, call pmap_multipage_op_submit_disjoint() and return to step 1).
3785 * 4) If the SPTM options need to change, call pmap_multipage_op_add_page() to insert
3786 * a new header with the updated options and, using the return value as the new
3787 * insertion point for the ops array, resume step 2).
3788 * 5) Upon completion, if there are any pending not-yet-submitted mappings, do not
3789 * submit those mappings to the SPTM as would ordinarily be done for a single-page
3790 * call. These trailing mappings will be submitted as part of the next batch,
3791 * or by the next-higher caller if the range operation is complete.
3792 *
3793 * Note that, as a performance optimization, the caller may track the insertion
3794 * point in the disjoint ops array locally (i.e. without incrementing
3795 * flush_range->pending_disjoint_entries on every iteration, as long as it takes care to do the
3796 * following:
3797 * 1) Initialize and update that insertion point as described in steps 1) and 4) above.
3798 * 2) Pass the updated insertion point as the 'pending_disjoint_entries' parameter into the calls
3799 * in steps 3) and 4) above.
3800 * 3) Update flush_range->pending_disjoint_entries with the locally-maintained value along with
3801 * step 5) above.
3802 */
3803
3804 /**
3805 * Submit any pending disjoint multi-page mapping updates to the SPTM.
3806 *
3807 * @note This function must be called with preemption disabled, and will drop
3808 * the preemption-disable count upon submitting to the SPTM.
3809 * @note [pending_disjoint_entries] must include *all* pending entries in the SPTM ops array,
3810 * including physical address "header" entries.
3811 * @note This function automatically updates the per_paddr_header.num_mappings field
3812 * for the most recent physical address header in the SPTM ops array to its final
3813 * value.
3814 *
3815 * @param pending_disjoint_entries The number of not-yet-submitted mappings according to the caller.
3816 * This value may be greater than [flush_range]->pending_disjoint_entries if
3817 * the caller has inserted mappings into the ops array without
3818 * updating [flush_range]->pending_disjoint_entries, in which case this
3819 * function will update [flush_range]->pending_disjoint_entries with the
3820 * caller's value.
3821 * @param flush_range The object tracking the current state of the multipage disjoint
3822 * operation.
3823 */
3824 static inline void
3825 pmap_multipage_op_submit_disjoint(unsigned int pending_disjoint_entries, pmap_tlb_flush_range_t *flush_range)
3826 {
3827 /**
3828 * Reconcile the number of pending entries as tracked by the caller with the
3829 * number of pending entries tracked by flush_range. If the caller's value is
3830 * greater, we assume the caller has inserted locally-tracked mappings into the
3831 * array without directly updating flush_range->pending_disjoint_entries. Otherwise, we
3832 * assume the caller has no locally-tracked mappings and is simply trying to
3833 * purge any pending mappings from a prior call sequence.
3834 */
3835 if (pending_disjoint_entries > flush_range->pending_disjoint_entries) {
3836 flush_range->pending_disjoint_entries = pending_disjoint_entries;
3837 } else {
3838 assert(pending_disjoint_entries == 0);
3839 }
3840 if (flush_range->pending_disjoint_entries != 0) {
3841 assert(get_preemption_level() > 0);
3842 /**
3843 * Compute the correct number of mappings for the most recent paddr
3844 * header based on the current position in the SPTM ops array.
3845 */
3846 flush_range->current_header->per_paddr_header.num_mappings =
3847 flush_range->pending_disjoint_entries - flush_range->current_header_first_mapping_index;
3848 const sptm_return_t sptm_return = sptm_update_disjoint_multipage(
3849 PERCPU_GET(pmap_sptm_percpu)->sptm_ops_pa, flush_range->pending_disjoint_entries);
3850
3851 /**
3852 * We may be submitting the batch and exiting the epoch partway through
3853 * processing the PV list for a page. That's fine, because in that case we'll
3854 * hold the PV lock for that page, which will prevent mappings of that page from
3855 * being disconnected and will prevent the completion of pmap_remove() against
3856 * any of those mappings, thus also guaranteeing the relevant page table pages
3857 * can't be freed. The epoch still protects mappings for any prior page in
3858 * the batch, whose PV locks are no longer held.
3859 */
3860 pmap_retype_epoch_exit();
3861 enable_preemption();
3862 if (flush_range->pending_region_entries != 0) {
3863 flush_range->processed_entries += flush_range->pending_disjoint_entries;
3864 } else {
3865 flush_range->processed_entries = 0;
3866 }
3867 flush_range->pending_disjoint_entries = 0;
3868 if (sptm_return == SPTM_UPDATE_DELAYED_TLBI) {
3869 flush_range->ptfr_flush_needed = true;
3870 }
3871 }
3872 }
3873
3874 /**
3875 * Insert a new physical address "header" entry into the per-CPU SPTM ops array for a
3876 * multi-page SPTM operation. It is expected that the caller will subsequently add
3877 * mapping entries for this physical address into the array.
3878 *
3879 * @note This function will disable preemption upon creation of the first paddr header
3880 * (index 0 in the per-CPU SPTM ops array) and it is expected that
3881 * pmap_multipage_op_submit() will subsequently be called on the same CPU.
3882 * @note Before inserting the new header, this function automatically updates the
3883 * per_paddr_header.num_mappings field for the previous physical address header
3884 * (if present) in the SPTM ops array to its final value.
3885 *
3886 * @param phys The physical address for which to insert a header entry.
3887 * @param inout_pending_disjoint_entries
3888 * [input] The number of not-yet-submitted mappings according to the caller.
3889 * This value may be greater than [flush_range]->pending_disjoint_entries if
3890 * the caller has inserted mappings into the ops array without
3891 * updating [flush_range]->pending_disjoint_entries, in which case this
3892 * function will update [flush_range]->pending_disjoint_entries with the
3893 * caller's value.
3894 * [output] Returns the starting index at which the caller should insert mapping
3895 * entries into the per-CPU SPTM ops array.
3896 * @param sptm_update_options SPTM_UPDATE_* flags to pass to the SPTM call.
3897 * SPTM_UPDATE_SKIP_PAPT is automatically inserted by this
3898 * function.
3899 * @param flush_range The object tracking the current state of the multipage operation.
3900 *
3901 * @return True if the region operation was submitted to the SPTM due to the ops array already
3902 * being full, false otherwise. In the former case, the new header will not be added
3903 * to the array; the caller will need to re-invoke this function after taking any
3904 * necessary post-submission action (such as enabling preemption).
3905 */
3906 static inline bool
3907 pmap_multipage_op_add_page(
3908 pmap_paddr_t phys,
3909 unsigned int *inout_pending_disjoint_entries,
3910 uint32_t sptm_update_options,
3911 pmap_tlb_flush_range_t *flush_range)
3912 {
3913 unsigned int pending_disjoint_entries = *inout_pending_disjoint_entries;
3914
3915 /**
3916 * Reconcile the number of pending entries as tracked by the caller with the
3917 * number of pending entries tracked by flush_range. If the caller's value is
3918 * greater, we assume the caller has inserted locally-tracked mappings into the
3919 * array without directly updating flush_range->pending_disjoint_entries. Otherwise, we
3920 * assume the caller has no locally-tracked mappings and is adding its paddr
3921 * header for the first time.
3922 */
3923 if (pending_disjoint_entries > flush_range->pending_disjoint_entries) {
3924 flush_range->pending_disjoint_entries = pending_disjoint_entries;
3925 } else {
3926 assert(pending_disjoint_entries == 0);
3927 }
3928 if (flush_range->pending_disjoint_entries >= (SPTM_MAPPING_LIMIT - 1)) {
3929 /**
3930 * If the SPTM ops array is either full or only has space for the paddr
3931 * header, there won't be room for mapping entries, so submit the pending
3932 * mappings to the SPTM now, and return to allow the caller to take
3933 * any necessary post-submission action.
3934 */
3935 pmap_multipage_op_submit_disjoint(pending_disjoint_entries, flush_range);
3936 *inout_pending_disjoint_entries = 0;
3937 return true;
3938 }
3939 pending_disjoint_entries = flush_range->pending_disjoint_entries;
3940
3941 sptm_update_options |= SPTM_UPDATE_SKIP_PAPT;
3942 if (pending_disjoint_entries == 0) {
3943 disable_preemption();
3944 /**
3945 * Enter the retype epoch while we gather the disjoint update arguments
3946 * and issue the SPTM call. Since this operation may cover multiple physical
3947 * pages, we may construct the argument array and invoke the SPTM without holding
3948 * all relevant PVH locks or pmap locks. We therefore need to record that we are
3949 * collecting and modifying mapping state so that e.g. pmap_page_protect() does
3950 * not attempt to retype the underlying pages and pmap_remove() does not attempt
3951 * to free the page tables used for these mappings without first draining our epoch.
3952 */
3953 pmap_retype_epoch_enter();
3954 flush_range->pending_disjoint_entries = 1;
3955 } else {
3956 /**
3957 * Before inserting the new header, update the prior header's number
3958 * of paddr-specific mappings to its final value.
3959 */
3960 assert(flush_range->current_header != NULL);
3961 flush_range->current_header->per_paddr_header.num_mappings =
3962 pending_disjoint_entries - flush_range->current_header_first_mapping_index;
3963 }
3964 sptm_disjoint_op_t *sptm_ops = PERCPU_GET(pmap_sptm_percpu)->sptm_ops;
3965 flush_range->current_header = (sptm_update_disjoint_multipage_op_t*)&sptm_ops[pending_disjoint_entries];
3966 flush_range->current_header_first_mapping_index = ++pending_disjoint_entries;
3967 flush_range->current_header->per_paddr_header.paddr = phys;
3968 flush_range->current_header->per_paddr_header.num_mappings = 0;
3969 flush_range->current_header->per_paddr_header.options = sptm_update_options;
3970
3971 *inout_pending_disjoint_entries = pending_disjoint_entries;
3972 return false;
3973 }
3974
3975 /**
3976 * The following two functions, pmap_multipage_op_submit_region() and
3977 * pmap_insert_flush_range_template(), are meant to be used in a similar fashion
3978 * to pmap_multipage_op_submit_disjoint() and pmap_multipage_op_add_page(),
3979 * but for the specific case in which a given mapping within a PV list happens
3980 * to map the current VA within a VA region being operated on by
3981 * phys_attribute_clear_range(). This allows the pmap to further optimize
3982 * the SPTM calls by using sptm_update_region() to modify all mappings within
3983 * the VA region, which requires far fewer table walks than a disjoint operation.
3984 * Since the starting VA of the region, the owning pmap, and the insertion point
3985 * within the per-CPU region templates array are already known, these functions
3986 * don't require the special "header" entry or the complex array position tracking
3987 * of their disjoint equivalents above.
3988 * Note that these functions may be used together with the disjoint functions above;
3989 * these functions can be used for the "primary" mappings corresponding to the VA
3990 * region being manipulated by the VM layer, while the disjoint functions can be
3991 * used for any alias mappings of the underlying pages which fall outside that
3992 * VA region.
3993 */
3994
3995 /**
3996 * Submit any pending region-based templates for the specified flush_range.
3997 *
3998 * @note This function must be called with preemption disabled, and will drop
3999 * the preemption-disable count upon submitting to the SPTM.
4000 *
4001 * @param flush_range The object tracking the current state of the region operation.
4002 */
4003 static inline void
4004 pmap_multipage_op_submit_region(pmap_tlb_flush_range_t *flush_range)
4005 {
4006 if (flush_range->pending_region_entries != 0) {
4007 assert(get_preemption_level() > 0);
4008 pmap_assert_locked(flush_range->ptfr_pmap, PMAP_LOCK_SHARED);
4009 /**
4010 * If there are any pending disjoint entries, we're already in a retype epoch.
4011 * For disjoint entries, we need to hold the epoch during the entire time we
4012 * construct the disjoint ops array because those ops may point to some arbitrary
4013 * pmap and we need to ensure the relevant page tables and even the pmap itself
4014 * aren't concurrently reclaimed while our ops array points to them.
4015 * But for a region op like this, we know we already hold the relevant pmap lock
4016 * so none of the above can happen concurrently. We therefore only need to hold
4017 * the epoch across the SPTM call itself to prevent a concurrent unmap operation
4018 * from attempting to retype the mapped pages while our SPTM call has them in-
4019 * flight.
4020 */
4021 if (flush_range->pending_disjoint_entries == 0) {
4022 pmap_retype_epoch_enter();
4023 }
4024 const sptm_return_t sptm_return = sptm_update_region(flush_range->ptfr_pmap->ttep,
4025 flush_range->pending_region_start, flush_range->pending_region_entries,
4026 PERCPU_GET(pmap_sptm_percpu)->sptm_templates_pa,
4027 SPTM_UPDATE_PERMS_AND_WAS_WRITABLE | SPTM_UPDATE_AF | SPTM_UPDATE_DEFER_TLBI);
4028 if (flush_range->pending_disjoint_entries == 0) {
4029 pmap_retype_epoch_exit();
4030 }
4031 enable_preemption();
4032 if (flush_range->pending_disjoint_entries != 0) {
4033 flush_range->processed_entries += flush_range->pending_region_entries;
4034 } else {
4035 flush_range->processed_entries = 0;
4036 }
4037 flush_range->pending_region_start += (flush_range->pending_region_entries <<
4038 pmap_get_pt_attr(flush_range->ptfr_pmap)->pta_page_shift);
4039 flush_range->pending_region_entries = 0;
4040 if (sptm_return == SPTM_UPDATE_DELAYED_TLBI) {
4041 flush_range->ptfr_flush_needed = true;
4042 }
4043 }
4044 }
4045
4046 /**
4047 * Insert a PTE template into the per-CPU SPTM region ops array.
4048 * This is meant to be used as a performance optimization for the case in which a given
4049 * mapping being processed by a function such as pmap_page_protect_options_with_flush_range()
4050 * happens to map the current iteration position within [flush_range]'s VA region.
4051 * In this case the mapping can be inserted as a region-based template rather than a disjoint
4052 * operation as would be done in the general case. The idea is that region-based SPTM
4053 * operations are significantly less expensive than disjoint operations, because each region
4054 * operation only requires a single page table walk at the beginning vs. a table walk for
4055 * each mapping in the disjoint case. Since the majority of mappings processed by a flush
4056 * range operation belong to the main flush range VA region (i.e. alias mappings outside
4057 * the region are less common), the performance improvement can be significant.
4058 *
4059 * @note This function will disable preemption upon inserting the first entry into the
4060 * per-CPU templates array, and will re-enable preemption upon submitting the region
4061 * operation to the SPTM.
4062 *
4063 * @param template The PTE template to insert into the per-CPU templates array.
4064 * @param flush_range The object tracking the current state of the region operation.
4065 *
4066 * @return True if the region operation was submitted to the SPTM, false otherwise.
4067 */
4068 static inline bool
4069 pmap_insert_flush_range_template(pt_entry_t template, pmap_tlb_flush_range_t *flush_range)
4070 {
4071 if (flush_range->pending_region_entries == 0) {
4072 disable_preemption();
4073 }
4074 flush_range->region_entry_added = true;
4075 PERCPU_GET(pmap_sptm_percpu)->sptm_templates[flush_range->pending_region_entries++] = template;
4076 if (flush_range->pending_region_entries == SPTM_MAPPING_LIMIT) {
4077 pmap_multipage_op_submit_region(flush_range);
4078 return true;
4079 }
4080 return false;
4081 }
4082
4083 /**
4084 * Wrapper function for submitting any pending operations, region-based or disjoint,
4085 * tracked by a flush range object. This is meant to be used by the top-level caller that
4086 * iterates over the flush range's VA region and calls functions such as
4087 * pmap_page_protect_options_with_flush_range() or arm_force_fast_fault_with_flush_range()
4088 * to construct the relevant SPTM operations arrays.
4089 *
4090 * @param flush_range The object tracking the current state of region and/or disjoint operations.
4091 */
4092 static inline void
4093 pmap_multipage_op_submit(pmap_tlb_flush_range_t *flush_range)
4094 {
4095 pmap_multipage_op_submit_disjoint(0, flush_range);
4096 pmap_multipage_op_submit_region(flush_range);
4097 }
4098
4099 /**
4100 * This is an internal-only flag that indicates the caller of pmap_page_protect_options_with_flush_range()
4101 * is removing/updating all mappings in preparation for a retype operation. In this case
4102 * pmap_page_protect_options() will assume (and assert) that the PVH lock for the physical page is held
4103 * by the calller, and will perform the necessary retype epoch drain prior to returning.
4104 */
4105 #define PMAP_OPTIONS_PPO_PENDING_RETYPE 0x80000000
4106 _Static_assert(PMAP_OPTIONS_PPO_PENDING_RETYPE & PMAP_OPTIONS_RESERVED_MASK,
4107 "PMAP_OPTIONS_PPO_PENDING_RETYPE outside reserved encoding space");
4108
4109 /**
4110 * Lower the permission for all mappings to a given page. If VM_PROT_NONE is specified,
4111 * the mappings will be removed.
4112 *
4113 * @param ppnum Page number to lower the permission of.
4114 * @param prot The permission to lower to.
4115 * @param options PMAP_OPTIONS_NOFLUSH indicates TLBI flush is not needed.
4116 * PMAP_OPTIONS_PPO_PENDING_RETYPE indicates the PVH lock for ppnum is
4117 * already locked and a retype epoch drain shold be performed.
4118 * PMAP_OPTIONS_COMPRESSOR indicates the function is called by the
4119 * VM compressor.
4120 * @param locked_pvh If non-NULL, this indicates the PVH lock for [ppnum] is already locked
4121 * by the caller. This is an input/output parameter which may be updated
4122 * to reflect a new PV head value to be passed to a later call to pvh_unlock().
4123 * @param flush_range When present, this function will skip the TLB flush for the
4124 * mappings that are covered by the range, leaving that to be
4125 * done later by the caller. It may also avoid submitting mapping
4126 * updates directly to the SPTM, instead accumulating them in a
4127 * per-CPU array to be submitted later by the caller.
4128 *
4129 * @note PMAP_OPTIONS_NOFLUSH and flush_range cannot both be specified.
4130 */
4131 MARK_AS_PMAP_TEXT static void
4132 pmap_page_protect_options_with_flush_range(
4133 ppnum_t ppnum,
4134 vm_prot_t prot,
4135 unsigned int options,
4136 locked_pvh_t *locked_pvh,
4137 pmap_tlb_flush_range_t *flush_range)
4138 {
4139 pmap_paddr_t phys = ptoa(ppnum);
4140 locked_pvh_t local_locked_pvh = {.pvh = 0};
4141 pv_entry_t *pve_p = NULL;
4142 pv_entry_t *pveh_p = NULL;
4143 pv_entry_t *pvet_p = NULL;
4144 pt_entry_t *pte_p = NULL;
4145 pv_entry_t *new_pve_p = NULL;
4146 pt_entry_t *new_pte_p = NULL;
4147
4148 bool remove = false;
4149 unsigned int pvh_cnt = 0;
4150 unsigned int num_mappings = 0, num_skipped_mappings = 0;
4151
4152 assert(ppnum != vm_page_fictitious_addr);
4153
4154 /**
4155 * Assert that PMAP_OPTIONS_NOFLUSH and flush_range cannot both be specified.
4156 *
4157 * PMAP_OPTIONS_NOFLUSH indicates there is no need of flushing the TLB in the entire operation, and
4158 * flush_range indicates the caller requests deferral of the TLB flushing. Fundemantally, the two
4159 * semantics conflict with each other, so assert they are not both true.
4160 */
4161 assert(!(flush_range && (options & PMAP_OPTIONS_NOFLUSH)));
4162
4163 /* Only work with managed pages. */
4164 if (!pa_valid(phys)) {
4165 return;
4166 }
4167
4168 /*
4169 * Determine the new protection.
4170 */
4171 switch (prot) {
4172 case VM_PROT_ALL:
4173 return; /* nothing to do */
4174 case VM_PROT_READ:
4175 case VM_PROT_READ | VM_PROT_EXECUTE:
4176 break;
4177 default:
4178 /* PPL security model requires that we flush TLBs before we exit if the page may be recycled. */
4179 options = options & ~PMAP_OPTIONS_NOFLUSH;
4180 remove = true;
4181 break;
4182 }
4183
4184 /**
4185 * We don't support cross-page batching (indicated by flush_range being non-NULL) for removals,
4186 * as removals must use the SPTM prev_ptes array for accounting, which isn't supported for cross-
4187 * page batches.
4188 */
4189 assert((flush_range == NULL) || !remove);
4190
4191 unsigned int pai = pa_index(phys);
4192 if (__probable(locked_pvh == NULL)) {
4193 if (flush_range != NULL) {
4194 /**
4195 * If we're partway through processing a multi-page batched call,
4196 * preemption will already be disabled so we can't simply call
4197 * pvh_lock() which may block. Instead, we first try to acquire
4198 * the lock without waiting, which in most cases should succeed.
4199 * If it fails, we submit the pending batched operations to re-
4200 * enable preemption and then acquire the lock normally.
4201 */
4202 local_locked_pvh = pvh_try_lock(pai);
4203 if (__improbable(!pvh_try_lock_success(&local_locked_pvh))) {
4204 pmap_multipage_op_submit(flush_range);
4205 local_locked_pvh = pvh_lock(pai);
4206 }
4207 } else {
4208 local_locked_pvh = pvh_lock(pai);
4209 }
4210 } else {
4211 local_locked_pvh = *locked_pvh;
4212 assert(pai == local_locked_pvh.pai);
4213 }
4214 assert(local_locked_pvh.pvh != 0);
4215 pvh_assert_locked(pai);
4216
4217 bool pvh_lock_sleep_mode_needed = false;
4218
4219 /*
4220 * PVH should be locked before accessing per-CPU data, as we're relying on the lock
4221 * to disable preemption.
4222 */
4223 pmap_cpu_data_t *pmap_cpu_data = NULL;
4224 pmap_sptm_percpu_data_t *sptm_pcpu = NULL;
4225 sptm_disjoint_op_t *sptm_ops = NULL;
4226 pt_desc_t **sptm_ptds = NULL;
4227 ptd_info_t **sptm_ptd_info = NULL;
4228
4229 /* BEGIN IGNORE CODESTYLE */
4230
4231 /**
4232 * This would also work as a block, with the above variables declared using the
4233 * __block qualifier, but the extra runtime overhead of block syntax (e.g.
4234 * dereferencing __block variables through stack forwarding pointers) isn't needed
4235 * here, as we never need to use this code sequence as a closure.
4236 */
4237 #define PPO_PERCPU_INIT() do { \
4238 disable_preemption(); \
4239 pmap_cpu_data = pmap_get_cpu_data(); \
4240 sptm_pcpu = PERCPU_GET(pmap_sptm_percpu); \
4241 sptm_ops = sptm_pcpu->sptm_ops; \
4242 sptm_ptds = sptm_pcpu->sptm_ptds; \
4243 sptm_ptd_info = sptm_pcpu->sptm_ptd_info; \
4244 if (remove) { \
4245 os_atomic_store(&pmap_cpu_data->inflight_disconnect, true, relaxed); \
4246 /* \
4247 * Ensure the store to inflight_disconnect will be observed before any of the
4248 * ensuing PTE/refcount stores in this function. This flag is used to avoid
4249 * a race in which the VM may clear a pmap's mappings and destroy the pmap on
4250 * another CPU, in between this function's clearing a PTE and dropping the
4251 * corresponding pagetable refcount. That can lead to a panic if the
4252 * destroying thread observes a non-zero refcount. For this we need a store-
4253 * store barrier; a store-release operation would not be sufficient.
4254 */ \
4255 os_atomic_thread_fence(release); \
4256 } \
4257 } while (0)
4258
4259 /* END IGNORE CODESTYLE */
4260
4261
4262 PPO_PERCPU_INIT();
4263
4264 pv_entry_t **pve_pp = NULL;
4265
4266 if (pvh_test_type(local_locked_pvh.pvh, PVH_TYPE_PTEP)) {
4267 pte_p = pvh_ptep(local_locked_pvh.pvh);
4268 } else if (pvh_test_type(local_locked_pvh.pvh, PVH_TYPE_PVEP)) {
4269 pve_p = pvh_pve_list(local_locked_pvh.pvh);
4270 pveh_p = pve_p;
4271 } else if (__improbable(!pvh_test_type(local_locked_pvh.pvh, PVH_TYPE_NULL))) {
4272 panic("%s: invalid PV head 0x%llx for PA 0x%llx", __func__, (uint64_t)local_locked_pvh.pvh, (uint64_t)phys);
4273 }
4274
4275 int pve_ptep_idx = 0;
4276 const bool compress = (options & PMAP_OPTIONS_COMPRESSOR);
4277
4278 /*
4279 * We need to keep track of whether a particular PVE list contains IOMMU
4280 * mappings when removing entries, because we should only remove CPU
4281 * mappings. If a PVE list contains at least one IOMMU mapping, we keep
4282 * it around.
4283 */
4284 bool iommu_mapping_in_pve = false;
4285
4286 /**
4287 * With regard to TLBI, there are three cases:
4288 *
4289 * 1. PMAP_OPTIONS_NOFLUSH is specified. In such case, SPTM doesn't need to flush TLB and neither does pmap.
4290 * 2. PMAP_OPTIONS_NOFLUSH is not specified, but flush_range is, indicating the caller intends to flush TLB
4291 * itself (with range TLBI). In such case, we check the flush_range limits and only issue the TLBI if a
4292 * mapping is out of the range.
4293 * 3. Neither PMAP_OPTIONS_NOFLUSH nor a valid flush_range pointer is specified. In such case, we should just
4294 * let SPTM handle TLBI flushing.
4295 */
4296 const bool defer_tlbi = (options & PMAP_OPTIONS_NOFLUSH) || flush_range;
4297 const uint32_t sptm_update_options = SPTM_UPDATE_PERMS_AND_WAS_WRITABLE | (defer_tlbi ? SPTM_UPDATE_DEFER_TLBI : 0);
4298
4299 while ((pve_p != PV_ENTRY_NULL) || (pte_p != PT_ENTRY_NULL)) {
4300 if (__improbable(pvh_lock_sleep_mode_needed)) {
4301 assert((num_mappings == 0) && (num_skipped_mappings == 0));
4302 if (remove) {
4303 /**
4304 * Clear the in-flight disconnect indicator for the current CPU, as we've
4305 * already submitted any prior pending SPTM operations, and we're about to
4306 * briefly re-enable preemption which may cause this thread to be migrated.
4307 */
4308 os_atomic_store(&pmap_cpu_data->inflight_disconnect, false, release);
4309 }
4310 /**
4311 * Undo the explicit preemption disable done in the last call to PPO_PER_CPU_INIT().
4312 * If the PVH lock is placed in sleep mode, we can't rely on it to disable preemption,
4313 * so we need these explicit preemption twiddles to ensure we don't get migrated off-
4314 * core while processing SPTM per-CPU data. At the same time, we also want preemption
4315 * to briefly be re-enabled every SPTM_MAPPING_LIMIT mappings so that any pending
4316 * urgent ASTs can be handled.
4317 */
4318 enable_preemption();
4319 pvh_lock_enter_sleep_mode(&local_locked_pvh);
4320 pvh_lock_sleep_mode_needed = false;
4321 PPO_PERCPU_INIT();
4322 }
4323
4324 if (pve_p != PV_ENTRY_NULL) {
4325 pte_p = pve_get_ptep(pve_p, pve_ptep_idx);
4326 if (pte_p == PT_ENTRY_NULL) {
4327 goto protect_skip_pve;
4328 }
4329 }
4330
4331 #ifdef PVH_FLAG_IOMMU
4332 if (pvh_ptep_is_iommu(pte_p)) {
4333 iommu_mapping_in_pve = true;
4334 if (__improbable(remove && (options & PMAP_OPTIONS_COMPRESSOR))) {
4335 const iommu_instance_t iommu = ptep_get_iommu(pte_p);
4336 panic("%s: attempt to compress ppnum 0x%x owned by iommu driver "
4337 "%u (token: %#x), pve_p=%p", __func__, ppnum, GET_IOMMU_ID(iommu),
4338 GET_IOMMU_TOKEN(iommu), pve_p);
4339 }
4340 if (remove && (pve_p == PV_ENTRY_NULL)) {
4341 /*
4342 * We've found an IOMMU entry and it's the only entry in the PV list.
4343 * We don't discard IOMMU entries, so simply set up the new PV list to
4344 * contain the single IOMMU PTE and exit the loop.
4345 */
4346 new_pte_p = pte_p;
4347 break;
4348 }
4349 ++num_skipped_mappings;
4350 goto protect_skip_pve;
4351 }
4352 #endif
4353
4354 const pt_entry_t spte = os_atomic_load(pte_p, relaxed);
4355
4356 if (__improbable(!remove && ((spte & ARM_PTE_TYPE_MASK) == ARM_PTE_TYPE_FAULT))) {
4357 ++num_skipped_mappings;
4358 goto protect_skip_pve;
4359 }
4360
4361 pt_desc_t *ptdp = NULL;
4362 pmap_t pmap = NULL;
4363 vm_map_address_t va = 0;
4364
4365 if ((flush_range != NULL) && (pte_p == flush_range->current_ptep)) {
4366 /**
4367 * If the current mapping matches the flush range's current iteration position,
4368 * there's no need to do the work of getting the PTD. We already know the pmap,
4369 * and the VA is implied by flush_range->pending_region_start.
4370 */
4371 pmap = flush_range->ptfr_pmap;
4372 } else {
4373 ptdp = ptep_get_ptd(pte_p);
4374 pmap = ptdp->pmap;
4375 va = ptd_get_va(ptdp, pte_p);
4376 }
4377
4378 /**
4379 * If the PTD is NULL, we're adding the current mapping to the pending region templates instead of the
4380 * pending disjoint ops, so we don't need to do flush range disjoint op management.
4381 */
4382 if ((flush_range != NULL) && (ptdp != NULL)) {
4383 /**
4384 * Insert a "header" entry for this physical page into the SPTM disjoint ops array.
4385 * We do this in three cases:
4386 * 1) We're at the beginning of the SPTM ops array (num_mappings == 0, flush_range->pending_disjoint_entries == 0).
4387 * 2) We may not be at the beginning of the SPTM ops array, but we are about to add the first operation
4388 * for this physical page (num_mappings == 0, flush_range->pending_disjoint_entries == ?).
4389 * 3) We need to change the options passed to the SPTM for a run of one or more mappings. Specifically,
4390 * if we encounter a run of mappings that reside outside the VA region of our flush_range, or that
4391 * belong to a pmap other than the one targeted by our flush_range, we should ask the SPTM to flush
4392 * the TLB for us (i.e., clear SPTM_UPDATE_DEFER_TLBI), but only for those specific mappings.
4393 */
4394 uint32_t per_mapping_sptm_update_options = sptm_update_options;
4395 if ((flush_range->ptfr_pmap != pmap) || (va >= flush_range->ptfr_end) || (va < flush_range->ptfr_start)) {
4396 per_mapping_sptm_update_options &= ~SPTM_UPDATE_DEFER_TLBI;
4397 }
4398 if ((num_mappings == 0) ||
4399 (flush_range->current_header->per_paddr_header.options != per_mapping_sptm_update_options)) {
4400 if (pmap_multipage_op_add_page(phys, &num_mappings, per_mapping_sptm_update_options, flush_range)) {
4401 /**
4402 * If we needed to submit the pending disjoint ops to make room for the new page,
4403 * flush any pending region ops to reenable preemption and restart the loop with
4404 * the lock in sleep mode. This prevents preemption from being held disabled
4405 * for an arbitrary amount of time in the pathological case in which we have
4406 * both pending region ops and an excessively long PV list that repeatedly
4407 * requires new page headers with SPTM_MAPPING_LIMIT - 1 entries already pending.
4408 */
4409 pmap_multipage_op_submit_region(flush_range);
4410 assert(num_mappings == 0);
4411 num_skipped_mappings = 0;
4412 pvh_lock_sleep_mode_needed = true;
4413 continue;
4414 }
4415 }
4416 }
4417
4418 if (__improbable((pmap == NULL) ||
4419 (((spte & ARM_PTE_TYPE_MASK) != ARM_PTE_TYPE_FAULT) && (atop(pte_to_pa(spte)) != ppnum)))) {
4420 #if MACH_ASSERT
4421 if ((pmap != NULL) && (pve_p != PV_ENTRY_NULL) && (kern_feature_override(KF_PMAPV_OVRD) == FALSE)) {
4422 /* Temporarily set PTEP to NULL so that the logic below doesn't pick it up as duplicate. */
4423 pt_entry_t *temp_ptep = pve_get_ptep(pve_p, pve_ptep_idx);
4424 pve_set_ptep(pve_p, pve_ptep_idx, PT_ENTRY_NULL);
4425
4426 pv_entry_t *check_pvep = pve_p;
4427
4428 do {
4429 if (pve_find_ptep_index(check_pvep, pte_p) != -1) {
4430 panic_plain("%s: duplicate pve entry ptep=%p pmap=%p, pvh=%p, "
4431 "pvep=%p, pai=0x%x", __func__, pte_p, pmap, (void*)local_locked_pvh.pvh, pve_p, pai);
4432 }
4433 } while ((check_pvep = pve_next(check_pvep)) != PV_ENTRY_NULL);
4434
4435 /* Restore previous PTEP value. */
4436 pve_set_ptep(pve_p, pve_ptep_idx, temp_ptep);
4437 }
4438 #endif
4439 panic("%s: bad PVE pte_p=%p pmap=%p prot=%d options=%u, pvh=%p, pveh_p=%p, pve_p=%p, pte=0x%llx, va=0x%llx ppnum: 0x%x",
4440 __func__, pte_p, pmap, prot, options, (void*)local_locked_pvh.pvh, pveh_p, pve_p, (uint64_t)*pte_p, (uint64_t)va, ppnum);
4441 }
4442
4443 pt_entry_t pte_template = ARM_PTE_EMPTY;
4444
4445 if (ptdp != NULL) {
4446 sptm_ops[num_mappings].root_pt_paddr = pmap->ttep;
4447 sptm_ops[num_mappings].vaddr = va;
4448 }
4449
4450 /* Remove the mapping if new protection is NONE */
4451 if (remove) {
4452 sptm_ptds[num_mappings] = ptdp;
4453 sptm_ptd_info[num_mappings] = ptd_get_info(ptdp);
4454 sptm_pcpu->sptm_acct_flags[num_mappings] = 0;
4455 if (pmap != kernel_pmap) {
4456 const bool is_internal = ppattr_pve_is_internal(pai, pve_p, pve_ptep_idx);
4457 const bool is_altacct = ppattr_pve_is_altacct(pai, pve_p, pve_ptep_idx);
4458
4459 if (is_internal) {
4460 sptm_pcpu->sptm_acct_flags[num_mappings] |= PMAP_SPTM_FLAG_INTERNAL;
4461 ppattr_pve_clr_internal(pai, pve_p, pve_ptep_idx);
4462 }
4463 if (is_altacct) {
4464 sptm_pcpu->sptm_acct_flags[num_mappings] |= PMAP_SPTM_FLAG_ALTACCT;
4465 ppattr_pve_clr_altacct(pai, pve_p, pve_ptep_idx);
4466 }
4467 if (compress && is_internal) {
4468 pte_template = ARM_PTE_COMPRESSED;
4469 if (is_altacct) {
4470 pte_template |= ARM_PTE_COMPRESSED_ALT;
4471 }
4472 }
4473 }
4474 /* Remove this CPU mapping from PVE list. */
4475 if (pve_p != PV_ENTRY_NULL) {
4476 pve_set_ptep(pve_p, pve_ptep_idx, PT_ENTRY_NULL);
4477 }
4478 } else {
4479 const pt_attr_t *const pt_attr = pmap_get_pt_attr(pmap);
4480
4481 if (pmap == kernel_pmap) {
4482 pte_template = ((spte & ~ARM_PTE_APMASK) | ARM_PTE_AP(AP_RONA));
4483 } else {
4484 pte_template = ((spte & ~ARM_PTE_APMASK) | pt_attr_leaf_ro(pt_attr));
4485 }
4486
4487 /*
4488 * We must at least clear the 'was writeable' flag, as we're at least revoking write access,
4489 * meaning that the VM is effectively requesting that subsequent write accesses to these mappings
4490 * go through vm_fault() instead of being handled by arm_fast_fault().
4491 */
4492 pte_set_was_writeable(pte_template, false);
4493
4494 /*
4495 * While the naive implementation of this would serve to add execute
4496 * permission, this is not how the VM uses this interface, or how
4497 * x86_64 implements it. So ignore requests to add execute permissions.
4498 */
4499 #if DEVELOPMENT || DEBUG
4500 if ((!(prot & VM_PROT_EXECUTE) && nx_enabled && pmap->nx_enabled) ||
4501 (pte_to_xprr_perm(spte) == XPRR_USER_TPRO_PERM))
4502 #else
4503 if (!(prot & VM_PROT_EXECUTE) ||
4504 (pte_to_xprr_perm(spte) == XPRR_USER_TPRO_PERM))
4505 #endif
4506 {
4507 pte_template |= pt_attr_leaf_xn(pt_attr);
4508 }
4509 }
4510
4511 if (ptdp != NULL) {
4512 sptm_ops[num_mappings].pte_template = pte_template;
4513 ++num_mappings;
4514 } else if (pmap_insert_flush_range_template(pte_template, flush_range)) {
4515 /**
4516 * We submit both the pending disjoint and pending region ops whenever
4517 * either category reaches the mapping limit. Having pending operations
4518 * in either category will keep preemption disabled, and we want to ensure
4519 * that we can at least temporarily re-enable preemption roughly every
4520 * SPTM_MAPPING_LIMIT mappings.
4521 */
4522 pmap_multipage_op_submit_disjoint(num_mappings, flush_range);
4523 pvh_lock_sleep_mode_needed = true;
4524 num_mappings = num_skipped_mappings = 0;
4525 }
4526
4527 protect_skip_pve:
4528 if ((num_mappings + num_skipped_mappings) >= SPTM_MAPPING_LIMIT) {
4529 if (flush_range != NULL) {
4530 /* See comment above for why we submit both disjoint and region ops when we hit the limit. */
4531 pmap_multipage_op_submit_disjoint(num_mappings, flush_range);
4532 pmap_multipage_op_submit_region(flush_range);
4533 } else if (num_mappings > 0) {
4534 if (remove) {
4535 pmap_disjoint_unmap(phys, num_mappings);
4536 } else {
4537 sptm_update_disjoint(phys, sptm_pcpu->sptm_ops_pa, num_mappings, sptm_update_options);
4538 }
4539 }
4540 pvh_lock_sleep_mode_needed = true;
4541 num_mappings = num_skipped_mappings = 0;
4542 }
4543 pte_p = PT_ENTRY_NULL;
4544 if ((pve_p != PV_ENTRY_NULL) && (++pve_ptep_idx == PTE_PER_PVE)) {
4545 pve_ptep_idx = 0;
4546
4547 if (remove) {
4548 /**
4549 * If there are any IOMMU mappings in the PVE list, preserve
4550 * those mappings in a new PVE list (new_pve_p) which will later
4551 * become the new PVH entry. Keep track of the CPU mappings in
4552 * pveh_p/pvet_p so they can be deallocated later.
4553 */
4554 if (iommu_mapping_in_pve) {
4555 iommu_mapping_in_pve = false;
4556 pv_entry_t *temp_pve_p = pve_next(pve_p);
4557 pve_remove(&local_locked_pvh, pve_pp, pve_p);
4558 if (pvh_test_type(local_locked_pvh.pvh, PVH_TYPE_PVEP)) {
4559 pveh_p = pvh_pve_list(local_locked_pvh.pvh);
4560 } else {
4561 assert(pvh_test_type(local_locked_pvh.pvh, PVH_TYPE_NULL));
4562 pveh_p = PV_ENTRY_NULL;
4563 }
4564 pve_p->pve_next = new_pve_p;
4565 new_pve_p = pve_p;
4566 pve_p = temp_pve_p;
4567 continue;
4568 } else {
4569 pvet_p = pve_p;
4570 pvh_cnt++;
4571 }
4572 }
4573
4574 pve_pp = pve_next_ptr(pve_p);
4575 pve_p = pve_next(pve_p);
4576 iommu_mapping_in_pve = false;
4577 }
4578 }
4579
4580 if (num_mappings != 0) {
4581 if (remove) {
4582 pmap_disjoint_unmap(phys, num_mappings);
4583 } else if (flush_range == NULL) {
4584 sptm_update_disjoint(phys, sptm_pcpu->sptm_ops_pa, num_mappings, sptm_update_options);
4585 } else {
4586 /* Resync the pending mapping state in flush_range with our local state. */
4587 assert(num_mappings >= flush_range->pending_disjoint_entries);
4588 flush_range->pending_disjoint_entries = num_mappings;
4589 }
4590 }
4591
4592 if (remove) {
4593 os_atomic_store(&pmap_cpu_data->inflight_disconnect, false, release);
4594 }
4595
4596 /**
4597 * Undo the explicit disable_preemption() done in PPO_PERCPU_INIT().
4598 * Note that enable_preemption() decrements a per-thread counter, so if
4599 * we happen to still hold the PVH lock in spin mode then preemption won't
4600 * actually be re-enabled until we drop the lock (which also decrements
4601 * the per-thread counter.
4602 */
4603 enable_preemption();
4604
4605 /* if we removed a bunch of entries, take care of them now */
4606 if (remove) {
4607 /**
4608 * If we (or our caller as indicated by PMAP_OPTIONS_PPO_PENDING_RETYPE) will
4609 * be retyping the page, we need to drain the epochs to ensure that concurrent
4610 * calls to batched operations such as pmap_remove() and the various multipage
4611 * attribute update functions have finished consuming mappings of this page.
4612 */
4613 const bool needs_retyping = pmap_prepare_unmapped_page_for_retype(phys);
4614 if ((options & PMAP_OPTIONS_PPO_PENDING_RETYPE) && !needs_retyping) {
4615 /**
4616 * pmap_prepare_unmapped_page_for_retype() will only return true if
4617 * the page belongs to a certain set of types that need to be auto-
4618 * retyped back to XNU_DEFAULT when they are unmapped. But if the
4619 * caller indicated that it's going to retype the page, we need
4620 * to drain the epochs regardless of the current page type.
4621 */
4622 pmap_retype_epoch_prepare_drain();
4623 }
4624 if (new_pve_p != PV_ENTRY_NULL) {
4625 pvh_update_head(&local_locked_pvh, new_pve_p, PVH_TYPE_PVEP);
4626 } else if (new_pte_p != PT_ENTRY_NULL) {
4627 pvh_update_head(&local_locked_pvh, new_pte_p, PVH_TYPE_PTEP);
4628 } else {
4629 pvh_set_flags(&local_locked_pvh, 0);
4630 pvh_update_head(&local_locked_pvh, PV_ENTRY_NULL, PVH_TYPE_NULL);
4631 }
4632
4633 /* If removing the last mapping to a specially-protected page, retype the page back to XNU_DEFAULT. */
4634 const bool retype_needed = pmap_retype_unmapped_page(phys);
4635 if ((options & PMAP_OPTIONS_PPO_PENDING_RETYPE) && !retype_needed) {
4636 pmap_retype_epoch_drain();
4637 }
4638 }
4639
4640 if (__probable(locked_pvh == NULL)) {
4641 pvh_unlock(&local_locked_pvh);
4642 } else {
4643 *locked_pvh = local_locked_pvh;
4644 }
4645
4646 if (remove && (pvet_p != PV_ENTRY_NULL)) {
4647 assert(pveh_p != PV_ENTRY_NULL);
4648 pv_list_free(pveh_p, pvet_p, pvh_cnt);
4649 }
4650
4651 if ((flush_range != NULL) && !preemption_enabled()) {
4652 flush_range->processed_entries += num_skipped_mappings;
4653 }
4654 }
4655
4656 MARK_AS_PMAP_TEXT void
4657 pmap_page_protect_options_internal(
4658 ppnum_t ppnum,
4659 vm_prot_t prot,
4660 unsigned int options,
4661 void *arg)
4662 {
4663 if (arg != NULL) {
4664 /*
4665 * This is a legacy argument from pre-ARM era that the VM layer passes in to hint that it will call
4666 * pmap_flush() later to flush the TLB. On ARM platforms, however, pmap_flush() is not implemented,
4667 * as it's typically more efficient to perform the TLB flushing inline with the page table updates
4668 * themselves. Therefore, if the argument is non-NULL, pmap will take care of TLB flushing itself
4669 * by clearing PMAP_OPTIONS_NOFLUSH.
4670 */
4671 options &= ~PMAP_OPTIONS_NOFLUSH;
4672 }
4673 pmap_page_protect_options_with_flush_range(ppnum, prot, options, NULL, NULL);
4674 }
4675
4676 void
4677 pmap_page_protect_options(
4678 ppnum_t ppnum,
4679 vm_prot_t prot,
4680 unsigned int options,
4681 void *arg)
4682 {
4683 pmap_paddr_t phys = ptoa(ppnum);
4684
4685 assert(ppnum != vm_page_fictitious_addr);
4686
4687 /* Only work with managed pages. */
4688 if (!pa_valid(phys)) {
4689 return;
4690 }
4691
4692 /*
4693 * Determine the new protection.
4694 */
4695 if (prot == VM_PROT_ALL) {
4696 return; /* nothing to do */
4697 }
4698
4699 PMAP_TRACE(2, PMAP_CODE(PMAP__PAGE_PROTECT) | DBG_FUNC_START, ppnum, prot);
4700
4701 pmap_page_protect_options_internal(ppnum, prot, options, arg);
4702
4703 PMAP_TRACE(2, PMAP_CODE(PMAP__PAGE_PROTECT) | DBG_FUNC_END);
4704 }
4705
4706
4707 #if __has_feature(ptrauth_calls) && (defined(XNU_TARGET_OS_OSX) || (DEVELOPMENT || DEBUG))
4708 MARK_AS_PMAP_TEXT void
4709 pmap_disable_user_jop_internal(pmap_t pmap)
4710 {
4711 if (pmap == kernel_pmap) {
4712 panic("%s: called with kernel_pmap", __func__);
4713 }
4714 validate_pmap_mutable(pmap);
4715 sptm_configure_root(pmap->ttep, 0, SPTM_ROOT_PT_FLAG_JOP);
4716 pmap->disable_jop = true;
4717 }
4718
4719 void
4720 pmap_disable_user_jop(pmap_t pmap)
4721 {
4722 pmap_disable_user_jop_internal(pmap);
4723 }
4724 #endif /* __has_feature(ptrauth_calls) && (defined(XNU_TARGET_OS_OSX) || (DEVELOPMENT || DEBUG)) */
4725
4726 /*
4727 * Indicates if the pmap layer enforces some additional restrictions on the
4728 * given set of protections.
4729 */
4730 bool
4731 pmap_has_prot_policy(__unused pmap_t pmap, __unused bool translated_allow_execute, __unused vm_prot_t prot)
4732 {
4733 return false;
4734 }
4735
4736 /*
4737 * Set the physical protection on the
4738 * specified range of this map as requested.
4739 * VERY IMPORTANT: Will not increase permissions.
4740 * VERY IMPORTANT: Only pmap_enter() is allowed to grant permissions.
4741 */
4742 void
4743 pmap_protect(
4744 pmap_t pmap,
4745 vm_map_address_t b,
4746 vm_map_address_t e,
4747 vm_prot_t prot)
4748 {
4749 pmap_protect_options(pmap, b, e, prot, 0, NULL);
4750 }
4751
4752 static bool
4753 pmap_protect_strong_sync(unsigned int num_mappings __unused)
4754 {
4755 return false;
4756 }
4757
4758 MARK_AS_PMAP_TEXT vm_map_address_t
4759 pmap_protect_options_internal(
4760 pmap_t pmap,
4761 vm_map_address_t start,
4762 vm_map_address_t end,
4763 vm_prot_t prot,
4764 unsigned int options,
4765 __unused void *args)
4766 {
4767 pt_entry_t *pte_p;
4768 bool set_NX = true;
4769 bool set_XO = false;
4770 bool should_have_removed = false;
4771 bool need_strong_sync = false;
4772
4773 /* Validate the pmap input before accessing its data. */
4774 validate_pmap_mutable(pmap);
4775
4776 const pt_attr_t *const pt_attr = pmap_get_pt_attr(pmap);
4777
4778 if (__improbable((end < start) || (end > ((start + pt_attr_twig_size(pt_attr)) & ~pt_attr_twig_offmask(pt_attr))))) {
4779 panic("%s: invalid address range %p, %p", __func__, (void*)start, (void*)end);
4780 }
4781
4782 #if DEVELOPMENT || DEBUG
4783 if (options & PMAP_OPTIONS_PROTECT_IMMEDIATE) {
4784 if ((prot & VM_PROT_ALL) == VM_PROT_NONE) {
4785 should_have_removed = true;
4786 }
4787 } else
4788 #endif
4789 {
4790 /* Determine the new protection. */
4791 switch (prot) {
4792 case VM_PROT_EXECUTE:
4793 set_XO = true;
4794 OS_FALLTHROUGH;
4795 case VM_PROT_READ:
4796 case VM_PROT_READ | VM_PROT_EXECUTE:
4797 break;
4798 case VM_PROT_READ | VM_PROT_WRITE:
4799 case VM_PROT_ALL:
4800 return end; /* nothing to do */
4801 default:
4802 should_have_removed = true;
4803 }
4804 }
4805
4806 if (__improbable(should_have_removed)) {
4807 panic("%s: should have been a remove operation, "
4808 "pmap=%p, start=%p, end=%p, prot=%#x, options=%#x, args=%p",
4809 __FUNCTION__,
4810 pmap, (void *)start, (void *)end, prot, options, args);
4811 }
4812
4813 #if DEVELOPMENT || DEBUG
4814 bool force_write = false;
4815 if ((options & PMAP_OPTIONS_PROTECT_IMMEDIATE) && (prot & VM_PROT_WRITE)) {
4816 force_write = true;
4817 }
4818 if ((prot & VM_PROT_EXECUTE) || !nx_enabled || !pmap->nx_enabled)
4819 #else
4820 if ((prot & VM_PROT_EXECUTE))
4821 #endif
4822 {
4823 set_NX = false;
4824 } else {
4825 set_NX = true;
4826 }
4827
4828 const uint64_t pmap_page_size = PAGE_RATIO * pt_attr_page_size(pt_attr);
4829 vm_map_address_t va = start;
4830 vm_map_address_t sptm_start_va = start;
4831 unsigned int num_mappings = 0;
4832
4833 pmap_lock(pmap, PMAP_LOCK_SHARED);
4834
4835 pte_p = pmap_pte(pmap, start);
4836
4837 if (pte_p == NULL) {
4838 pmap_unlock(pmap, PMAP_LOCK_SHARED);
4839 return end;
4840 }
4841
4842 pmap_sptm_percpu_data_t *sptm_pcpu = NULL;
4843 #if DEVELOPMENT || DEBUG
4844 if (!force_write)
4845 #endif
4846 {
4847 disable_preemption();
4848 sptm_pcpu = PERCPU_GET(pmap_sptm_percpu);
4849 }
4850
4851 pt_entry_t tmplate = ARM_PTE_EMPTY;
4852
4853 if (pmap == kernel_pmap) {
4854 #if DEVELOPMENT || DEBUG
4855 if (force_write) {
4856 tmplate = ARM_PTE_AP(AP_RWNA);
4857 } else
4858 #endif
4859 {
4860 tmplate = ARM_PTE_AP(AP_RONA);
4861 }
4862 } else {
4863 #if DEVELOPMENT || DEBUG
4864 if (force_write) {
4865 assert(pmap->type != PMAP_TYPE_NESTED);
4866 tmplate = pt_attr_leaf_rw(pt_attr);
4867 } else
4868 #endif
4869 if (set_XO) {
4870 tmplate = pt_attr_leaf_rona(pt_attr);
4871 } else {
4872 tmplate = pt_attr_leaf_ro(pt_attr);
4873 }
4874 }
4875
4876 if (set_NX) {
4877 tmplate |= pt_attr_leaf_xn(pt_attr);
4878 }
4879
4880 while (va < end) {
4881 pt_entry_t spte = ARM_PTE_EMPTY;
4882
4883 /**
4884 * Removing "NX" would grant "execute" access immediately, bypassing any
4885 * checks VM might want to do in its soft fault path.
4886 * pmap_protect() and co. are not allowed to increase access permissions,
4887 * except in the PMAP_PROTECT_OPTIONS_IMMEDIATE internal-only case.
4888 * Therefore, if we are not explicitly clearing execute permissions, inherit
4889 * the existing permissions.
4890 */
4891 if (!set_NX) {
4892 spte = os_atomic_load(pte_p, relaxed);
4893 if (__improbable((spte & ARM_PTE_TYPE_MASK) == ARM_PTE_TYPE_FAULT)) {
4894 tmplate |= pt_attr_leaf_xn(pt_attr);
4895 } else {
4896 tmplate |= (spte & ARM_PTE_XMASK);
4897 }
4898 }
4899
4900 #if DEVELOPMENT || DEBUG
4901 /*
4902 * PMAP_OPTIONS_PROTECT_IMMEDIATE is an internal-only option that's intended to
4903 * provide a "backdoor" to allow normally write-protected compressor pages to be
4904 * be temporarily written without triggering expensive write faults.
4905 * SPTM TODO: Given the intended use of this flag, we may be able to relax some
4906 * of our assumptions below when it comes to ref/mod accounting, and we may be
4907 * able to avoid holding the PVH lock across the SPTM mapping operation and the
4908 * ref/mod updates. This will be important if we move to a batched SPTM mapping
4909 * API.
4910 */
4911 if (force_write) {
4912 if (spte == ARM_PTE_EMPTY) {
4913 spte = os_atomic_load(pte_p, relaxed);
4914 }
4915
4916 /* A concurrent remove or disconnect may have cleared the PTE. */
4917 if (__improbable((spte & ARM_PTE_TYPE_MASK) == ARM_PTE_TYPE_FAULT)) {
4918 goto pmap_protect_insert_mapping;
4919 }
4920
4921 /* Inherit permissions and "was_writeable" from the template. */
4922 spte = (spte & ~(ARM_PTE_APMASK | ARM_PTE_XMASK | ARM_PTE_WRITEABLE)) |
4923 (tmplate & (ARM_PTE_APMASK | ARM_PTE_XMASK | ARM_PTE_WRITEABLE));
4924
4925 /* Access flag should be set for any immediate change in protections */
4926 spte |= ARM_PTE_AF;
4927 const pmap_paddr_t pa = pte_to_pa(spte);
4928 const unsigned int pai = pa_index(pa);
4929 locked_pvh_t locked_pvh;
4930 if (pa_valid(pa)) {
4931 locked_pvh = pvh_lock(pai);
4932 ppattr_modify_bits(pai, PP_ATTR_REFFAULT | PP_ATTR_MODFAULT,
4933 PP_ATTR_REFERENCED | PP_ATTR_MODIFIED);
4934 }
4935
4936 __assert_only const sptm_return_t sptm_status = sptm_map_page(pmap->ttep, va, spte);
4937
4938 /*
4939 * We don't expect the VM to be concurrently removing these compressor mappings.
4940 * If it does for some reason, we can check for SPTM_MAP_FLUSH_PENDING and continue
4941 * the main loop.
4942 */
4943 assert((sptm_status == SPTM_SUCCESS) || (sptm_status == SPTM_MAP_VALID));
4944
4945 if (pa_valid(pa)) {
4946 pvh_unlock(&locked_pvh);
4947 }
4948 }
4949
4950 pmap_protect_insert_mapping:
4951 #endif /* DEVELOPMENT || DEBUG */
4952
4953 va += pmap_page_size;
4954 ++pte_p;
4955
4956 #if DEVELOPMENT || DEBUG
4957 if (!force_write)
4958 #endif
4959 {
4960 sptm_pcpu->sptm_templates[num_mappings] = tmplate;
4961 ++num_mappings;
4962 if (num_mappings == SPTM_MAPPING_LIMIT) {
4963 /**
4964 * Enter the retype epoch for the batched update operation. This is necessary because we
4965 * cannot reasonably hold the PVH locks for all pages mapped by the region during this
4966 * call, so a concurrent pmap_page_protect() operation against one of those pages may
4967 * race this call. That should be perfectly fine as far as the PTE updates are concerned,
4968 * but if pmap_page_protect() then needs to retype the page, an SPTM violation may result
4969 * if it does not first drain our epoch.
4970 */
4971 pmap_retype_epoch_enter();
4972 sptm_update_region(pmap->ttep, sptm_start_va, num_mappings, sptm_pcpu->sptm_templates_pa,
4973 SPTM_UPDATE_PERMS_AND_WAS_WRITABLE);
4974 pmap_retype_epoch_exit();
4975 need_strong_sync = need_strong_sync || pmap_protect_strong_sync(num_mappings);
4976
4977 /* Temporarily re-enable preemption to allow any urgent ASTs to be processed. */
4978 enable_preemption();
4979 num_mappings = 0;
4980 sptm_start_va = va;
4981 disable_preemption();
4982 sptm_pcpu = PERCPU_GET(pmap_sptm_percpu);
4983 }
4984 }
4985 }
4986
4987 /* This won't happen in the force_write case as we should never increment num_mappings. */
4988 if (num_mappings != 0) {
4989 pmap_retype_epoch_enter();
4990 sptm_update_region(pmap->ttep, sptm_start_va, num_mappings, sptm_pcpu->sptm_templates_pa,
4991 SPTM_UPDATE_PERMS_AND_WAS_WRITABLE);
4992 pmap_retype_epoch_exit();
4993 need_strong_sync = need_strong_sync || pmap_protect_strong_sync(num_mappings);
4994 }
4995
4996 #if DEVELOPMENT || DEBUG
4997 if (!force_write)
4998 #endif
4999 {
5000 enable_preemption();
5001 }
5002 pmap_unlock(pmap, PMAP_LOCK_SHARED);
5003 if (__improbable(need_strong_sync)) {
5004 arm64_sync_tlb(true);
5005 }
5006 return va;
5007 }
5008
5009 void
5010 pmap_protect_options(
5011 pmap_t pmap,
5012 vm_map_address_t b,
5013 vm_map_address_t e,
5014 vm_prot_t prot,
5015 unsigned int options,
5016 __unused void *args)
5017 {
5018 vm_map_address_t l, beg;
5019
5020 __unused const pt_attr_t * const pt_attr = pmap_get_pt_attr(pmap);
5021
5022 if ((b | e) & pt_attr_leaf_offmask(pt_attr)) {
5023 panic("pmap_protect_options() pmap %p start 0x%llx end 0x%llx",
5024 pmap, (uint64_t)b, (uint64_t)e);
5025 }
5026
5027 /*
5028 * We allow single-page requests to execute non-preemptibly,
5029 * as it doesn't make sense to sample AST_URGENT for a single-page
5030 * operation, and there are a couple of special use cases that
5031 * require a non-preemptible single-page operation.
5032 */
5033 if ((e - b) > (pt_attr_page_size(pt_attr) * PAGE_RATIO)) {
5034 pmap_verify_preemptible();
5035 }
5036
5037 #if DEVELOPMENT || DEBUG
5038 if (options & PMAP_OPTIONS_PROTECT_IMMEDIATE) {
5039 if ((prot & VM_PROT_ALL) == VM_PROT_NONE) {
5040 pmap_remove_options(pmap, b, e, options);
5041 return;
5042 }
5043 } else
5044 #endif
5045 {
5046 /* Determine the new protection. */
5047 switch (prot) {
5048 case VM_PROT_EXECUTE:
5049 case VM_PROT_READ:
5050 case VM_PROT_READ | VM_PROT_EXECUTE:
5051 break;
5052 case VM_PROT_READ | VM_PROT_WRITE:
5053 case VM_PROT_ALL:
5054 return; /* nothing to do */
5055 default:
5056 pmap_remove_options(pmap, b, e, options);
5057 return;
5058 }
5059 }
5060
5061 PMAP_TRACE(2, PMAP_CODE(PMAP__PROTECT) | DBG_FUNC_START,
5062 VM_KERNEL_ADDRHIDE(pmap), VM_KERNEL_ADDRHIDE(b),
5063 VM_KERNEL_ADDRHIDE(e));
5064
5065 beg = b;
5066
5067 while (beg < e) {
5068 l = ((beg + pt_attr_twig_size(pt_attr)) & ~pt_attr_twig_offmask(pt_attr));
5069
5070 if (l > e) {
5071 l = e;
5072 }
5073
5074 beg = pmap_protect_options_internal(pmap, beg, l, prot, options, args);
5075 }
5076
5077 PMAP_TRACE(2, PMAP_CODE(PMAP__PROTECT) | DBG_FUNC_END);
5078 }
5079
5080 /**
5081 * Inserts an arbitrary number of physical pages ("block") in a pmap.
5082 *
5083 * @param pmap pmap to insert the pages into.
5084 * @param va virtual address to map the pages into.
5085 * @param pa page number of the first physical page to map.
5086 * @param size block size, in number of pages.
5087 * @param prot mapping protection attributes.
5088 * @param attr flags to pass to pmap_enter().
5089 *
5090 * @return KERN_SUCCESS.
5091 */
5092 kern_return_t
5093 pmap_map_block(
5094 pmap_t pmap,
5095 addr64_t va,
5096 ppnum_t pa,
5097 uint32_t size,
5098 vm_prot_t prot,
5099 int attr,
5100 unsigned int flags)
5101 {
5102 return pmap_map_block_addr(pmap, va, ((pmap_paddr_t)pa) << PAGE_SHIFT, size, prot, attr, flags);
5103 }
5104
5105 /**
5106 * Inserts an arbitrary number of physical pages ("block") in a pmap.
5107 * As opposed to pmap_map_block(), this function takes
5108 * a physical address as an input and operates using the
5109 * page size associated with the input pmap.
5110 *
5111 * @param pmap pmap to insert the pages into.
5112 * @param va virtual address to map the pages into.
5113 * @param pa physical address of the first physical page to map.
5114 * @param size block size, in number of pages.
5115 * @param prot mapping protection attributes.
5116 * @param attr flags to pass to pmap_enter().
5117 *
5118 * @return KERN_SUCCESS.
5119 */
5120 kern_return_t
5121 pmap_map_block_addr(
5122 pmap_t pmap,
5123 addr64_t va,
5124 pmap_paddr_t pa,
5125 uint32_t size,
5126 vm_prot_t prot,
5127 int attr,
5128 unsigned int flags)
5129 {
5130 #if __ARM_MIXED_PAGE_SIZE__
5131 const pt_attr_t * const pt_attr = pmap_get_pt_attr(pmap);
5132 const uint64_t pmap_page_size = pt_attr_page_size(pt_attr);
5133 #else
5134 const uint64_t pmap_page_size = PAGE_SIZE;
5135 #endif
5136
5137 for (ppnum_t page = 0; page < size; page++) {
5138 if (pmap_enter_addr(pmap, va, pa, prot, VM_PROT_NONE, attr, TRUE, PMAP_MAPPING_TYPE_INFER) != KERN_SUCCESS) {
5139 panic("%s: failed pmap_enter_addr, "
5140 "pmap=%p, va=%#llx, pa=%llu, size=%u, prot=%#x, flags=%#x",
5141 __FUNCTION__,
5142 pmap, va, (uint64_t)pa, size, prot, flags);
5143 }
5144
5145 va += pmap_page_size;
5146 pa += pmap_page_size;
5147 }
5148
5149 return KERN_SUCCESS;
5150 }
5151
5152 kern_return_t
5153 pmap_enter_addr(
5154 pmap_t pmap,
5155 vm_map_address_t v,
5156 pmap_paddr_t pa,
5157 vm_prot_t prot,
5158 vm_prot_t fault_type,
5159 unsigned int flags,
5160 boolean_t wired,
5161 pmap_mapping_type_t mapping_type)
5162 {
5163 return pmap_enter_options_addr(pmap, v, pa, prot, fault_type, flags, wired, 0, NULL, mapping_type);
5164 }
5165
5166 /*
5167 * Insert the given physical page (p) at
5168 * the specified virtual address (v) in the
5169 * target physical map with the protection requested.
5170 *
5171 * If specified, the page will be wired down, meaning
5172 * that the related pte can not be reclaimed.
5173 *
5174 * NB: This is the only routine which MAY NOT lazy-evaluate
5175 * or lose information. That is, this routine must actually
5176 * insert this page into the given map eventually (must make
5177 * forward progress eventually.
5178 */
5179 kern_return_t
5180 pmap_enter(
5181 pmap_t pmap,
5182 vm_map_address_t v,
5183 ppnum_t pn,
5184 vm_prot_t prot,
5185 vm_prot_t fault_type,
5186 unsigned int flags,
5187 boolean_t wired,
5188 pmap_mapping_type_t mapping_type)
5189 {
5190 return pmap_enter_addr(pmap, v, ((pmap_paddr_t)pn) << PAGE_SHIFT, prot, fault_type, flags, wired, mapping_type);
5191 }
5192
5193 /*
5194 * Attempt to update a PTE constructed by pmap_enter_options().
5195 *
5196 * @note performs no page table or accounting modifications, nor any lasting SPTM page type modification, on failure.
5197 * @note expects to be called with preemption disabled to guarantee safe access to SPTM per-CPU data.
5198 *
5199 * @param pmap The pmap representing the address space in which to store the new PTE
5200 * @param pte_p The physical aperture KVA of the PTE to store
5201 * @param new_pte The new value to store in *pte_p
5202 * @param v The virtual address mapped by pte_p
5203 * @param locked_pvh Input/Output parameter pointing to a wrapped pv_head_table entry returned by
5204 * a previous call to pvh_lock(). *locked_pvh will be updated if existing mappings
5205 * need to be disconnected prior to retyping.
5206 * @param old_pte Returns the prior PTE contents, iff the PTE is successfully updated
5207 * @param options bitmask of PMAP_OPTIONS_* flags passed to pmap_enter_options().
5208 * @param mapping_type The type of the new mapping, this defines which SPTM frame type to use.
5209 *
5210 * @return SPTM_SUCCESS iff able to successfully update *pte_p to new_pte via sptm_map_page(),
5211 * SPTM_MAP_VALID if an existing mapping was successfully upgraded via sptm_map_page(),
5212 * SPTM_MAP_FLUSH_PENDING if the TLB flush of a previous mapping is still in-flight and
5213 * the mapping operation should be retried, or if the mapping operation should be retried
5214 * because we had to temporarily re-enable preemption which would invalidate caller-held
5215 * per-CPU data.
5216 * Otherwise an appropriate SPTM or TXM error code; in these cases the mapping should not be
5217 * retried and the caller should return an error.
5218 */
5219 static inline sptm_return_t
5220 pmap_enter_pte(
5221 pmap_t pmap,
5222 pt_entry_t *pte_p,
5223 pt_entry_t new_pte,
5224 locked_pvh_t *locked_pvh,
5225 pt_entry_t *old_pte,
5226 vm_map_address_t v,
5227 unsigned int options,
5228 pmap_mapping_type_t mapping_type)
5229 {
5230 sptm_pte_t prev_pte;
5231 bool changed_wiring = false;
5232
5233 assert(pte_p != NULL);
5234 assert(old_pte != NULL);
5235
5236 /* SPTM TODO: handle PAGE_RATIO_4 configurations if those devices remain supported. */
5237
5238 assert(get_preemption_level() > 0);
5239 const pmap_paddr_t pa = pte_to_pa(new_pte) & ~PAGE_MASK;
5240 sptm_frame_type_t prev_frame_type = XNU_DEFAULT;
5241 sptm_frame_type_t new_frame_type = XNU_DEFAULT;
5242
5243 /*
5244 * If the caller specified a mapping type of PMAP_MAPPINGS_TYPE_INFER, then we
5245 * keep the existing logic of deriving the SPTM frame type from the XPRR permissions.
5246 *
5247 * If the caller specified another mapping type, we simply follow that. This refactor was
5248 * needed for the XNU_KERNEL_RESTRICTED work, and it also allows us to be more precise at
5249 * what we want. It's better to let the caller specify the mapping type rather than use the
5250 * permissions for that.
5251 *
5252 * In the future, we should move entirely to use pmap_mapping_type_t; see rdar://114886323.
5253 */
5254 if (mapping_type != PMAP_MAPPING_TYPE_INFER) {
5255 switch (mapping_type) {
5256 case PMAP_MAPPING_TYPE_DEFAULT:
5257 new_frame_type = (sptm_frame_type_t)mapping_type;
5258 break;
5259 case PMAP_MAPPING_TYPE_ROZONE:
5260 assert(((pmap == kernel_pmap) && zone_spans_ro_va(v, v + pt_attr_page_size(pmap_get_pt_attr(pmap)))));
5261 new_frame_type = (sptm_frame_type_t)mapping_type;
5262 break;
5263 case PMAP_MAPPING_TYPE_RESTRICTED:
5264 if (use_xnu_restricted) {
5265 new_frame_type = (sptm_frame_type_t)mapping_type;
5266 } else {
5267 new_frame_type = XNU_DEFAULT;
5268 }
5269 break;
5270 default:
5271 panic("invalid mapping type: %d", mapping_type);
5272 }
5273 } else if (__improbable(pte_to_xprr_perm(new_pte) == XPRR_USER_JIT_PERM)) {
5274 /*
5275 * Always check for XPRR_USER_JIT_PERM before we check for anything else. When using
5276 * RWX permissions, the only allowed type is XNU_USER_JIT, regardless of any other
5277 * flags which the VM may have provided.
5278 *
5279 * TODO: Assert that the PMAP_OPTIONS_XNU_USER_DEBUG flag isn't set when entering
5280 * this case. We can't do this for now because this might trigger on some macOS
5281 * systems where applications use MAP_JIT with RW/RX permissions, and then later
5282 * switch to RWX (which will cause a switch to XNU_USER_JIT from XNU_USER_DEBUG
5283 * but the VM will still have PMAP_OPTIONS_XNU_USER_DEBUG set). If the VM can
5284 * catch this case, and remove PMAP_OPTIONS_XNU_USER_DEBUG when an application
5285 * switches to RWX, then we can start asserting this requirement.
5286 */
5287 new_frame_type = XNU_USER_JIT;
5288 } else if (__improbable(options & PMAP_OPTIONS_XNU_USER_DEBUG)) {
5289 /*
5290 * Both XNU_USER_DEBUG and XNU_USER_EXEC allow RX permissions. Given that, we must
5291 * test for PMAP_OPTIONS_XNU_USER_DEBUG before we test for XNU_USER_EXEC since the
5292 * XNU_USER_DEBUG type overlays the XNU_USER_EXEC type.
5293 */
5294 new_frame_type = XNU_USER_DEBUG;
5295 } else if (pte_to_xprr_perm(new_pte) == XPRR_USER_RX_PERM) {
5296 new_frame_type = XNU_USER_EXEC;
5297 }
5298
5299 if (__improbable(new_frame_type != XNU_DEFAULT)) {
5300 prev_frame_type = sptm_get_frame_type(pa);
5301 }
5302
5303 if (__improbable(new_frame_type != prev_frame_type)) {
5304 /**
5305 * Remove all existing mappings prior to retyping, so that we can safely retype without having to worry
5306 * about a concurrent operation on one of those mappings triggering an SPTM violation. In particular,
5307 * pmap_remove() may clear a mapping to this page without holding its PVH lock. This approach works
5308 * because we hold the PVH lock during this call, and any attempt to enter a new mapping for the page
5309 * will also need to grab the PVH lock and call this function.
5310 */
5311 pmap_page_protect_options_with_flush_range((ppnum_t)atop(pa), VM_PROT_NONE,
5312 PMAP_OPTIONS_PPO_PENDING_RETYPE, locked_pvh, NULL);
5313 /**
5314 * In the unlikely event that pmap_page_protect_options_with_flush_range() had to process
5315 * an excessively long PV list, it will have enabled preemption by placing the PVH lock
5316 * in sleep mode. In this case, we may have been migrated to a different CPU, and caller
5317 * assumptions about the state of per-CPU data (such as per-CPU PVE availability) will no
5318 * longer hold true. Ask the caller to retry by pretending we encountered a pending flush.
5319 */
5320 if (__improbable(preemption_enabled())) {
5321 return SPTM_MAP_FLUSH_PENDING;
5322 }
5323 sptm_retype_params_t retype_params = {.raw = SPTM_RETYPE_PARAMS_NULL};
5324 /* Reload the existing frame type, as pmap_page_protect_options() may have changed it back to XNU_DEFAULT. */
5325 prev_frame_type = sptm_get_frame_type(pa);
5326 sptm_retype(pa, prev_frame_type, new_frame_type, retype_params);
5327 }
5328
5329 const sptm_return_t sptm_status = sptm_map_page(pmap->ttep, v, new_pte);
5330 if (__improbable((sptm_status != SPTM_SUCCESS) && (sptm_status != SPTM_MAP_VALID))) {
5331 /*
5332 * We should always undo our previous retype, even if the SPTM returned SPTM_MAP_FLUSH_PENDING as
5333 * opposed to a TXM error. In the case of SPTM_MAP_FLUSH_PENDING, pmap_enter() will drop the PVH
5334 * lock before turning around to retry the mapping operation. It may then be possible for the
5335 * mapping state of the page to change such that our next attempt to map it will fail with a TXM
5336 * error, so if we were to leave the new type in place here we would then have lost our record
5337 * of the previous type and would effectively leave the page in an inconsistent state.
5338 */
5339 if (__improbable(new_frame_type != prev_frame_type)) {
5340 sptm_retype_params_t retype_params = {.raw = SPTM_RETYPE_PARAMS_NULL};
5341 sptm_retype(pa, new_frame_type, prev_frame_type, retype_params);
5342 }
5343 return sptm_status;
5344 }
5345
5346 *old_pte = prev_pte = PERCPU_GET(pmap_sptm_percpu)->sptm_prev_ptes[0];
5347
5348 if (prev_pte != new_pte) {
5349 changed_wiring = pte_is_compressed(prev_pte, pte_p) ?
5350 (new_pte & ARM_PTE_WIRED) != 0 :
5351 (new_pte & ARM_PTE_WIRED) != (prev_pte & ARM_PTE_WIRED);
5352
5353 if ((pmap != kernel_pmap) && changed_wiring) {
5354 pte_update_wiredcnt(pmap, pte_p, (new_pte & ARM_PTE_WIRED) != 0);
5355 }
5356
5357 PMAP_TRACE(4 + pt_attr_leaf_level(pmap_get_pt_attr(pmap)), PMAP_CODE(PMAP__TTE),
5358 VM_KERNEL_ADDRHIDE(pmap), VM_KERNEL_ADDRHIDE(v),
5359 VM_KERNEL_ADDRHIDE(v + (pt_attr_page_size(pmap_get_pt_attr(pmap)) * PAGE_RATIO)), new_pte);
5360 }
5361
5362 return sptm_status;
5363 }
5364
5365 MARK_AS_PMAP_TEXT static pt_entry_t
5366 wimg_to_pte(unsigned int wimg, pmap_paddr_t pa)
5367 {
5368 pt_entry_t pte;
5369
5370 switch (wimg & (VM_WIMG_MASK)) {
5371 case VM_WIMG_IO:
5372 // Map DRAM addresses with VM_WIMG_IO as Device-GRE instead of
5373 // Device-nGnRnE. On H14+, accesses to them can be reordered by
5374 // AP, while preserving the security benefits of using device
5375 // mapping against side-channel attacks. On pre-H14 platforms,
5376 // the accesses will still be strongly ordered.
5377 if (is_dram_addr(pa)) {
5378 pte = ARM_PTE_ATTRINDX(CACHE_ATTRINDX_POSTED_COMBINED_REORDERED);
5379 } else {
5380 pte = ARM_PTE_ATTRINDX(CACHE_ATTRINDX_DISABLE);
5381 #if HAS_FEAT_XS
5382 pmap_io_range_t *io_rgn = pmap_find_io_attr(pa);
5383 if (__improbable((io_rgn != NULL) && (io_rgn->wimg & PMAP_IO_RANGE_STRONG_SYNC))) {
5384 pte = ARM_PTE_ATTRINDX(CACHE_ATTRINDX_DISABLE_XS);
5385 }
5386 #endif /* HAS_FEAT_XS */
5387 }
5388 pte |= ARM_PTE_NX | ARM_PTE_PNX;
5389 break;
5390 case VM_WIMG_RT:
5391 pte = ARM_PTE_ATTRINDX(CACHE_ATTRINDX_RT);
5392 pte |= ARM_PTE_NX | ARM_PTE_PNX;
5393 break;
5394 case VM_WIMG_POSTED:
5395 if (is_dram_addr(pa)) {
5396 pte = ARM_PTE_ATTRINDX(CACHE_ATTRINDX_POSTED_COMBINED_REORDERED);
5397 } else {
5398 pte = ARM_PTE_ATTRINDX(CACHE_ATTRINDX_POSTED);
5399 }
5400 pte |= ARM_PTE_NX | ARM_PTE_PNX;
5401 break;
5402 case VM_WIMG_POSTED_REORDERED:
5403 if (is_dram_addr(pa)) {
5404 pte = ARM_PTE_ATTRINDX(CACHE_ATTRINDX_POSTED_COMBINED_REORDERED);
5405 } else {
5406 pte = ARM_PTE_ATTRINDX(CACHE_ATTRINDX_POSTED_REORDERED);
5407 }
5408 pte |= ARM_PTE_NX | ARM_PTE_PNX;
5409 break;
5410 case VM_WIMG_POSTED_COMBINED_REORDERED:
5411 pte = ARM_PTE_ATTRINDX(CACHE_ATTRINDX_POSTED_COMBINED_REORDERED);
5412 #if HAS_FEAT_XS
5413 if (!is_dram_addr(pa)) {
5414 pte = ARM_PTE_ATTRINDX(CACHE_ATTRINDX_POSTED_COMBINED_REORDERED_XS);
5415 }
5416 #endif /* HAS_FEAT_XS */
5417 pte |= ARM_PTE_NX | ARM_PTE_PNX;
5418 break;
5419 case VM_WIMG_WCOMB:
5420 pte = ARM_PTE_ATTRINDX(CACHE_ATTRINDX_WRITECOMB);
5421 pte |= ARM_PTE_NX | ARM_PTE_PNX;
5422 break;
5423 case VM_WIMG_WTHRU:
5424 pte = ARM_PTE_ATTRINDX(CACHE_ATTRINDX_WRITETHRU);
5425 pte |= ARM_PTE_SH(SH_OUTER_MEMORY);
5426 break;
5427 case VM_WIMG_COPYBACK:
5428 pte = ARM_PTE_ATTRINDX(CACHE_ATTRINDX_WRITEBACK);
5429 pte |= ARM_PTE_SH(SH_OUTER_MEMORY);
5430 break;
5431 case VM_WIMG_INNERWBACK:
5432 pte = ARM_PTE_ATTRINDX(CACHE_ATTRINDX_INNERWRITEBACK);
5433 pte |= ARM_PTE_SH(SH_INNER_MEMORY);
5434 break;
5435 default:
5436 pte = ARM_PTE_ATTRINDX(CACHE_ATTRINDX_DEFAULT);
5437 pte |= ARM_PTE_SH(SH_OUTER_MEMORY);
5438 }
5439
5440 return pte;
5441 }
5442
5443
5444 /*
5445 * Construct a PTE (and the physical page attributes) for the given virtual to
5446 * physical mapping.
5447 *
5448 * This function has no side effects and is safe to call so that it is safe to
5449 * call while attempting a pmap_enter transaction.
5450 */
5451 MARK_AS_PMAP_TEXT static pt_entry_t
5452 pmap_construct_pte(
5453 const pmap_t pmap,
5454 vm_map_address_t va,
5455 pmap_paddr_t pa,
5456 vm_prot_t prot,
5457 vm_prot_t fault_type,
5458 boolean_t wired,
5459 const pt_attr_t* const pt_attr,
5460 uint16_t *pp_attr_bits /* OUTPUT */
5461 )
5462 {
5463 bool set_NX = false, set_XO = false;
5464 pt_entry_t pte = pa_to_pte(pa) | ARM_PTE_TYPE;
5465 assert(pp_attr_bits != NULL);
5466 *pp_attr_bits = 0;
5467
5468 if (wired) {
5469 pte |= ARM_PTE_WIRED;
5470 }
5471
5472 #if DEVELOPMENT || DEBUG
5473 if ((prot & VM_PROT_EXECUTE) || !nx_enabled || !pmap->nx_enabled)
5474 #else
5475 if ((prot & VM_PROT_EXECUTE))
5476 #endif
5477 {
5478 set_NX = false;
5479 } else {
5480 set_NX = true;
5481 }
5482
5483 if (prot == VM_PROT_EXECUTE) {
5484 set_XO = true;
5485
5486 }
5487
5488 if (set_NX) {
5489 pte |= pt_attr_leaf_xn(pt_attr);
5490 } else {
5491 if (pmap == kernel_pmap) {
5492 pte |= ARM_PTE_NX;
5493 } else {
5494 pte |= pt_attr_leaf_x(pt_attr);
5495 }
5496 }
5497
5498 if (pmap == kernel_pmap) {
5499 #if __ARM_KERNEL_PROTECT__
5500 pte |= ARM_PTE_NG;
5501 #endif /* __ARM_KERNEL_PROTECT__ */
5502 if (prot & VM_PROT_WRITE) {
5503 pte |= ARM_PTE_AP(AP_RWNA);
5504 *pp_attr_bits |= PP_ATTR_MODIFIED | PP_ATTR_REFERENCED;
5505 } else {
5506 pte |= ARM_PTE_AP(AP_RONA);
5507 *pp_attr_bits |= PP_ATTR_REFERENCED;
5508 }
5509 } else {
5510 if (pmap->type != PMAP_TYPE_NESTED) {
5511 pte |= ARM_PTE_NG;
5512 } else if ((pmap->nested_region_unnested_table_bitmap)
5513 && (va >= pmap->nested_region_addr)
5514 && (va < (pmap->nested_region_addr + pmap->nested_region_size))) {
5515 unsigned int index = (unsigned int)((va - pmap->nested_region_addr) >> pt_attr_twig_shift(pt_attr));
5516
5517 if ((pmap->nested_region_unnested_table_bitmap)
5518 && bitmap_test(pmap->nested_region_unnested_table_bitmap, index)) {
5519 pte |= ARM_PTE_NG;
5520 }
5521 }
5522 if (prot & VM_PROT_WRITE) {
5523 assert(pmap->type != PMAP_TYPE_NESTED);
5524 if (pa_valid(pa) && (!ppattr_pa_test_bits(pa, PP_ATTR_MODIFIED))) {
5525 if (fault_type & VM_PROT_WRITE) {
5526 pte |= pt_attr_leaf_rw(pt_attr);
5527 *pp_attr_bits |= PP_ATTR_REFERENCED | PP_ATTR_MODIFIED;
5528 } else {
5529 pte |= pt_attr_leaf_ro(pt_attr);
5530 /*
5531 * Mark the page as MODFAULT so that a subsequent write
5532 * may be handled through arm_fast_fault().
5533 */
5534 *pp_attr_bits |= PP_ATTR_REFERENCED | PP_ATTR_MODFAULT;
5535 pte_set_was_writeable(pte, true);
5536 }
5537 } else {
5538 pte |= pt_attr_leaf_rw(pt_attr);
5539 *pp_attr_bits |= (PP_ATTR_REFERENCED | PP_ATTR_MODIFIED);
5540 }
5541 } else {
5542 if (set_XO) {
5543 pte |= pt_attr_leaf_rona(pt_attr);
5544 } else {
5545 pte |= pt_attr_leaf_ro(pt_attr);
5546 }
5547 *pp_attr_bits |= PP_ATTR_REFERENCED;
5548 }
5549 }
5550
5551 pte |= ARM_PTE_AF;
5552 return pte;
5553 }
5554
5555 MARK_AS_PMAP_TEXT kern_return_t
5556 pmap_enter_options_internal(
5557 pmap_t pmap,
5558 vm_map_address_t v,
5559 pmap_paddr_t pa,
5560 vm_prot_t prot,
5561 vm_prot_t fault_type,
5562 unsigned int flags,
5563 boolean_t wired,
5564 unsigned int options,
5565 pmap_mapping_type_t mapping_type)
5566 {
5567 ppnum_t pn = (ppnum_t)atop(pa);
5568 pt_entry_t *pte_p;
5569 unsigned int wimg_bits;
5570 bool committed = false;
5571 kern_return_t kr = KERN_SUCCESS;
5572 uint16_t pp_attr_bits;
5573 volatile uint16_t *wiredcnt = NULL;
5574 pv_free_list_t *local_pv_free;
5575
5576 validate_pmap_mutable(pmap);
5577
5578 /**
5579 * Prepare for the SPTM call early by prefetching the relavant FTEs. Cache misses
5580 * in SPTM accessing these turn out to contribute to a large portion of delay on
5581 * the critical path. Technically, sptm_prefetch_fte may not find an FTE associated
5582 * with pa and return LIBSPTM_FAILURE. However, we are okay with that as it's only
5583 * a best-effort performance optimization.
5584 */
5585 sptm_prefetch_fte(pmap->ttep);
5586 sptm_prefetch_fte(pa);
5587
5588 __unused const pt_attr_t * const pt_attr = pmap_get_pt_attr(pmap);
5589
5590 if ((v) & pt_attr_leaf_offmask(pt_attr)) {
5591 panic("pmap_enter_options() pmap %p v 0x%llx",
5592 pmap, (uint64_t)v);
5593 }
5594
5595 if (__improbable((pmap == kernel_pmap) && (v >= CPUWINDOWS_BASE) && (v < CPUWINDOWS_TOP))) {
5596 panic("pmap_enter_options() kernel pmap %p v 0x%llx belongs to [CPUWINDOWS_BASE: 0x%llx, CPUWINDOWS_TOP: 0x%llx)",
5597 pmap, (uint64_t)v, (uint64_t)CPUWINDOWS_BASE, (uint64_t)CPUWINDOWS_TOP);
5598 }
5599
5600 if ((pa) & pt_attr_leaf_offmask(pt_attr)) {
5601 panic("pmap_enter_options() pmap %p pa 0x%llx",
5602 pmap, (uint64_t)pa);
5603 }
5604
5605 /* The PA should not extend beyond the architected physical address space */
5606 pa &= ARM_PTE_PAGE_MASK;
5607
5608 if ((prot & VM_PROT_EXECUTE) && (pmap == kernel_pmap)) {
5609 #if defined(KERNEL_INTEGRITY_CTRR) && defined(CONFIG_XNUPOST)
5610 extern vm_offset_t ctrr_test_page;
5611 if (__probable(v != ctrr_test_page))
5612 #endif
5613 panic("pmap_enter_options(): attempt to add executable mapping to kernel_pmap");
5614 }
5615 assert(pn != vm_page_fictitious_addr);
5616
5617 pmap_lock(pmap, PMAP_LOCK_SHARED);
5618
5619 /*
5620 * Expand pmap to include this pte. Assume that
5621 * pmap is always expanded to include enough hardware
5622 * pages to map one VM page.
5623 */
5624 while ((pte_p = pmap_pte(pmap, v)) == PT_ENTRY_NULL) {
5625 /* Must unlock to expand the pmap. */
5626 pmap_unlock(pmap, PMAP_LOCK_SHARED);
5627
5628 kr = pmap_expand(pmap, v, options, pt_attr_leaf_level(pt_attr));
5629
5630 if (kr != KERN_SUCCESS) {
5631 return kr;
5632 }
5633
5634 pmap_lock(pmap, PMAP_LOCK_SHARED);
5635 }
5636
5637 if (options & PMAP_OPTIONS_NOENTER) {
5638 pmap_unlock(pmap, PMAP_LOCK_SHARED);
5639 return KERN_SUCCESS;
5640 }
5641
5642 /*
5643 * Since we may not hold the pmap lock exclusive, updating the pte is
5644 * done via a cmpxchg loop.
5645 * We need to be careful about modifying non-local data structures before commiting
5646 * the new pte since we may need to re-do the transaction.
5647 */
5648 const pt_entry_t prev_pte = os_atomic_load(pte_p, relaxed);
5649
5650 if (((prev_pte & ARM_PTE_TYPE_VALID) == ARM_PTE_TYPE) && (pte_to_pa(prev_pte) != pa)) {
5651 /*
5652 * There is already a mapping here & it's for a different physical page.
5653 * First remove that mapping.
5654 * We assume that we can leave the pmap lock held for shared access rather
5655 * than exclusive access here, because we assume that the VM won't try to
5656 * simultaneously map the same VA to multiple different physical pages.
5657 * If that assumption is violated, sptm_map_page() will panic as the architecture
5658 * does not allow the output address of a mapping to be changed without a break-
5659 * before-make sequence.
5660 */
5661 pmap_remove_range(pmap, v, v + PAGE_SIZE);
5662 }
5663
5664 if (pmap != kernel_pmap) {
5665 ptd_info_t *ptd_info = ptep_get_info(pte_p);
5666 wiredcnt = &ptd_info->wiredcnt;
5667 }
5668
5669 while (!committed) {
5670 pt_entry_t spte = ARM_PTE_TYPE_FAULT;
5671 pv_alloc_return_t pv_status = PV_ALLOC_SUCCESS;
5672 bool skip_footprint_debit = false;
5673
5674 /*
5675 * The XO index is used for TPRO mappings. To avoid exposing them as --x,
5676 * the VM code tracks VM_MAP_TPRO requests and couples them with the proper
5677 * read-write protection. The PMAP layer though still needs to use the right
5678 * index, which is the older XO-now-TPRO one and that is specially selected
5679 * here thanks to PMAP_OPTIONS_MAP_TPRO.
5680 *
5681 * Note that pmap_construct_pte() may check the nested region ASID bitmap,
5682 * which needs to happen at every iteration of the commit loop in case we
5683 * previously dropped the pmap lock.
5684 */
5685 pt_entry_t pte = pmap_construct_pte(pmap, v, pa,
5686 ((options & PMAP_OPTIONS_MAP_TPRO) ? VM_PROT_RORW_TP : prot), fault_type, wired, pt_attr, &pp_attr_bits);
5687
5688
5689 if (pa_valid(pa)) {
5690 unsigned int pai;
5691 boolean_t is_altacct = FALSE, is_internal = FALSE, is_reusable = FALSE, is_external = FALSE;
5692
5693 is_internal = FALSE;
5694 is_altacct = FALSE;
5695
5696 pai = pa_index(pa);
5697 locked_pvh_t locked_pvh;
5698
5699 if (__improbable(options & PMAP_OPTIONS_NOPREEMPT)) {
5700 locked_pvh = pvh_lock_nopreempt(pai);
5701 } else {
5702 locked_pvh = pvh_lock(pai);
5703 }
5704
5705 /*
5706 * Make sure that the current per-cpu PV free list has
5707 * enough entries (2 in the worst-case scenario) to handle the enter_pv
5708 * if the transaction succeeds. At this point, preemption has either
5709 * been disabled by the caller or by pvh_lock() above.
5710 * Note that we can still be interrupted, but a primary
5711 * interrupt handler can never enter the pmap.
5712 */
5713 assert(get_preemption_level() > 0);
5714 local_pv_free = &pmap_get_cpu_data()->pv_free;
5715 const bool allocation_required = !pvh_test_type(locked_pvh.pvh, PVH_TYPE_NULL) &&
5716 !(pvh_test_type(locked_pvh.pvh, PVH_TYPE_PTEP) && pvh_ptep(locked_pvh.pvh) == pte_p);
5717
5718 if (__improbable(allocation_required && (local_pv_free->count < 2))) {
5719 pv_entry_t *new_pve_p[2] = {PV_ENTRY_NULL};
5720 int new_allocated_pves = 0;
5721
5722 while (new_allocated_pves < 2) {
5723 local_pv_free = &pmap_get_cpu_data()->pv_free;
5724 pv_status = pv_alloc(pmap, PMAP_LOCK_SHARED, options, &new_pve_p[new_allocated_pves], &locked_pvh, wiredcnt);
5725 if (pv_status == PV_ALLOC_FAIL) {
5726 break;
5727 } else if (pv_status == PV_ALLOC_RETRY) {
5728 /*
5729 * In the case that pv_alloc() had to grab a new page of PVEs,
5730 * it will have dropped the pmap lock while doing so.
5731 * On non-PPL devices, dropping the lock re-enables preemption so we may
5732 * be on a different CPU now.
5733 */
5734 local_pv_free = &pmap_get_cpu_data()->pv_free;
5735 } else {
5736 /* If we've gotten this far then a node should've been allocated. */
5737 assert(new_pve_p[new_allocated_pves] != PV_ENTRY_NULL);
5738
5739 new_allocated_pves++;
5740 }
5741 }
5742
5743 for (int i = 0; i < new_allocated_pves; i++) {
5744 pv_free(new_pve_p[i]);
5745 }
5746 }
5747
5748 if (pv_status == PV_ALLOC_FAIL) {
5749 pvh_unlock(&locked_pvh);
5750 kr = KERN_RESOURCE_SHORTAGE;
5751 break;
5752 } else if (pv_status == PV_ALLOC_RETRY) {
5753 pvh_unlock(&locked_pvh);
5754 /* We dropped the pmap and PVH locks to allocate. Retry transaction. */
5755 continue;
5756 }
5757
5758 if ((flags & (VM_WIMG_MASK | VM_WIMG_USE_DEFAULT))) {
5759 wimg_bits = (flags & (VM_WIMG_MASK | VM_WIMG_USE_DEFAULT));
5760 } else {
5761 wimg_bits = pmap_cache_attributes(pn);
5762 }
5763
5764 /**
5765 * We may be retrying this operation after dropping the PVH lock.
5766 * Cache attributes for the physical page may have changed while the lock
5767 * was dropped, so update PTE cache attributes on each loop iteration.
5768 */
5769 pte |= pmap_get_pt_ops(pmap)->wimg_to_pte(wimg_bits, pa);
5770
5771
5772 const sptm_return_t sptm_status = pmap_enter_pte(pmap, pte_p, pte, &locked_pvh, &spte, v, options, mapping_type);
5773 assert(committed == false);
5774 if ((sptm_status == SPTM_SUCCESS) || (sptm_status == SPTM_MAP_VALID)) {
5775 committed = true;
5776 } else if (sptm_status == SPTM_MAP_FLUSH_PENDING) {
5777 pvh_unlock(&locked_pvh);
5778 continue;
5779 } else if (sptm_status == SPTM_MAP_CODESIGN_ERROR) {
5780 pvh_unlock(&locked_pvh);
5781 kr = KERN_CODESIGN_ERROR;
5782 break;
5783 } else {
5784 pvh_unlock(&locked_pvh);
5785 kr = KERN_FAILURE;
5786 break;
5787 }
5788 const bool had_valid_mapping = (sptm_status == SPTM_MAP_VALID);
5789 /* End of transaction. Commit pv changes, pa bits, and memory accounting. */
5790 if (!had_valid_mapping) {
5791 pv_entry_t *new_pve_p = PV_ENTRY_NULL;
5792 int pve_ptep_idx = 0;
5793 pv_status = pmap_enter_pv(pmap, pte_p, options, PMAP_LOCK_SHARED, &locked_pvh, &new_pve_p, &pve_ptep_idx);
5794 /* We did all the allocations up top. So this shouldn't be able to fail. */
5795 if (pv_status != PV_ALLOC_SUCCESS) {
5796 panic("%s: unexpected pmap_enter_pv ret code: %d. new_pve_p=%p pmap=%p",
5797 __func__, pv_status, new_pve_p, pmap);
5798 }
5799
5800 if (pmap != kernel_pmap) {
5801 if (options & PMAP_OPTIONS_INTERNAL) {
5802 ppattr_pve_set_internal(pai, new_pve_p, pve_ptep_idx);
5803 if ((options & PMAP_OPTIONS_ALT_ACCT) ||
5804 PMAP_FOOTPRINT_SUSPENDED(pmap)) {
5805 /*
5806 * Make a note to ourselves that this
5807 * mapping is using alternative
5808 * accounting. We'll need this in order
5809 * to know which ledger to debit when
5810 * the mapping is removed.
5811 *
5812 * The altacct bit must be set while
5813 * the pv head is locked. Defer the
5814 * ledger accounting until after we've
5815 * dropped the lock.
5816 */
5817 ppattr_pve_set_altacct(pai, new_pve_p, pve_ptep_idx);
5818 is_altacct = TRUE;
5819 }
5820 }
5821 if (ppattr_test_reusable(pai) &&
5822 !is_altacct) {
5823 is_reusable = TRUE;
5824 } else if (options & PMAP_OPTIONS_INTERNAL) {
5825 is_internal = TRUE;
5826 } else {
5827 is_external = TRUE;
5828 }
5829 }
5830 }
5831
5832 pvh_unlock(&locked_pvh);
5833
5834 if (pp_attr_bits != 0) {
5835 ppattr_pa_set_bits(pa, pp_attr_bits);
5836 }
5837
5838 if (!had_valid_mapping && (pmap != kernel_pmap)) {
5839 pmap_ledger_credit(pmap, task_ledgers.phys_mem, pt_attr_page_size(pt_attr) * PAGE_RATIO);
5840
5841 if (is_internal) {
5842 /*
5843 * Make corresponding adjustments to
5844 * phys_footprint statistics.
5845 */
5846 pmap_ledger_credit(pmap, task_ledgers.internal, pt_attr_page_size(pt_attr) * PAGE_RATIO);
5847 if (is_altacct) {
5848 /*
5849 * If this page is internal and
5850 * in an IOKit region, credit
5851 * the task's total count of
5852 * dirty, internal IOKit pages.
5853 * It should *not* count towards
5854 * the task's total physical
5855 * memory footprint, because
5856 * this entire region was
5857 * already billed to the task
5858 * at the time the mapping was
5859 * created.
5860 *
5861 * Put another way, this is
5862 * internal++ and
5863 * alternate_accounting++, so
5864 * net effect on phys_footprint
5865 * is 0. That means: don't
5866 * touch phys_footprint here.
5867 */
5868 pmap_ledger_credit(pmap, task_ledgers.alternate_accounting, pt_attr_page_size(pt_attr) * PAGE_RATIO);
5869 } else {
5870 if (pte_is_compressed(spte, pte_p) && !(spte & ARM_PTE_COMPRESSED_ALT)) {
5871 /* Replacing a compressed page (with internal accounting). No change to phys_footprint. */
5872 skip_footprint_debit = true;
5873 } else {
5874 pmap_ledger_credit(pmap, task_ledgers.phys_footprint, pt_attr_page_size(pt_attr) * PAGE_RATIO);
5875 }
5876 }
5877 }
5878 if (is_reusable) {
5879 pmap_ledger_credit(pmap, task_ledgers.reusable, pt_attr_page_size(pt_attr) * PAGE_RATIO);
5880 } else if (is_external) {
5881 pmap_ledger_credit(pmap, task_ledgers.external, pt_attr_page_size(pt_attr) * PAGE_RATIO);
5882 }
5883 }
5884 } else {
5885 if (prot & VM_PROT_EXECUTE) {
5886 kr = KERN_FAILURE;
5887 break;
5888 }
5889
5890 wimg_bits = pmap_cache_attributes(pn);
5891 if ((flags & (VM_WIMG_MASK | VM_WIMG_USE_DEFAULT))) {
5892 wimg_bits = (wimg_bits & (~VM_WIMG_MASK)) | (flags & (VM_WIMG_MASK | VM_WIMG_USE_DEFAULT));
5893 }
5894
5895 pte |= pmap_get_pt_ops(pmap)->wimg_to_pte(wimg_bits, pa);
5896
5897
5898 /**
5899 * pmap_enter_pte() expects to be called with preemption disabled so it can access
5900 * the per-CPU prev_ptes array.
5901 */
5902 disable_preemption();
5903 const sptm_return_t sptm_status = pmap_enter_pte(pmap, pte_p, pte, NULL, &spte, v, options, mapping_type);
5904 enable_preemption();
5905 assert(committed == false);
5906 if ((sptm_status == SPTM_SUCCESS) || (sptm_status == SPTM_MAP_VALID)) {
5907 committed = true;
5908
5909 /**
5910 * If there was already a valid pte here then we reuse its
5911 * reference on the ptd and drop the one that we took above.
5912 */
5913 } else if (__improbable(sptm_status != SPTM_MAP_FLUSH_PENDING)) {
5914 panic("%s: Unexpected SPTM return code %u for non-managed PA 0x%llx", __func__, (unsigned int)sptm_status, (unsigned long long)pa);
5915 }
5916 }
5917 if (committed) {
5918 if (pte_is_compressed(spte, pte_p)) {
5919 assert(pmap != kernel_pmap);
5920
5921 /* One less "compressed" */
5922 pmap_ledger_debit(pmap, task_ledgers.internal_compressed,
5923 pt_attr_page_size(pt_attr) * PAGE_RATIO);
5924
5925 if (spte & ARM_PTE_COMPRESSED_ALT) {
5926 pmap_ledger_debit(pmap, task_ledgers.alternate_accounting_compressed, pt_attr_page_size(pt_attr) * PAGE_RATIO);
5927 } else if (!skip_footprint_debit) {
5928 /* Was part of the footprint */
5929 pmap_ledger_debit(pmap, task_ledgers.phys_footprint, pt_attr_page_size(pt_attr) * PAGE_RATIO);
5930 }
5931 }
5932 }
5933 }
5934
5935 pmap_unlock(pmap, PMAP_LOCK_SHARED);
5936
5937 if (kr == KERN_CODESIGN_ERROR) {
5938 /* Print any logs from TXM */
5939 txm_print_logs();
5940 }
5941 return kr;
5942 }
5943
5944 kern_return_t
5945 pmap_enter_options_addr(
5946 pmap_t pmap,
5947 vm_map_address_t v,
5948 pmap_paddr_t pa,
5949 vm_prot_t prot,
5950 vm_prot_t fault_type,
5951 unsigned int flags,
5952 boolean_t wired,
5953 unsigned int options,
5954 __unused void *arg,
5955 pmap_mapping_type_t mapping_type)
5956 {
5957 kern_return_t kr = KERN_FAILURE;
5958
5959
5960 PMAP_TRACE(2, PMAP_CODE(PMAP__ENTER) | DBG_FUNC_START,
5961 VM_KERNEL_ADDRHIDE(pmap), VM_KERNEL_ADDRHIDE(v), pa, prot);
5962
5963 kr = pmap_enter_options_internal(pmap, v, pa, prot, fault_type, flags, wired, options, mapping_type);
5964
5965 PMAP_TRACE(2, PMAP_CODE(PMAP__ENTER) | DBG_FUNC_END, kr);
5966
5967 return kr;
5968 }
5969
5970 kern_return_t
5971 pmap_enter_options(
5972 pmap_t pmap,
5973 vm_map_address_t v,
5974 ppnum_t pn,
5975 vm_prot_t prot,
5976 vm_prot_t fault_type,
5977 unsigned int flags,
5978 boolean_t wired,
5979 unsigned int options,
5980 __unused void *arg,
5981 pmap_mapping_type_t mapping_type)
5982 {
5983 return pmap_enter_options_addr(pmap, v, ((pmap_paddr_t)pn) << PAGE_SHIFT, prot,
5984 fault_type, flags, wired, options, arg, mapping_type);
5985 }
5986
5987 /*
5988 * Routine: pmap_change_wiring
5989 * Function: Change the wiring attribute for a map/virtual-address
5990 * pair.
5991 * In/out conditions:
5992 * The mapping must already exist in the pmap.
5993 */
5994 MARK_AS_PMAP_TEXT void
5995 pmap_change_wiring_internal(
5996 pmap_t pmap,
5997 vm_map_address_t v,
5998 boolean_t wired)
5999 {
6000 pt_entry_t *pte_p, prev_pte;
6001
6002 validate_pmap_mutable(pmap);
6003
6004 pmap_lock(pmap, PMAP_LOCK_SHARED);
6005
6006 const pt_entry_t new_wiring = (wired ? ARM_PTE_WIRED : 0);
6007
6008 pte_p = pmap_pte(pmap, v);
6009 if (pte_p == PT_ENTRY_NULL) {
6010 if (!wired) {
6011 /*
6012 * The PTE may have already been cleared by a disconnect/remove operation, and the L3 table
6013 * may have been freed by a remove operation.
6014 */
6015 goto pmap_change_wiring_return;
6016 } else {
6017 panic("%s: Attempt to wire nonexistent PTE for pmap %p", __func__, pmap);
6018 }
6019 }
6020
6021 disable_preemption();
6022 pmap_sptm_percpu_data_t *sptm_pcpu = PERCPU_GET(pmap_sptm_percpu);
6023 sptm_pcpu->sptm_templates[0] = (*pte_p & ~ARM_PTE_WIRED) | new_wiring;
6024
6025 pmap_retype_epoch_enter();
6026 sptm_update_region(pmap->ttep, v, 1, sptm_pcpu->sptm_templates_pa, SPTM_UPDATE_SW_WIRED);
6027 pmap_retype_epoch_exit();
6028
6029 prev_pte = os_atomic_load(&sptm_pcpu->sptm_prev_ptes[0], relaxed);
6030 enable_preemption();
6031
6032 if ((prev_pte & ARM_PTE_TYPE_MASK) == ARM_PTE_TYPE_FAULT) {
6033 goto pmap_change_wiring_return;
6034 }
6035
6036 if ((pmap != kernel_pmap) && (wired != pte_is_wired(prev_pte))) {
6037 pte_update_wiredcnt(pmap, pte_p, wired);
6038 }
6039
6040 pmap_change_wiring_return:
6041 pmap_unlock(pmap, PMAP_LOCK_SHARED);
6042 }
6043
6044 void
6045 pmap_change_wiring(
6046 pmap_t pmap,
6047 vm_map_address_t v,
6048 boolean_t wired)
6049 {
6050 pmap_change_wiring_internal(pmap, v, wired);
6051 }
6052
6053 MARK_AS_PMAP_TEXT pmap_paddr_t
6054 pmap_find_pa_internal(
6055 pmap_t pmap,
6056 addr64_t va)
6057 {
6058 pmap_paddr_t pa = 0;
6059
6060 validate_pmap(pmap);
6061
6062 if (pmap != kernel_pmap) {
6063 pmap_lock(pmap, PMAP_LOCK_SHARED);
6064 }
6065
6066 pa = pmap_vtophys(pmap, va);
6067
6068 if (pmap != kernel_pmap) {
6069 pmap_unlock(pmap, PMAP_LOCK_SHARED);
6070 }
6071
6072 return pa;
6073 }
6074
6075 pmap_paddr_t
6076 pmap_find_pa_nofault(pmap_t pmap, addr64_t va)
6077 {
6078 pmap_paddr_t pa = 0;
6079
6080 if (pmap == kernel_pmap) {
6081 pa = mmu_kvtop(va);
6082 } else if ((current_thread()->map) && (pmap == vm_map_pmap(current_thread()->map))) {
6083 /*
6084 * Note that this doesn't account for PAN: mmu_uvtop() may return a valid
6085 * translation even if PAN would prevent kernel access through the translation.
6086 * It's therefore assumed the UVA will be accessed in a PAN-disabled context.
6087 */
6088 pa = mmu_uvtop(va);
6089 }
6090 return pa;
6091 }
6092
6093 pmap_paddr_t
6094 pmap_find_pa(
6095 pmap_t pmap,
6096 addr64_t va)
6097 {
6098 pmap_paddr_t pa = pmap_find_pa_nofault(pmap, va);
6099
6100 if (pa != 0) {
6101 return pa;
6102 }
6103
6104 if (not_in_kdp) {
6105 return pmap_find_pa_internal(pmap, va);
6106 } else {
6107 return pmap_vtophys(pmap, va);
6108 }
6109 }
6110
6111 ppnum_t
6112 pmap_find_phys_nofault(
6113 pmap_t pmap,
6114 addr64_t va)
6115 {
6116 ppnum_t ppn;
6117 ppn = atop(pmap_find_pa_nofault(pmap, va));
6118 return ppn;
6119 }
6120
6121 ppnum_t
6122 pmap_find_phys(
6123 pmap_t pmap,
6124 addr64_t va)
6125 {
6126 ppnum_t ppn;
6127 ppn = atop(pmap_find_pa(pmap, va));
6128 return ppn;
6129 }
6130
6131 /**
6132 * Translate a kernel virtual address into a physical address.
6133 *
6134 * @param va The kernel virtual address to translate. Does not work on user
6135 * virtual addresses.
6136 *
6137 * @return The physical address if the translation was successful, or zero if
6138 * no valid mappings were found for the given virtual address.
6139 */
6140 pmap_paddr_t
6141 kvtophys(vm_offset_t va)
6142 {
6143 sptm_paddr_t pa;
6144
6145 if (sptm_kvtophys(va, &pa) != LIBSPTM_SUCCESS) {
6146 return 0;
6147 }
6148
6149 return pa;
6150 }
6151
6152 /**
6153 * Variant of kvtophys that can't fail. If no mapping is found or the mapping
6154 * points to a non-kernel-managed physical page, then this call will panic().
6155 *
6156 * @note The output of this function is guaranteed to be a kernel-managed
6157 * physical page, which means it's safe to pass the output directly to
6158 * pa_index() to create a physical address index for various pmap data
6159 * structures.
6160 *
6161 * @param va The kernel virtual address to translate. Does not work on user
6162 * virtual addresses.
6163 *
6164 * @return The translated physical address for the given virtual address.
6165 */
6166 pmap_paddr_t
6167 kvtophys_nofail(vm_offset_t va)
6168 {
6169 pmap_paddr_t pa;
6170
6171 if (__improbable(sptm_kvtophys(va, &pa) != LIBSPTM_SUCCESS)) {
6172 panic("%s: VA->PA translation failed for va %p", __func__, (void *)va);
6173 }
6174
6175 return pa;
6176 }
6177
6178 pmap_paddr_t
6179 pmap_vtophys(
6180 pmap_t pmap,
6181 addr64_t va)
6182 {
6183 if ((va < pmap->min) || (va >= pmap->max)) {
6184 return 0;
6185 }
6186
6187 const pt_attr_t * const pt_attr = pmap_get_pt_attr(pmap);
6188
6189 tt_entry_t * ttp = NULL;
6190 tt_entry_t * ttep = NULL;
6191 tt_entry_t tte = ARM_TTE_EMPTY;
6192 pmap_paddr_t pa = 0;
6193 unsigned int cur_level;
6194
6195 ttp = pmap->tte;
6196
6197 for (cur_level = pt_attr_root_level(pt_attr); cur_level <= pt_attr_leaf_level(pt_attr); cur_level++) {
6198 ttep = &ttp[ttn_index(pt_attr, va, cur_level)];
6199
6200 tte = *ttep;
6201
6202 const uint64_t valid_mask = pt_attr->pta_level_info[cur_level].valid_mask;
6203 const uint64_t type_mask = pt_attr->pta_level_info[cur_level].type_mask;
6204 const uint64_t type_block = pt_attr->pta_level_info[cur_level].type_block;
6205 const uint64_t offmask = pt_attr->pta_level_info[cur_level].offmask;
6206
6207 if ((tte & valid_mask) != valid_mask) {
6208 return (pmap_paddr_t) 0;
6209 }
6210
6211 /* This detects both leaf entries and intermediate block mappings. */
6212 if ((tte & type_mask) == type_block) {
6213 pa = ((tte & ARM_TTE_PA_MASK & ~offmask) | (va & offmask));
6214 break;
6215 }
6216
6217 ttp = (tt_entry_t*)phystokv(tte & ARM_TTE_TABLE_MASK);
6218 }
6219
6220 return pa;
6221 }
6222
6223 /*
6224 * pmap_init_pte_page - Initialize a page table page.
6225 */
6226 MARK_AS_PMAP_TEXT void
6227 pmap_init_pte_page(
6228 pmap_t pmap,
6229 pt_entry_t *pte_p,
6230 vm_offset_t va,
6231 unsigned int ttlevel,
6232 boolean_t alloc_ptd)
6233 {
6234 pt_desc_t *ptdp = NULL;
6235 unsigned int pai = pa_index(kvtophys_nofail((vm_offset_t)pte_p));
6236 const uintptr_t pvh = pai_to_pvh(pai);
6237
6238 if (pvh_test_type(pvh, PVH_TYPE_NULL)) {
6239 if (alloc_ptd) {
6240 /*
6241 * This path should only be invoked from arm_vm_init. If we are emulating 16KB pages
6242 * on 4KB hardware, we may already have allocated a page table descriptor for a
6243 * bootstrap request, so we check for an existing PTD here.
6244 */
6245 ptdp = ptd_alloc(pmap, PMAP_PAGE_ALLOCATE_NOWAIT);
6246 if (ptdp == NULL) {
6247 panic("%s: unable to allocate PTD", __func__);
6248 }
6249 locked_pvh_t locked_pvh = pvh_lock(pai);
6250 pvh_update_head(&locked_pvh, ptdp, PVH_TYPE_PTDP);
6251 pvh_unlock(&locked_pvh);
6252 } else {
6253 panic("pmap_init_pte_page(): no PTD for pte_p %p", pte_p);
6254 }
6255 } else if (pvh_test_type(pvh, PVH_TYPE_PTDP)) {
6256 ptdp = pvh_ptd(pvh);
6257 } else {
6258 panic("pmap_init_pte_page(): invalid PVH type for pte_p %p", pte_p);
6259 }
6260
6261 // pagetable zero-fill and barrier should be guaranteed by the SPTM
6262 ptd_info_init(ptdp, pmap, va, ttlevel, pte_p);
6263 }
6264
6265 /*
6266 * This function guarantees that a pmap has the necessary page tables in place
6267 * to map the specified VA. If necessary, it will allocate new tables at any
6268 * non-root level in the hierarchy (the root table is always already allocated
6269 * and stored in the pmap).
6270 *
6271 * @note This function is expected to be called without any pmap or PVH lock
6272 * held.
6273 *
6274 * @note It is possible for an L3 table newly allocated by this function to be
6275 * deleted by another thread before control returns to the caller, iff that
6276 * table is an ordinary userspace table. Callers that use this function
6277 * to allocate new user L3 tables are therefore expected to keep calling
6278 * this function until they observe a successful L3 PTE lookup with the pmap
6279 * lock held. As long as it does not drop the pmap lock, the caller may
6280 * then safely use the looked-up L3 table. See the use of this function in
6281 * pmap_enter_options_internal() for an example.
6282 *
6283 * @param pmap The pmap for which to ensure mapping space is present.
6284 * @param v The virtual address for which to ensure mapping space is present
6285 * in [pmap].
6286 * @param options Flags to pass to pmap_tt_allocate() if a new table needs to be
6287 * allocated. The only valid option is PMAP_OPTIONS_NOWAIT, which
6288 * specifies that the allocation must not block.
6289 * @param level The maximum paging level for which to ensure a table is present.
6290 *
6291 * @return KERN_INVALID_ADDRESS if [v] is outside the pmap's mappable range,
6292 * KERN_RESOURCE_SHORTAGE if a new table can't be allocated,
6293 * KERN_SUCCESS otherwise.
6294 */
6295 MARK_AS_PMAP_TEXT static kern_return_t
6296 pmap_expand(
6297 pmap_t pmap,
6298 vm_map_address_t vaddr,
6299 unsigned int options,
6300 unsigned int level)
6301 {
6302 __unused const pt_attr_t * const pt_attr = pmap_get_pt_attr(pmap);
6303
6304 if (__improbable((vaddr < pmap->min) || (vaddr >= pmap->max))) {
6305 return KERN_INVALID_ADDRESS;
6306 }
6307 pmap_paddr_t pa;
6308 const uint64_t pmap_page_size = pt_attr_page_size(pt_attr);
6309 const uint64_t table_align_mask = (PAGE_SIZE / pmap_page_size) - 1;
6310 unsigned int ttlevel = pt_attr_root_level(pt_attr);
6311 tt_entry_t *table_ttep = pmap->tte;
6312 tt_entry_t *ttep;
6313 tt_entry_t old_tte = ARM_TTE_EMPTY;
6314
6315 pa = 0x0ULL;
6316
6317 for (; ttlevel < level; ttlevel++) {
6318 /**
6319 * If the previous iteration didn't allocate a new table, obtain the table from the previous TTE.
6320 * Doing this step at the beginning of the loop instead of the end (which would make it part of
6321 * the prior iteration) avoids the possibility of executing this step to extract an L3 table KVA
6322 * from an L2 TTE, which would be useless because there would be no next iteration to make use
6323 * of the table KVA.
6324 */
6325 if (table_ttep == NULL) {
6326 assert((old_tte & (ARM_TTE_TYPE_MASK | ARM_TTE_VALID)) == (ARM_TTE_TYPE_TABLE | ARM_TTE_VALID));
6327 table_ttep = (tt_entry_t*)phystokv(old_tte & ARM_TTE_TABLE_MASK);
6328 }
6329
6330 vm_map_address_t v = pt_attr_align_va(pt_attr, ttlevel, vaddr);
6331
6332 /**
6333 * We don't need to hold the pmap lock while walking the paging hierarchy. Only L3 tables are
6334 * allowed to be dynamically removed, and only for regular user pmaps at that. We may allocate
6335 * a new L3 table below, but we will only access L0-L2 tables, so there's no risk of a table
6336 * being deleted while we are using it for the next level(s) of lookup.
6337 */
6338 ttep = &table_ttep[ttn_index(pt_attr, vaddr, ttlevel)];
6339 old_tte = os_atomic_load(ttep, relaxed);
6340 table_ttep = NULL;
6341 if ((old_tte & (ARM_TTE_TYPE_MASK | ARM_TTE_VALID)) != (ARM_TTE_TYPE_TABLE | ARM_TTE_VALID)) {
6342 tt_entry_t new_tte, *new_ttep;
6343 while (pmap_tt_allocate(pmap, &new_ttep, ttlevel + 1, options | PMAP_PAGE_NOZEROFILL) != KERN_SUCCESS) {
6344 if (options & PMAP_OPTIONS_NOWAIT) {
6345 return KERN_RESOURCE_SHORTAGE;
6346 }
6347 VM_PAGE_WAIT();
6348 }
6349 /* Grab the pmap lock to ensure we don't try to concurrently map different tables at the same TTE. */
6350 pmap_lock(pmap, PMAP_LOCK_EXCLUSIVE);
6351 old_tte = os_atomic_load(ttep, relaxed);
6352 if ((old_tte & (ARM_TTE_TYPE_MASK | ARM_TTE_VALID)) != (ARM_TTE_TYPE_TABLE | ARM_TTE_VALID)) {
6353 pmap_init_pte_page(pmap, (pt_entry_t *) new_ttep, v, ttlevel + 1, FALSE);
6354 pa = kvtophys_nofail((vm_offset_t)new_ttep);
6355 /*
6356 * If the table is going to map a kernel RO zone VA region, then we must
6357 * upgrade its SPTM type to XNU_PAGE_TABLE_ROZONE. The SPTM's type system
6358 * requires the table to be transitioned through XNU_DEFAULT for refcount
6359 * enforcement, which is fine since this path is expected to execute only
6360 * once during boot.
6361 */
6362 if (__improbable(ttlevel == pt_attr_twig_level(pt_attr)) &&
6363 (pmap == kernel_pmap) && zone_spans_ro_va(vaddr, vaddr + PAGE_SIZE)) {
6364 sptm_retype_params_t retype_params = {.raw = SPTM_RETYPE_PARAMS_NULL};
6365 sptm_retype(pa, XNU_PAGE_TABLE, XNU_DEFAULT, retype_params);
6366 retype_params.level = (sptm_pt_level_t)pt_attr_leaf_level(pt_attr);
6367 sptm_retype(pa, XNU_DEFAULT, XNU_PAGE_TABLE_ROZONE, retype_params);
6368 }
6369 new_tte = (pa & ARM_TTE_TABLE_MASK) | ARM_TTE_TYPE_TABLE | ARM_TTE_VALID;
6370 sptm_map_table(pmap->ttep, v, (sptm_pt_level_t)ttlevel, new_tte);
6371 PMAP_TRACE(4 + ttlevel, PMAP_CODE(PMAP__TTE), VM_KERNEL_ADDRHIDE(pmap), VM_KERNEL_ADDRHIDE(v & ~pt_attr_ln_offmask(pt_attr, ttlevel)),
6372 VM_KERNEL_ADDRHIDE((v & ~pt_attr_ln_offmask(pt_attr, ttlevel)) + pt_attr_ln_size(pt_attr, ttlevel)), new_tte);
6373 /**
6374 * If we need to set up multiple TTEs mapping different parts of the same page
6375 * (e.g. because we're carving multiple 4K page tables out of a 16K native page,
6376 * determine which of the grouped TTEs is the one that we need to follow for the
6377 * next level of the table walk.
6378 */
6379 table_ttep = new_ttep + ((((uintptr_t)ttep / sizeof(tt_entry_t)) & table_align_mask) *
6380 (pmap_page_size / sizeof(tt_entry_t)));
6381 pa = 0x0ULL;
6382 new_ttep = (tt_entry_t *)NULL;
6383 }
6384 pmap_unlock(pmap, PMAP_LOCK_EXCLUSIVE);
6385
6386 if (new_ttep != (tt_entry_t *)NULL) {
6387 pmap_tt_deallocate(pmap, new_ttep, ttlevel + 1);
6388 new_ttep = (tt_entry_t *)NULL;
6389 }
6390 }
6391 }
6392
6393 return KERN_SUCCESS;
6394 }
6395
6396 /*
6397 * Routine: pmap_gc
6398 * Function:
6399 * Pmap garbage collection
6400 * Called by the pageout daemon when pages are scarce.
6401 *
6402 */
6403 void
6404 pmap_gc(void)
6405 {
6406 /*
6407 * TODO: as far as I can tell this has never been implemented to do anything meaninful.
6408 * We can't just destroy any old pmap on the chance that it may be active on a CPU
6409 * or may contain wired mappings. However, it may make sense to scan the pmap VM
6410 * object here, and for each page consult the SPTM frame table and if necessary
6411 * the PTD in the PV head table. If the frame table indicates the page is a leaf
6412 * page table page and the PTD indicates it has no wired mappings, we can call
6413 * pmap_remove() on the VA region mapped by the page and therein return the page
6414 * to the VM.
6415 */
6416 }
6417
6418 /*
6419 * By default, don't attempt pmap GC more frequently
6420 * than once / 1 minutes.
6421 */
6422
6423 void
6424 compute_pmap_gc_throttle(
6425 void *arg __unused)
6426 {
6427 }
6428
6429 /*
6430 * pmap_attribute_cache_sync(vm_offset_t pa)
6431 *
6432 * Invalidates all of the instruction cache on a physical page and
6433 * pushes any dirty data from the data cache for the same physical page
6434 */
6435
6436 kern_return_t
6437 pmap_attribute_cache_sync(
6438 ppnum_t pp,
6439 vm_size_t size,
6440 __unused vm_machine_attribute_t attribute,
6441 __unused vm_machine_attribute_val_t * value)
6442 {
6443 if (size > PAGE_SIZE) {
6444 panic("pmap_attribute_cache_sync size: 0x%llx", (uint64_t)size);
6445 } else {
6446 cache_sync_page(pp);
6447 }
6448
6449 return KERN_SUCCESS;
6450 }
6451
6452 /*
6453 * pmap_sync_page_data_phys(ppnum_t pp)
6454 *
6455 * Invalidates all of the instruction cache on a physical page and
6456 * pushes any dirty data from the data cache for the same physical page.
6457 * Not required on SPTM systems, because the SPTM automatically performs
6458 * the invalidate operation when retyping to one of the types that allow
6459 * for executable permissions.
6460 */
6461 void
6462 pmap_sync_page_data_phys(
6463 __unused ppnum_t pp)
6464 {
6465 return;
6466 }
6467
6468 /*
6469 * pmap_sync_page_attributes_phys(ppnum_t pp)
6470 *
6471 * Write back and invalidate all cachelines on a physical page.
6472 */
6473 void
6474 pmap_sync_page_attributes_phys(
6475 ppnum_t pp)
6476 {
6477 flush_dcache((vm_offset_t) (pp << PAGE_SHIFT), PAGE_SIZE, TRUE);
6478 }
6479
6480 #if CONFIG_COREDUMP
6481 /* temporary workaround */
6482 boolean_t
6483 coredumpok(
6484 vm_map_t map,
6485 mach_vm_offset_t va)
6486 {
6487 pt_entry_t *pte_p;
6488 pt_entry_t spte;
6489
6490 pte_p = pmap_pte(map->pmap, va);
6491 if (0 == pte_p) {
6492 return FALSE;
6493 }
6494 if (vm_map_entry_has_device_pager(map, va)) {
6495 return FALSE;
6496 }
6497 spte = *pte_p;
6498 return (spte & ARM_PTE_ATTRINDXMASK) == ARM_PTE_ATTRINDX(CACHE_ATTRINDX_DEFAULT);
6499 }
6500 #endif
6501
6502 void
6503 fillPage(
6504 ppnum_t pn,
6505 unsigned int fill)
6506 {
6507 unsigned int *addr;
6508 int count;
6509
6510 addr = (unsigned int *) phystokv(ptoa(pn));
6511 count = PAGE_SIZE / sizeof(unsigned int);
6512 while (count--) {
6513 *addr++ = fill;
6514 }
6515 }
6516
6517 extern void mapping_set_mod(ppnum_t pn);
6518
6519 void
6520 mapping_set_mod(
6521 ppnum_t pn)
6522 {
6523 pmap_set_modify(pn);
6524 }
6525
6526 extern void mapping_set_ref(ppnum_t pn);
6527
6528 void
6529 mapping_set_ref(
6530 ppnum_t pn)
6531 {
6532 pmap_set_reference(pn);
6533 }
6534
6535 /*
6536 * Clear specified attribute bits.
6537 *
6538 * Try to force an arm_fast_fault() for all mappings of
6539 * the page - to force attributes to be set again at fault time.
6540 * If the forcing succeeds, clear the cached bits at the head.
6541 * Otherwise, something must have been wired, so leave the cached
6542 * attributes alone.
6543 */
6544 MARK_AS_PMAP_TEXT static void
6545 phys_attribute_clear_with_flush_range(
6546 ppnum_t pn,
6547 unsigned int bits,
6548 int options,
6549 void *arg,
6550 pmap_tlb_flush_range_t *flush_range)
6551 {
6552 pmap_paddr_t pa = ptoa(pn);
6553 vm_prot_t allow_mode = VM_PROT_ALL;
6554
6555 if ((arg != NULL) || (flush_range != NULL)) {
6556 options = options & ~PMAP_OPTIONS_NOFLUSH;
6557 }
6558
6559 if (__improbable((options & PMAP_OPTIONS_FF_WIRED) != 0)) {
6560 panic("phys_attribute_clear(%#010x,%#010x,%#010x,%p,%p): "
6561 "invalid options",
6562 pn, bits, options, arg, flush_range);
6563 }
6564
6565 if (__improbable((bits & PP_ATTR_MODIFIED) &&
6566 (options & PMAP_OPTIONS_NOFLUSH))) {
6567 panic("phys_attribute_clear(%#010x,%#010x,%#010x,%p,%p): "
6568 "should not clear 'modified' without flushing TLBs",
6569 pn, bits, options, arg, flush_range);
6570 }
6571
6572 assert(pn != vm_page_fictitious_addr);
6573
6574 if (options & PMAP_OPTIONS_CLEAR_WRITE) {
6575 assert(bits == PP_ATTR_MODIFIED);
6576
6577 pmap_page_protect_options_with_flush_range(pn, (VM_PROT_ALL & ~VM_PROT_WRITE), options, NULL, flush_range);
6578 /*
6579 * We short circuit this case; it should not need to
6580 * invoke arm_force_fast_fault, so just clear the modified bit.
6581 * pmap_page_protect has taken care of resetting
6582 * the state so that we'll see the next write as a fault to
6583 * the VM (i.e. we don't want a fast fault).
6584 */
6585 ppattr_pa_clear_bits(pa, (pp_attr_t)bits);
6586 return;
6587 }
6588 if (bits & PP_ATTR_REFERENCED) {
6589 allow_mode &= ~(VM_PROT_READ | VM_PROT_EXECUTE);
6590 }
6591 if (bits & PP_ATTR_MODIFIED) {
6592 allow_mode &= ~VM_PROT_WRITE;
6593 }
6594
6595 if (bits == PP_ATTR_NOENCRYPT) {
6596 /*
6597 * We short circuit this case; it should not need to
6598 * invoke arm_force_fast_fault, so just clear and
6599 * return. On ARM, this bit is just a debugging aid.
6600 */
6601 ppattr_pa_clear_bits(pa, (pp_attr_t)bits);
6602 return;
6603 }
6604
6605 arm_force_fast_fault_with_flush_range(pn, allow_mode, options, NULL, (pp_attr_t)bits, flush_range);
6606 }
6607
6608 MARK_AS_PMAP_TEXT void
6609 phys_attribute_clear_internal(
6610 ppnum_t pn,
6611 unsigned int bits,
6612 int options,
6613 void *arg)
6614 {
6615 phys_attribute_clear_with_flush_range(pn, bits, options, arg, NULL);
6616 }
6617
6618 #if __ARM_RANGE_TLBI__
6619
6620 MARK_AS_PMAP_TEXT static vm_map_address_t
6621 phys_attribute_clear_twig_internal(
6622 pmap_t pmap,
6623 vm_map_address_t start,
6624 vm_map_address_t end,
6625 unsigned int bits,
6626 unsigned int options,
6627 pmap_tlb_flush_range_t *flush_range)
6628 {
6629 pmap_assert_locked(pmap, PMAP_LOCK_SHARED);
6630 const pt_attr_t * const pt_attr = pmap_get_pt_attr(pmap);
6631 assert(end >= start);
6632 assert((end - start) <= pt_attr_twig_size(pt_attr));
6633 const uint64_t pmap_page_size = pt_attr_page_size(pt_attr);
6634 vm_map_address_t va = start;
6635 pt_entry_t *pte_p, *start_pte_p, *end_pte_p, *curr_pte_p;
6636 tt_entry_t *tte_p;
6637 tte_p = pmap_tte(pmap, start);
6638
6639 /**
6640 * It's possible that this portion of our VA region has never been paged in, in which case
6641 * there may not be a valid twig or leaf table here.
6642 */
6643 if ((tte_p == (tt_entry_t *) NULL) || ((*tte_p & ARM_TTE_TYPE_MASK) != ARM_TTE_TYPE_TABLE)) {
6644 assert(flush_range->pending_region_entries == 0);
6645 return end;
6646 }
6647
6648 pte_p = (pt_entry_t *) ttetokv(*tte_p);
6649
6650 start_pte_p = &pte_p[pte_index(pt_attr, start)];
6651 end_pte_p = start_pte_p + ((end - start) >> pt_attr_leaf_shift(pt_attr));
6652 assert(end_pte_p >= start_pte_p);
6653 for (curr_pte_p = start_pte_p; curr_pte_p < end_pte_p; curr_pte_p++, va += pmap_page_size) {
6654 if (flush_range->pending_region_entries == 0) {
6655 flush_range->pending_region_start = va;
6656 } else {
6657 assertf((flush_range->pending_region_start +
6658 (flush_range->pending_region_entries * pmap_page_size)) == va,
6659 "pending_region_start 0x%llx + 0x%lx pages != va 0%llx",
6660 (unsigned long long)flush_range->pending_region_start,
6661 (unsigned long)flush_range->pending_region_entries,
6662 (unsigned long long)va);
6663 }
6664 flush_range->current_ptep = curr_pte_p;
6665 const pt_entry_t spte = os_atomic_load(curr_pte_p, relaxed);
6666 const pmap_paddr_t pa = pte_to_pa(spte);
6667 if (((spte & ARM_PTE_TYPE_MASK) != ARM_PTE_TYPE_FAULT) && pa_valid(pa)) {
6668 /* The PTE maps a managed page, so do the appropriate PV list-based permission changes. */
6669 const ppnum_t pn = (ppnum_t) atop(pa);
6670 phys_attribute_clear_with_flush_range(pn, bits, options, NULL, flush_range);
6671 if (__probable(flush_range->region_entry_added)) {
6672 flush_range->region_entry_added = false;
6673 } else {
6674 /**
6675 * It's possible that some other thread removed the mapping between our check
6676 * of the PTE above and taking the PVH lock in the
6677 * phys_attribute_clear_with_flush_range() path. In that case we have a
6678 * discontinuity in the region to update, so just submit any pending region
6679 * templates and start a new region op on the next iteration.
6680 */
6681 pmap_multipage_op_submit_region(flush_range);
6682 }
6683 } else if (__improbable((spte & ARM_PTE_TYPE_MASK) == ARM_PTE_TYPE_FAULT)) {
6684 /**
6685 * We've found an invalid mapping, so we have a discontinuity in the the region to
6686 * update. Handle this by submitting any pending region templates and starting a new
6687 * region on the next iteration. In theory we could instead handle this by installing
6688 * a "safe" (AF bit cleared, minimal permissions) PTE template; the SPTM would just
6689 * ignore the update on finding an invalid mapping in the PTE. But we don't know
6690 * what a "safe" template will be in all cases: for example, JIT regions require all
6691 * mapping to either be invalid or to have full RWX permissions.
6692 */
6693 pmap_multipage_op_submit_region(flush_range);
6694 } else if (pmap_insert_flush_range_template(spte, flush_range)) {
6695 /**
6696 * We've found a mapping to a non-managed page, so just insert the existing
6697 * PTE into the pending region ops since we don't manage attributes for non-managed
6698 * pages.
6699 * If pmap_insert_flush_range_template() returns true, indicating that it reached
6700 * the mapping limit and submitted the SPTM call, then we also submit any pending
6701 * disjoint ops. Having pending operations in either category will keep preemption
6702 * disabled, and we want to ensure that we can at least temporarily
6703 * re-enable preemption every SPTM_MAPPING_LIMIT mappings.
6704 */
6705 pmap_multipage_op_submit_disjoint(0, flush_range);
6706 }
6707 if (((flush_range->processed_entries + flush_range->pending_disjoint_entries +
6708 flush_range->pending_region_entries) >= SPTM_MAPPING_LIMIT) &&
6709 pmap_pending_preemption()) {
6710 pmap_multipage_op_submit(flush_range);
6711 assert(preemption_enabled());
6712 }
6713 }
6714
6715 /* SPTM region ops can't span L3 table boundaries, so submit any pending region templates now. */
6716 pmap_multipage_op_submit_region(flush_range);
6717 return end;
6718 }
6719
6720 MARK_AS_PMAP_TEXT vm_map_address_t
6721 phys_attribute_clear_range_internal(
6722 pmap_t pmap,
6723 vm_map_address_t start,
6724 vm_map_address_t end,
6725 unsigned int bits,
6726 unsigned int options)
6727 {
6728 if (__improbable(end < start)) {
6729 panic("%s: invalid address range %p, %p", __func__, (void*)start, (void*)end);
6730 }
6731 validate_pmap_mutable(pmap);
6732
6733 vm_map_address_t va = start;
6734 pmap_tlb_flush_range_t flush_range = {
6735 .ptfr_pmap = pmap,
6736 .ptfr_start = start,
6737 .ptfr_end = end,
6738 .current_ptep = NULL,
6739 .pending_region_start = 0,
6740 .pending_region_entries = 0,
6741 .region_entry_added = false,
6742 .current_header = NULL,
6743 .current_header_first_mapping_index = 0,
6744 .processed_entries = 0,
6745 .pending_disjoint_entries = 0,
6746 .ptfr_flush_needed = false
6747 };
6748
6749 pmap_lock(pmap, PMAP_LOCK_SHARED);
6750 const pt_attr_t * const pt_attr = pmap_get_pt_attr(pmap);
6751
6752 while (va < end) {
6753 vm_map_address_t curr_end;
6754
6755 curr_end = ((va + pt_attr_twig_size(pt_attr)) & ~pt_attr_twig_offmask(pt_attr));
6756 if (curr_end > end) {
6757 curr_end = end;
6758 }
6759
6760 va = phys_attribute_clear_twig_internal(pmap, va, curr_end, bits, options, &flush_range);
6761 }
6762 pmap_multipage_op_submit(&flush_range);
6763 pmap_unlock(pmap, PMAP_LOCK_SHARED);
6764 assert((flush_range.pending_disjoint_entries == 0) && (flush_range.pending_region_entries == 0));
6765 if (flush_range.ptfr_flush_needed) {
6766 pmap_get_pt_ops(pmap)->flush_tlb_region_async(
6767 flush_range.ptfr_start,
6768 flush_range.ptfr_end - flush_range.ptfr_start,
6769 flush_range.ptfr_pmap,
6770 true);
6771 sync_tlb_flush();
6772 }
6773 return va;
6774 }
6775
6776 static void
6777 phys_attribute_clear_range(
6778 pmap_t pmap,
6779 vm_map_address_t start,
6780 vm_map_address_t end,
6781 unsigned int bits,
6782 unsigned int options)
6783 {
6784 /*
6785 * We allow single-page requests to execute non-preemptibly,
6786 * as it doesn't make sense to sample AST_URGENT for a single-page
6787 * operation, and there are a couple of special use cases that
6788 * require a non-preemptible single-page operation.
6789 */
6790 if ((end - start) > (pt_attr_page_size(pmap_get_pt_attr(pmap)) * PAGE_RATIO)) {
6791 pmap_verify_preemptible();
6792 }
6793 __assert_only const int preemption_level = get_preemption_level();
6794
6795 PMAP_TRACE(3, PMAP_CODE(PMAP__ATTRIBUTE_CLEAR_RANGE) | DBG_FUNC_START, bits);
6796
6797 phys_attribute_clear_range_internal(pmap, start, end, bits, options);
6798
6799 PMAP_TRACE(3, PMAP_CODE(PMAP__ATTRIBUTE_CLEAR_RANGE) | DBG_FUNC_END);
6800
6801 assert(preemption_level == get_preemption_level());
6802 }
6803 #endif /* __ARM_RANGE_TLBI__ */
6804
6805 static void
6806 phys_attribute_clear(
6807 ppnum_t pn,
6808 unsigned int bits,
6809 int options,
6810 void *arg)
6811 {
6812 /*
6813 * Do we really want this tracepoint? It will be extremely chatty.
6814 * Also, should we have a corresponding trace point for the set path?
6815 */
6816 PMAP_TRACE(3, PMAP_CODE(PMAP__ATTRIBUTE_CLEAR) | DBG_FUNC_START, pn, bits);
6817
6818 phys_attribute_clear_internal(pn, bits, options, arg);
6819
6820 PMAP_TRACE(3, PMAP_CODE(PMAP__ATTRIBUTE_CLEAR) | DBG_FUNC_END);
6821 }
6822
6823 /*
6824 * Set specified attribute bits.
6825 *
6826 * Set cached value in the pv head because we have
6827 * no per-mapping hardware support for referenced and
6828 * modify bits.
6829 */
6830 MARK_AS_PMAP_TEXT void
6831 phys_attribute_set_internal(
6832 ppnum_t pn,
6833 unsigned int bits)
6834 {
6835 pmap_paddr_t pa = ptoa(pn);
6836 assert(pn != vm_page_fictitious_addr);
6837
6838 ppattr_pa_set_bits(pa, (uint16_t)bits);
6839
6840 return;
6841 }
6842
6843 static void
6844 phys_attribute_set(
6845 ppnum_t pn,
6846 unsigned int bits)
6847 {
6848 phys_attribute_set_internal(pn, bits);
6849 }
6850
6851
6852 /*
6853 * Check specified attribute bits.
6854 *
6855 * use the software cached bits (since no hw support).
6856 */
6857 static boolean_t
6858 phys_attribute_test(
6859 ppnum_t pn,
6860 unsigned int bits)
6861 {
6862 pmap_paddr_t pa = ptoa(pn);
6863 assert(pn != vm_page_fictitious_addr);
6864 return ppattr_pa_test_bits(pa, (pp_attr_t)bits);
6865 }
6866
6867
6868 /*
6869 * Set the modify/reference bits on the specified physical page.
6870 */
6871 void
6872 pmap_set_modify(ppnum_t pn)
6873 {
6874 phys_attribute_set(pn, PP_ATTR_MODIFIED);
6875 }
6876
6877
6878 /*
6879 * Clear the modify bits on the specified physical page.
6880 */
6881 void
6882 pmap_clear_modify(
6883 ppnum_t pn)
6884 {
6885 phys_attribute_clear(pn, PP_ATTR_MODIFIED, 0, NULL);
6886 }
6887
6888
6889 /*
6890 * pmap_is_modified:
6891 *
6892 * Return whether or not the specified physical page is modified
6893 * by any physical maps.
6894 */
6895 boolean_t
6896 pmap_is_modified(
6897 ppnum_t pn)
6898 {
6899 return phys_attribute_test(pn, PP_ATTR_MODIFIED);
6900 }
6901
6902
6903 /*
6904 * Set the reference bit on the specified physical page.
6905 */
6906 static void
6907 pmap_set_reference(
6908 ppnum_t pn)
6909 {
6910 phys_attribute_set(pn, PP_ATTR_REFERENCED);
6911 }
6912
6913 /*
6914 * Clear the reference bits on the specified physical page.
6915 */
6916 void
6917 pmap_clear_reference(
6918 ppnum_t pn)
6919 {
6920 phys_attribute_clear(pn, PP_ATTR_REFERENCED, 0, NULL);
6921 }
6922
6923
6924 /*
6925 * pmap_is_referenced:
6926 *
6927 * Return whether or not the specified physical page is referenced
6928 * by any physical maps.
6929 */
6930 boolean_t
6931 pmap_is_referenced(
6932 ppnum_t pn)
6933 {
6934 return phys_attribute_test(pn, PP_ATTR_REFERENCED);
6935 }
6936
6937 /*
6938 * pmap_get_refmod(phys)
6939 * returns the referenced and modified bits of the specified
6940 * physical page.
6941 */
6942 unsigned int
6943 pmap_get_refmod(
6944 ppnum_t pn)
6945 {
6946 return ((phys_attribute_test(pn, PP_ATTR_MODIFIED)) ? VM_MEM_MODIFIED : 0)
6947 | ((phys_attribute_test(pn, PP_ATTR_REFERENCED)) ? VM_MEM_REFERENCED : 0);
6948 }
6949
6950 static inline unsigned int
6951 pmap_clear_refmod_mask_to_modified_bits(const unsigned int mask)
6952 {
6953 return ((mask & VM_MEM_MODIFIED) ? PP_ATTR_MODIFIED : 0) |
6954 ((mask & VM_MEM_REFERENCED) ? PP_ATTR_REFERENCED : 0);
6955 }
6956
6957 /*
6958 * pmap_clear_refmod(phys, mask)
6959 * clears the referenced and modified bits as specified by the mask
6960 * of the specified physical page.
6961 */
6962 void
6963 pmap_clear_refmod_options(
6964 ppnum_t pn,
6965 unsigned int mask,
6966 unsigned int options,
6967 void *arg)
6968 {
6969 unsigned int bits;
6970
6971 bits = pmap_clear_refmod_mask_to_modified_bits(mask);
6972 phys_attribute_clear(pn, bits, options, arg);
6973 }
6974
6975 /*
6976 * Perform pmap_clear_refmod_options on a virtual address range.
6977 * The operation will be performed in bulk & tlb flushes will be coalesced
6978 * if possible.
6979 *
6980 * Returns true if the operation is supported on this platform.
6981 * If this function returns false, the operation is not supported and
6982 * nothing has been modified in the pmap.
6983 */
6984 bool
6985 pmap_clear_refmod_range_options(
6986 pmap_t pmap __unused,
6987 vm_map_address_t start __unused,
6988 vm_map_address_t end __unused,
6989 unsigned int mask __unused,
6990 unsigned int options __unused)
6991 {
6992 #if __ARM_RANGE_TLBI__
6993 unsigned int bits;
6994 bits = pmap_clear_refmod_mask_to_modified_bits(mask);
6995 phys_attribute_clear_range(pmap, start, end, bits, options);
6996 return true;
6997 #else /* __ARM_RANGE_TLBI__ */
6998 #pragma unused(pmap, start, end, mask, options)
6999 /*
7000 * This operation allows the VM to bulk modify refmod bits on a virtually
7001 * contiguous range of addresses. This is large performance improvement on
7002 * platforms that support ranged tlbi instructions. But on older platforms,
7003 * we can only flush per-page or the entire asid. So we currently
7004 * only support this operation on platforms that support ranged tlbi.
7005 * instructions. On other platforms, we require that
7006 * the VM modify the bits on a per-page basis.
7007 */
7008 return false;
7009 #endif /* __ARM_RANGE_TLBI__ */
7010 }
7011
7012 void
7013 pmap_clear_refmod(
7014 ppnum_t pn,
7015 unsigned int mask)
7016 {
7017 pmap_clear_refmod_options(pn, mask, 0, NULL);
7018 }
7019
7020 unsigned int
7021 pmap_disconnect_options(
7022 ppnum_t pn,
7023 unsigned int options,
7024 void *arg)
7025 {
7026 if ((options & PMAP_OPTIONS_COMPRESSOR_IFF_MODIFIED)) {
7027 /*
7028 * On ARM, the "modified" bit is managed by software, so
7029 * we know up-front if the physical page is "modified",
7030 * without having to scan all the PTEs pointing to it.
7031 * The caller should have made the VM page "busy" so noone
7032 * should be able to establish any new mapping and "modify"
7033 * the page behind us.
7034 */
7035 if (pmap_is_modified(pn)) {
7036 /*
7037 * The page has been modified and will be sent to
7038 * the VM compressor.
7039 */
7040 options |= PMAP_OPTIONS_COMPRESSOR;
7041 } else {
7042 /*
7043 * The page hasn't been modified and will be freed
7044 * instead of compressed.
7045 */
7046 }
7047 }
7048
7049 /* disconnect the page */
7050 pmap_page_protect_options(pn, 0, options, arg);
7051
7052 /* return ref/chg status */
7053 return pmap_get_refmod(pn);
7054 }
7055
7056 /*
7057 * Routine:
7058 * pmap_disconnect
7059 *
7060 * Function:
7061 * Disconnect all mappings for this page and return reference and change status
7062 * in generic format.
7063 *
7064 */
7065 unsigned int
7066 pmap_disconnect(
7067 ppnum_t pn)
7068 {
7069 pmap_page_protect(pn, 0); /* disconnect the page */
7070 return pmap_get_refmod(pn); /* return ref/chg status */
7071 }
7072
7073 boolean_t
7074 pmap_has_managed_page(ppnum_t first, ppnum_t last)
7075 {
7076 if (ptoa(first) >= vm_last_phys) {
7077 return FALSE;
7078 }
7079 if (ptoa(last) < vm_first_phys) {
7080 return FALSE;
7081 }
7082
7083 return TRUE;
7084 }
7085
7086 /*
7087 * The state maintained by the noencrypt functions is used as a
7088 * debugging aid on ARM. This incurs some overhead on the part
7089 * of the caller. A special case check in phys_attribute_clear
7090 * (the most expensive path) currently minimizes this overhead,
7091 * but stubbing these functions out on RELEASE kernels yields
7092 * further wins.
7093 */
7094 boolean_t
7095 pmap_is_noencrypt(
7096 ppnum_t pn)
7097 {
7098 #if DEVELOPMENT || DEBUG
7099 boolean_t result = FALSE;
7100
7101 if (!pa_valid(ptoa(pn))) {
7102 return FALSE;
7103 }
7104
7105 result = (phys_attribute_test(pn, PP_ATTR_NOENCRYPT));
7106
7107 return result;
7108 #else
7109 #pragma unused(pn)
7110 return FALSE;
7111 #endif
7112 }
7113
7114 void
7115 pmap_set_noencrypt(
7116 ppnum_t pn)
7117 {
7118 #if DEVELOPMENT || DEBUG
7119 if (!pa_valid(ptoa(pn))) {
7120 return;
7121 }
7122
7123 phys_attribute_set(pn, PP_ATTR_NOENCRYPT);
7124 #else
7125 #pragma unused(pn)
7126 #endif
7127 }
7128
7129 void
7130 pmap_clear_noencrypt(
7131 ppnum_t pn)
7132 {
7133 #if DEVELOPMENT || DEBUG
7134 if (!pa_valid(ptoa(pn))) {
7135 return;
7136 }
7137
7138 phys_attribute_clear(pn, PP_ATTR_NOENCRYPT, 0, NULL);
7139 #else
7140 #pragma unused(pn)
7141 #endif
7142 }
7143
7144 void
7145 pmap_lock_phys_page(ppnum_t pn)
7146 {
7147 unsigned int pai;
7148 pmap_paddr_t phys = ptoa(pn);
7149
7150 if (pa_valid(phys)) {
7151 pai = pa_index(phys);
7152 __unused const locked_pvh_t locked_pvh = pvh_lock(pai);
7153 } else {
7154 simple_lock(&phys_backup_lock, LCK_GRP_NULL);
7155 }
7156 }
7157
7158
7159 void
7160 pmap_unlock_phys_page(ppnum_t pn)
7161 {
7162 unsigned int pai;
7163 pmap_paddr_t phys = ptoa(pn);
7164
7165 if (pa_valid(phys)) {
7166 pai = pa_index(phys);
7167 locked_pvh_t locked_pvh = {.pvh = pai_to_pvh(pai), .pai = pai};
7168 pvh_unlock(&locked_pvh);
7169 } else {
7170 simple_unlock(&phys_backup_lock);
7171 }
7172 }
7173
7174 MARK_AS_PMAP_TEXT void
7175 pmap_clear_user_ttb_internal(void)
7176 {
7177 set_mmu_ttb(invalid_ttep & TTBR_BADDR_MASK);
7178 }
7179
7180 void
7181 pmap_clear_user_ttb(void)
7182 {
7183 PMAP_TRACE(3, PMAP_CODE(PMAP__CLEAR_USER_TTB) | DBG_FUNC_START, NULL, 0, 0);
7184 pmap_clear_user_ttb_internal();
7185 PMAP_TRACE(3, PMAP_CODE(PMAP__CLEAR_USER_TTB) | DBG_FUNC_END);
7186 }
7187
7188 /**
7189 * Set up a "fast fault", or a page fault that won't go through the VM layer on
7190 * a page. This is primarily used to manage ref/mod bits in software. Depending
7191 * on the value of allow_mode, the next read and/or write of the page will fault
7192 * and the ref/mod bits will be updated.
7193 *
7194 * @param ppnum Page number to set up a fast fault on.
7195 * @param allow_mode VM_PROT_NONE will cause the next read and write access to
7196 * fault.
7197 * VM_PROT_READ will only cause the next write access to fault.
7198 * Other values are undefined.
7199 * @param options PMAP_OPTIONS_NOFLUSH indicates TLBI flush is not needed.
7200 * PMAP_OPTIONS_FF_WIRED forces a fast fault even on wired pages.
7201 * PMAP_OPTIONS_SET_REUSABLE/PMAP_OPTIONS_CLEAR_REUSABLE updates
7202 * the global reusable bit of the page.
7203 * @param locked_pvh If non-NULL, this indicates the PVH lock for [ppnum] is already locked
7204 * by the caller. This is an input/output parameter which may be updated
7205 * to reflect a new PV head value to be passed to a later call to pvh_unlock().
7206 * @param bits_to_clear Mask of additional pp_attr_t bits to clear for the physical
7207 * page, iff this function completes successfully and returns
7208 * TRUE. This is typically some combination of
7209 * the referenced, modified, and noencrypt bits.
7210 * @param flush_range When present, this function will skip the TLB flush for the
7211 * mappings that are covered by the range, leaving that to be
7212 * done later by the caller. It may also avoid submitting mapping
7213 * updates directly to the SPTM, instead accumulating them in a
7214 * per-CPU array to be submitted later by the caller.
7215 *
7216 * @return TRUE if the fast fault was successfully configured for all mappings
7217 * of the page, FALSE otherwise (e.g. if wired mappings are present and
7218 * PMAP_OPTIONS_FF_WIRED was not passed).
7219 *
7220 * @note PMAP_OPTIONS_NOFLUSH and flush_range cannot both be specified.
7221 *
7222 * @warning PMAP_OPTIONS_FF_WIRED should only be used with pages accessible from
7223 * EL0. The kernel may assume that accesses to wired, kernel-owned pages
7224 * won't fault.
7225 */
7226 MARK_AS_PMAP_TEXT static boolean_t
7227 arm_force_fast_fault_with_flush_range(
7228 ppnum_t ppnum,
7229 vm_prot_t allow_mode,
7230 int options,
7231 locked_pvh_t *locked_pvh,
7232 pp_attr_t bits_to_clear,
7233 pmap_tlb_flush_range_t *flush_range)
7234 {
7235 pmap_paddr_t phys = ptoa(ppnum);
7236 pv_entry_t *pve_p;
7237 pt_entry_t *pte_p;
7238 unsigned int pai;
7239 boolean_t result;
7240 unsigned int num_mappings = 0, num_skipped_mappings = 0;
7241 bool ref_fault;
7242 bool mod_fault;
7243 bool clear_write_fault = false;
7244 bool ref_aliases_mod = false;
7245
7246 assert(ppnum != vm_page_fictitious_addr);
7247
7248 /**
7249 * Assert that PMAP_OPTIONS_NOFLUSH and flush_range cannot both be specified.
7250 *
7251 * PMAP_OPTIONS_NOFLUSH indicates there is no need of flushing the TLB in the entire operation, and
7252 * flush_range indicates the caller requests deferral of the TLB flushing. Fundemantally, the two
7253 * semantics conflict with each other, so assert they are not both true.
7254 */
7255 assert(!(flush_range && (options & PMAP_OPTIONS_NOFLUSH)));
7256
7257 if (!pa_valid(phys)) {
7258 return FALSE; /* Not a managed page. */
7259 }
7260
7261 result = TRUE;
7262 ref_fault = false;
7263 mod_fault = false;
7264 pai = pa_index(phys);
7265 locked_pvh_t local_locked_pvh = {.pvh = 0};
7266 if (__probable(locked_pvh == NULL)) {
7267 if (flush_range != NULL) {
7268 /**
7269 * If we're partway through processing a multi-page batched call,
7270 * preemption will already be disabled so we can't simply call
7271 * pvh_lock() which may block. Instead, we first try to acquire
7272 * the lock without waiting, which in most cases should succeed.
7273 * If it fails, we submit the pending batched operations to re-
7274 * enable preemption and then acquire the lock normally.
7275 */
7276 local_locked_pvh = pvh_try_lock(pai);
7277 if (__improbable(!pvh_try_lock_success(&local_locked_pvh))) {
7278 pmap_multipage_op_submit(flush_range);
7279 local_locked_pvh = pvh_lock(pai);
7280 }
7281 } else {
7282 local_locked_pvh = pvh_lock(pai);
7283 }
7284 } else {
7285 local_locked_pvh = *locked_pvh;
7286 assert(pai == local_locked_pvh.pai);
7287 }
7288 assert(local_locked_pvh.pvh != 0);
7289 pvh_assert_locked(pai);
7290
7291 pte_p = PT_ENTRY_NULL;
7292 pve_p = PV_ENTRY_NULL;
7293 if (pvh_test_type(local_locked_pvh.pvh, PVH_TYPE_PTEP)) {
7294 pte_p = pvh_ptep(local_locked_pvh.pvh);
7295 } else if (pvh_test_type(local_locked_pvh.pvh, PVH_TYPE_PVEP)) {
7296 pve_p = pvh_pve_list(local_locked_pvh.pvh);
7297 } else if (__improbable(!pvh_test_type(local_locked_pvh.pvh, PVH_TYPE_NULL))) {
7298 panic("%s: invalid PV head 0x%llx for PA 0x%llx", __func__, (uint64_t)local_locked_pvh.pvh, (uint64_t)phys);
7299 }
7300
7301 const bool is_reusable = ppattr_test_reusable(pai);
7302
7303 bool pvh_lock_sleep_mode_needed = false;
7304 pmap_sptm_percpu_data_t *sptm_pcpu = NULL;
7305 sptm_disjoint_op_t *sptm_ops = NULL;
7306
7307 /**
7308 * This would also work as a block, with the above variables declared using the
7309 * __block qualifier, but the extra runtime overhead of block syntax (e.g.
7310 * dereferencing __block variables through stack forwarding pointers) isn't needed
7311 * here, as we never need to use this code sequence as a closure.
7312 */
7313 #define FFF_PERCPU_INIT() do { \
7314 disable_preemption(); \
7315 sptm_pcpu = PERCPU_GET(pmap_sptm_percpu); \
7316 sptm_ops = sptm_pcpu->sptm_ops; \
7317 } while (0)
7318
7319 FFF_PERCPU_INIT();
7320
7321 int pve_ptep_idx = 0;
7322
7323 /**
7324 * With regard to TLBI, there are three cases:
7325 *
7326 * 1. PMAP_OPTIONS_NOFLUSH is specified. In such case, SPTM doesn't need to flush TLB and neither does pmap.
7327 * 2. PMAP_OPTIONS_NOFLUSH is not specified, but flush_range is, indicating the caller intends to flush TLB
7328 * itself (with range TLBI). In such case, we check the flush_range limits and only issue the TLBI if a
7329 * mapping is out of the range.
7330 * 3. Neither PMAP_OPTIONS_NOFLUSH nor a valid flush_range pointer is specified. In such case, we should just
7331 * let SPTM handle TLBI flushing.
7332 */
7333 const bool defer_tlbi = (options & PMAP_OPTIONS_NOFLUSH) || flush_range;
7334 const uint32_t sptm_update_options = SPTM_UPDATE_PERMS_AND_WAS_WRITABLE | SPTM_UPDATE_AF | (defer_tlbi ? SPTM_UPDATE_DEFER_TLBI : 0);
7335
7336 while ((pve_p != PV_ENTRY_NULL) || (pte_p != PT_ENTRY_NULL)) {
7337 pt_entry_t spte;
7338 pt_entry_t tmplate;
7339
7340 if (__improbable(pvh_lock_sleep_mode_needed)) {
7341 assert((num_mappings == 0) && (num_skipped_mappings == 0));
7342 /**
7343 * Undo the explicit preemption disable done in the last call to FFF_PER_CPU_INIT().
7344 * If the PVH lock is placed in sleep mode, we can't rely on it to disable preemption,
7345 * so we need these explicit preemption twiddles to ensure we don't get migrated off-
7346 * core while processing SPTM per-CPU data. At the same time, we also want preemption
7347 * to briefly be re-enabled every SPTM_MAPPING_LIMIT mappings so that any pending
7348 * urgent ASTs can be handled.
7349 */
7350 enable_preemption();
7351 pvh_lock_enter_sleep_mode(&local_locked_pvh);
7352 pvh_lock_sleep_mode_needed = false;
7353 FFF_PERCPU_INIT();
7354 }
7355
7356 if (pve_p != PV_ENTRY_NULL) {
7357 pte_p = pve_get_ptep(pve_p, pve_ptep_idx);
7358 if (pte_p == PT_ENTRY_NULL) {
7359 goto fff_skip_pve;
7360 }
7361 }
7362
7363 #ifdef PVH_FLAG_IOMMU
7364 if (pvh_ptep_is_iommu(pte_p)) {
7365 ++num_skipped_mappings;
7366 goto fff_skip_pve;
7367 }
7368 #endif
7369 spte = os_atomic_load(pte_p, relaxed);
7370 if (pte_is_compressed(spte, pte_p)) {
7371 panic("pte is COMPRESSED: pte_p=%p ppnum=0x%x", pte_p, ppnum);
7372 }
7373
7374 pt_desc_t *ptdp = NULL;
7375 pmap_t pmap = NULL;
7376 vm_map_address_t va = 0;
7377
7378 if ((flush_range != NULL) && (pte_p == flush_range->current_ptep)) {
7379 /**
7380 * If the current mapping matches the flush range's current iteration position,
7381 * there's no need to do the work of getting the PTD. We already know the pmap,
7382 * and the VA is implied by flush_range->pending_region_start.
7383 */
7384 pmap = flush_range->ptfr_pmap;
7385 } else {
7386 ptdp = ptep_get_ptd(pte_p);
7387 pmap = ptdp->pmap;
7388 va = ptd_get_va(ptdp, pte_p);
7389 assert(va >= pmap->min && va < pmap->max);
7390 }
7391
7392 bool skip_pte = pte_is_wired(spte) &&
7393 ((options & PMAP_OPTIONS_FF_WIRED) == 0);
7394
7395 if (skip_pte) {
7396 result = FALSE;
7397 }
7398
7399 // A concurrent pmap_remove() may have cleared the PTE
7400 if (__improbable((spte & ARM_PTE_TYPE_MASK) == ARM_PTE_TYPE_FAULT)) {
7401 skip_pte = true;
7402 }
7403
7404 /**
7405 * If the PTD is NULL, we're adding the current mapping to the pending region templates instead of the
7406 * pending disjoint ops, so we don't need to do flush range disjoint op management.
7407 */
7408 if ((flush_range != NULL) && (ptdp != NULL) && !skip_pte) {
7409 /**
7410 * Insert a "header" entry for this physical page into the SPTM disjoint ops array.
7411 * We do this in three cases:
7412 * 1) We're at the beginning of the SPTM ops array (num_mappings == 0, flush_range->pending_disjoint_entries == 0).
7413 * 2) We may not be at the beginning of the SPTM ops array, but we are about to add the first operation
7414 * for this physical page (num_mappings == 0, flush_range->pending_disjoint_entries == ?).
7415 * 3) We need to change the options passed to the SPTM for a run of one or more mappings. Specifically,
7416 * if we encounter a run of mappings that reside outside the VA region of our flush_range, or that
7417 * belong to a pmap other than the one targeted by our flush_range, we should ask the SPTM to flush
7418 * the TLB for us (i.e., clear SPTM_UPDATE_DEFER_TLBI), but only for those specific mappings.
7419 */
7420 uint32_t per_mapping_sptm_update_options = sptm_update_options;
7421 if ((flush_range->ptfr_pmap != pmap) || (va >= flush_range->ptfr_end) || (va < flush_range->ptfr_start)) {
7422 per_mapping_sptm_update_options &= ~SPTM_UPDATE_DEFER_TLBI;
7423 }
7424 if ((num_mappings == 0) ||
7425 (flush_range->current_header->per_paddr_header.options != per_mapping_sptm_update_options)) {
7426 if (pmap_multipage_op_add_page(phys, &num_mappings, per_mapping_sptm_update_options, flush_range)) {
7427 /**
7428 * If we needed to submit the pending disjoint ops to make room for the new page,
7429 * flush any pending region ops to reenable preemption and restart the loop with
7430 * the lock in sleep mode. This prevents preemption from being held disabled
7431 * for an arbitrary amount of time in the pathological case in which we have
7432 * both pending region ops and an excessively long PV list that repeatedly
7433 * requires new page headers with SPTM_MAPPING_LIMIT - 1 entries already pending.
7434 */
7435 pmap_multipage_op_submit_region(flush_range);
7436 assert(num_mappings == 0);
7437 num_skipped_mappings = 0;
7438 pvh_lock_sleep_mode_needed = true;
7439 continue;
7440 }
7441 }
7442 }
7443
7444 const pt_attr_t * const pt_attr = pmap_get_pt_attr(pmap);
7445
7446 /* update pmap stats and ledgers */
7447 const bool is_internal = ppattr_pve_is_internal(pai, pve_p, pve_ptep_idx);
7448 const bool is_altacct = ppattr_pve_is_altacct(pai, pve_p, pve_ptep_idx);
7449 if (is_altacct) {
7450 /*
7451 * We do not track "reusable" status for
7452 * "alternate accounting" mappings.
7453 */
7454 } else if ((options & PMAP_OPTIONS_CLEAR_REUSABLE) &&
7455 is_reusable &&
7456 is_internal &&
7457 pmap != kernel_pmap) {
7458 /* one less "reusable" */
7459 pmap_ledger_debit(pmap, task_ledgers.reusable, pt_attr_page_size(pt_attr) * PAGE_RATIO);
7460 /* one more "internal" */
7461 pmap_ledger_credit(pmap, task_ledgers.internal, pt_attr_page_size(pt_attr) * PAGE_RATIO);
7462 pmap_ledger_credit(pmap, task_ledgers.phys_footprint, pt_attr_page_size(pt_attr) * PAGE_RATIO);
7463
7464 /*
7465 * Since the page is being marked non-reusable, we assume that it will be
7466 * modified soon. Avoid the cost of another trap to handle the fast
7467 * fault when we next write to this page.
7468 */
7469 clear_write_fault = true;
7470 } else if ((options & PMAP_OPTIONS_SET_REUSABLE) &&
7471 !is_reusable &&
7472 is_internal &&
7473 pmap != kernel_pmap) {
7474 /* one more "reusable" */
7475 pmap_ledger_credit(pmap, task_ledgers.reusable, pt_attr_page_size(pt_attr) * PAGE_RATIO);
7476 pmap_ledger_debit(pmap, task_ledgers.internal, pt_attr_page_size(pt_attr) * PAGE_RATIO);
7477 pmap_ledger_debit(pmap, task_ledgers.phys_footprint, pt_attr_page_size(pt_attr) * PAGE_RATIO);
7478 }
7479
7480 if (skip_pte) {
7481 ++num_skipped_mappings;
7482 goto fff_skip_pve;
7483 }
7484
7485 tmplate = spte;
7486
7487 if ((allow_mode & VM_PROT_READ) != VM_PROT_READ) {
7488 /* read protection sets the pte to fault */
7489 tmplate = tmplate & ~ARM_PTE_AF;
7490 ref_fault = true;
7491 }
7492 if ((allow_mode & VM_PROT_WRITE) != VM_PROT_WRITE) {
7493 /* take away write permission if set */
7494 if (pmap == kernel_pmap) {
7495 if ((tmplate & ARM_PTE_APMASK) == ARM_PTE_AP(AP_RWNA)) {
7496 tmplate = ((tmplate & ~ARM_PTE_APMASK) | ARM_PTE_AP(AP_RONA));
7497 pte_set_was_writeable(tmplate, true);
7498 mod_fault = true;
7499 }
7500 } else {
7501 if ((tmplate & ARM_PTE_APMASK) == pt_attr_leaf_rw(pt_attr)) {
7502 tmplate = ((tmplate & ~ARM_PTE_APMASK) | pt_attr_leaf_ro(pt_attr));
7503 pte_set_was_writeable(tmplate, true);
7504 mod_fault = true;
7505 }
7506 }
7507 }
7508
7509 if (ptdp != NULL) {
7510 sptm_ops[num_mappings].root_pt_paddr = pmap->ttep;
7511 sptm_ops[num_mappings].vaddr = va;
7512 sptm_ops[num_mappings].pte_template = tmplate;
7513 ++num_mappings;
7514 } else if (pmap_insert_flush_range_template(tmplate, flush_range)) {
7515 /**
7516 * We submit both the pending disjoint and pending region ops whenever
7517 * either category reaches the mapping limit. Having pending operations
7518 * in either category will keep preemption disabled, and we want to ensure
7519 * that we can at least temporarily re-enable preemption roughly every
7520 * SPTM_MAPPING_LIMIT mappings.
7521 */
7522 pmap_multipage_op_submit_disjoint(num_mappings, flush_range);
7523 pvh_lock_sleep_mode_needed = true;
7524 num_mappings = num_skipped_mappings = 0;
7525 }
7526 fff_skip_pve:
7527 if ((num_mappings + num_skipped_mappings) >= SPTM_MAPPING_LIMIT) {
7528 if (flush_range != NULL) {
7529 /* See comment above for why we submit both disjoint and region ops when we hit the limit. */
7530 pmap_multipage_op_submit_disjoint(num_mappings, flush_range);
7531 pmap_multipage_op_submit_region(flush_range);
7532 } else if (num_mappings > 0) {
7533 sptm_update_disjoint(phys, sptm_pcpu->sptm_ops_pa, num_mappings, sptm_update_options);
7534 }
7535 pvh_lock_sleep_mode_needed = true;
7536 num_mappings = num_skipped_mappings = 0;
7537 }
7538 pte_p = PT_ENTRY_NULL;
7539 if ((pve_p != PV_ENTRY_NULL) && (++pve_ptep_idx == PTE_PER_PVE)) {
7540 pve_ptep_idx = 0;
7541 pve_p = pve_next(pve_p);
7542 }
7543 }
7544
7545 if (num_mappings != 0) {
7546 sptm_return_t sptm_ret;
7547
7548 if (flush_range == NULL) {
7549 sptm_ret = sptm_update_disjoint(phys, sptm_pcpu->sptm_ops_pa, num_mappings, sptm_update_options);
7550 } else {
7551 /* Resync the pending mapping state in flush_range with our local state. */
7552 assert(num_mappings >= flush_range->pending_disjoint_entries);
7553 flush_range->pending_disjoint_entries = num_mappings;
7554 }
7555 }
7556
7557 /**
7558 * Undo the explicit disable_preemption() done in FFF_PERCPU_INIT().
7559 * Note that enable_preemption() decrements a per-thread counter, so if
7560 * we happen to still hold the PVH lock in spin mode then preemption won't
7561 * actually be re-enabled until we drop the lock (which also decrements
7562 * the per-thread counter.
7563 */
7564 enable_preemption();
7565
7566 /*
7567 * If we are using the same approach for ref and mod
7568 * faults on this PTE, do not clear the write fault;
7569 * this would cause both ref and mod to be set on the
7570 * page again, and prevent us from taking ANY read/write
7571 * fault on the mapping.
7572 */
7573 if (clear_write_fault && !ref_aliases_mod) {
7574 arm_clear_fast_fault(ppnum, VM_PROT_WRITE, local_locked_pvh.pvh, PT_ENTRY_NULL, 0);
7575 }
7576
7577 pp_attr_t attrs_to_clear = (result ? bits_to_clear : 0);
7578 pp_attr_t attrs_to_set = 0;
7579 /* update global "reusable" status for this page */
7580 if ((options & PMAP_OPTIONS_CLEAR_REUSABLE) && is_reusable) {
7581 attrs_to_clear |= PP_ATTR_REUSABLE;
7582 } else if ((options & PMAP_OPTIONS_SET_REUSABLE) && !is_reusable) {
7583 attrs_to_set |= PP_ATTR_REUSABLE;
7584 }
7585
7586 if (mod_fault) {
7587 attrs_to_set |= PP_ATTR_MODFAULT;
7588 }
7589 if (ref_fault) {
7590 attrs_to_set |= PP_ATTR_REFFAULT;
7591 }
7592
7593 if (attrs_to_set | attrs_to_clear) {
7594 ppattr_modify_bits(pai, attrs_to_clear, attrs_to_set);
7595 }
7596
7597 if (__probable(locked_pvh == NULL)) {
7598 pvh_unlock(&local_locked_pvh);
7599 } else {
7600 *locked_pvh = local_locked_pvh;
7601 }
7602 if ((flush_range != NULL) && !preemption_enabled()) {
7603 flush_range->processed_entries += num_skipped_mappings;
7604 }
7605 return result;
7606 }
7607
7608 MARK_AS_PMAP_TEXT boolean_t
7609 arm_force_fast_fault_internal(
7610 ppnum_t ppnum,
7611 vm_prot_t allow_mode,
7612 int options)
7613 {
7614 if (__improbable((options & (PMAP_OPTIONS_FF_LOCKED | PMAP_OPTIONS_FF_WIRED | PMAP_OPTIONS_NOFLUSH)) != 0)) {
7615 panic("arm_force_fast_fault(0x%x, 0x%x, 0x%x): invalid options", ppnum, allow_mode, options);
7616 }
7617 return arm_force_fast_fault_with_flush_range(ppnum, allow_mode, options, NULL, 0, NULL);
7618 }
7619
7620 /*
7621 * Routine: arm_force_fast_fault
7622 *
7623 * Function:
7624 * Force all mappings for this page to fault according
7625 * to the access modes allowed, so we can gather ref/modify
7626 * bits again.
7627 */
7628
7629 boolean_t
7630 arm_force_fast_fault(
7631 ppnum_t ppnum,
7632 vm_prot_t allow_mode,
7633 int options,
7634 __unused void *arg)
7635 {
7636 pmap_paddr_t phys = ptoa(ppnum);
7637
7638 assert(ppnum != vm_page_fictitious_addr);
7639
7640 if (!pa_valid(phys)) {
7641 return FALSE; /* Not a managed page. */
7642 }
7643
7644 return arm_force_fast_fault_internal(ppnum, allow_mode, options);
7645 }
7646
7647 /**
7648 * Clear pending force fault for at most SPTM_MAPPING_LIMIT mappings for this
7649 * page based on the observed fault type, and update the appropriate ref/modify
7650 * bits for the physical page. This typically involves adding write permissions
7651 * back for write faults and setting the Access Flag for both read/write faults
7652 * (since the lack of those things is what caused the fault in the first place).
7653 *
7654 * @note Only SPTM_MAPPING_LIMIT number of mappings can be modified in a single
7655 * arm_clear_fast_fault() call to prevent excessive PVH lock contention as
7656 * the PVH lock should be held for `ppnum` already. If a fault is
7657 * subsequently taken on a mapping we haven't processed, arm_fast_fault()
7658 * will call this function with a non-NULL pte_p to perform a targeted
7659 * fixup.
7660 *
7661 * @param ppnum Page number of the page to clear a pending force fault on.
7662 * @param fault_type The type of access/fault that triggered us wanting to clear
7663 * the pending force fault status. This determines how we
7664 * modify the PTE to not cause a fault in the future and also
7665 * whether we mark the PTE as referenced or modified.
7666 * Typically a write fault would cause the page to be marked
7667 * as referenced and modified, and a read fault would only
7668 * cause the page to be marked as referenced.
7669 * @param pvh pv_head_table entry value for [ppnum] returned by a previous call
7670 * to pvh_lock().
7671 * @param pte_p If this value is non-PT_ENTRY_NULL then only this specified PTE
7672 * will be modified. If it is PT_ENTRY_NULL, then every mapping to
7673 * `ppnum` will be modified.
7674 * @param attrs_to_clear Mask of additional pp_attr_t bits to clear for the physical
7675 * page upon completion of this function. This is typically
7676 * some combination of the REFFAULT and MODFAULT bits.
7677 *
7678 * @return TRUE if any PTEs were modified, FALSE otherwise.
7679 */
7680 MARK_AS_PMAP_TEXT static boolean_t
7681 arm_clear_fast_fault(
7682 ppnum_t ppnum,
7683 vm_prot_t fault_type,
7684 uintptr_t pvh,
7685 pt_entry_t *pte_p,
7686 pp_attr_t attrs_to_clear)
7687 {
7688 const pmap_paddr_t pa = ptoa(ppnum);
7689 pv_entry_t *pve_p;
7690 boolean_t result;
7691 unsigned int num_mappings = 0, num_skipped_mappings = 0;
7692 pp_attr_t attrs_to_set = 0;
7693
7694 assert(ppnum != vm_page_fictitious_addr);
7695
7696 if (!pa_valid(pa)) {
7697 return FALSE; /* Not a managed page. */
7698 }
7699
7700 result = FALSE;
7701 pve_p = PV_ENTRY_NULL;
7702 if (pte_p == PT_ENTRY_NULL) {
7703 if (pvh_test_type(pvh, PVH_TYPE_PTEP)) {
7704 pte_p = pvh_ptep(pvh);
7705 } else if (pvh_test_type(pvh, PVH_TYPE_PVEP)) {
7706 pve_p = pvh_pve_list(pvh);
7707 } else if (__improbable(!pvh_test_type(pvh, PVH_TYPE_NULL))) {
7708 panic("%s: invalid PV head 0x%llx for PA 0x%llx", __func__, (uint64_t)pvh, (uint64_t)pa);
7709 }
7710 }
7711
7712 disable_preemption();
7713 pmap_sptm_percpu_data_t *sptm_pcpu = PERCPU_GET(pmap_sptm_percpu);
7714 sptm_disjoint_op_t *sptm_ops = sptm_pcpu->sptm_ops;
7715
7716 int pve_ptep_idx = 0;
7717
7718 while ((pve_p != PV_ENTRY_NULL) || (pte_p != PT_ENTRY_NULL)) {
7719 pt_entry_t spte;
7720 pt_entry_t tmplate;
7721
7722 if (pve_p != PV_ENTRY_NULL) {
7723 pte_p = pve_get_ptep(pve_p, pve_ptep_idx);
7724 if (pte_p == PT_ENTRY_NULL) {
7725 goto cff_skip_pve;
7726 }
7727 }
7728
7729 #ifdef PVH_FLAG_IOMMU
7730 if (pvh_ptep_is_iommu(pte_p)) {
7731 ++num_skipped_mappings;
7732 goto cff_skip_pve;
7733 }
7734 #endif
7735 spte = os_atomic_load(pte_p, relaxed);
7736 // A concurrent pmap_remove() may have cleared the PTE
7737 if (__improbable((spte & ARM_PTE_TYPE_MASK) == ARM_PTE_TYPE_FAULT)) {
7738 ++num_skipped_mappings;
7739 goto cff_skip_pve;
7740 }
7741
7742 const pt_desc_t * const ptdp = ptep_get_ptd(pte_p);
7743 const pmap_t pmap = ptdp->pmap;
7744 const vm_map_address_t va = ptd_get_va(ptdp, pte_p);
7745
7746 assert(va >= pmap->min && va < pmap->max);
7747
7748 tmplate = spte;
7749
7750 if ((fault_type & VM_PROT_WRITE) && (pte_was_writeable(spte))) {
7751 {
7752 if (pmap == kernel_pmap) {
7753 tmplate = ((spte & ~ARM_PTE_APMASK) | ARM_PTE_AP(AP_RWNA));
7754 } else {
7755 assert(pmap->type != PMAP_TYPE_NESTED);
7756 tmplate = ((spte & ~ARM_PTE_APMASK) | pt_attr_leaf_rw(pmap_get_pt_attr(pmap)));
7757 }
7758 }
7759
7760 tmplate |= ARM_PTE_AF;
7761
7762 pte_set_was_writeable(tmplate, false);
7763 attrs_to_set |= (PP_ATTR_REFERENCED | PP_ATTR_MODIFIED);
7764 } else if ((fault_type & VM_PROT_READ) && ((spte & ARM_PTE_AF) != ARM_PTE_AF)) {
7765 tmplate = spte | ARM_PTE_AF;
7766
7767 {
7768 attrs_to_set |= PP_ATTR_REFERENCED;
7769 }
7770 }
7771
7772 assert(spte != ARM_PTE_TYPE_FAULT);
7773
7774 if (spte != tmplate) {
7775 sptm_ops[num_mappings].root_pt_paddr = pmap->ttep;
7776 sptm_ops[num_mappings].vaddr = va;
7777 sptm_ops[num_mappings].pte_template = tmplate;
7778 ++num_mappings;
7779 result = TRUE;
7780 }
7781
7782 cff_skip_pve:
7783 if ((num_mappings + num_skipped_mappings) == SPTM_MAPPING_LIMIT) {
7784 if (num_mappings != 0) {
7785 sptm_update_disjoint(pa, sptm_pcpu->sptm_ops_pa, num_mappings,
7786 SPTM_UPDATE_PERMS_AND_WAS_WRITABLE | SPTM_UPDATE_AF);
7787 num_mappings = 0;
7788 }
7789 /*
7790 * We've reached the limit of mappings that can be processed in a single arm_clear_fast_fault()
7791 * call. Bail out here to avoid excessive PVH lock duration on the fault path. If a fault is
7792 * subsequently taken on a mapping we haven't processed, arm_fast_fault() will call this
7793 * function with a non-NULL pte_p to perform a targeted fixup.
7794 */
7795 break;
7796 }
7797
7798 pte_p = PT_ENTRY_NULL;
7799 if ((pve_p != PV_ENTRY_NULL) && (++pve_ptep_idx == PTE_PER_PVE)) {
7800 pve_ptep_idx = 0;
7801 pve_p = pve_next(pve_p);
7802 }
7803 }
7804
7805 if (num_mappings != 0) {
7806 assert(result == TRUE);
7807 sptm_update_disjoint(pa, sptm_pcpu->sptm_ops_pa, num_mappings,
7808 SPTM_UPDATE_PERMS_AND_WAS_WRITABLE | SPTM_UPDATE_AF);
7809 }
7810
7811 if (attrs_to_set | attrs_to_clear) {
7812 ppattr_modify_bits(pa_index(pa), attrs_to_clear, attrs_to_set);
7813 }
7814 enable_preemption();
7815
7816 return result;
7817 }
7818
7819 /*
7820 * Determine if the fault was induced by software tracking of
7821 * modify/reference bits. If so, re-enable the mapping (and set
7822 * the appropriate bits).
7823 *
7824 * Returns KERN_SUCCESS if the fault was induced and was
7825 * successfully handled.
7826 *
7827 * Returns KERN_FAILURE if the fault was not induced and
7828 * the function was unable to deal with it.
7829 *
7830 * Returns KERN_PROTECTION_FAILURE if the pmap layer explictly
7831 * disallows this type of access.
7832 */
7833 MARK_AS_PMAP_TEXT kern_return_t
7834 arm_fast_fault_internal(
7835 pmap_t pmap,
7836 vm_map_address_t va,
7837 vm_prot_t fault_type,
7838 __unused bool was_af_fault,
7839 __unused bool from_user)
7840 {
7841 kern_return_t result = KERN_FAILURE;
7842 pt_entry_t *ptep;
7843 pt_entry_t spte = ARM_PTE_TYPE_FAULT;
7844 locked_pvh_t locked_pvh = {.pvh = 0};
7845 unsigned int pai;
7846 pmap_paddr_t pa;
7847 validate_pmap_mutable(pmap);
7848
7849 if (__probable(preemption_enabled())) {
7850 pmap_lock(pmap, PMAP_LOCK_SHARED);
7851 } else if (__improbable(!pmap_try_lock(pmap, PMAP_LOCK_SHARED))) {
7852 /**
7853 * In certain cases, arm_fast_fault() may be invoked with preemption disabled
7854 * on the copyio path. In theses cases the (in-kernel) caller expects that any
7855 * faults taken against the user address may not be handled successfully
7856 * (vm_fault() allows non-preemptible callers with the possibility that the
7857 * fault may not be successfully handled) and will result in the copyio operation
7858 * returning EFAULT. It is then the caller's responsibility to retry the copyio
7859 * operation in a preemptible context.
7860 *
7861 * For these cases attempting to acquire the sleepable lock will panic, so
7862 * we simply make a best effort and return failure just as the VM does if we
7863 * can't acquire the lock without sleeping.
7864 */
7865 return result;
7866 }
7867
7868 /*
7869 * If the entry doesn't exist, is completely invalid, or is already
7870 * valid, we can't fix it here.
7871 */
7872
7873 const uint64_t pmap_page_size = pt_attr_page_size(pmap_get_pt_attr(pmap)) * PAGE_RATIO;
7874 ptep = pmap_pte(pmap, va & ~(pmap_page_size - 1));
7875 if (ptep != PT_ENTRY_NULL) {
7876 while (true) {
7877 spte = os_atomic_load(ptep, relaxed);
7878
7879 pa = pte_to_pa(spte);
7880
7881 if ((spte == ARM_PTE_TYPE_FAULT) ||
7882 pte_is_compressed(spte, ptep)) {
7883 pmap_unlock(pmap, PMAP_LOCK_SHARED);
7884 return result;
7885 }
7886
7887 if (!pa_valid(pa)) {
7888 const sptm_frame_type_t frame_type = sptm_get_frame_type(pa);
7889 if (frame_type == XNU_PROTECTED_IO) {
7890 result = KERN_PROTECTION_FAILURE;
7891 }
7892 pmap_unlock(pmap, PMAP_LOCK_SHARED);
7893 return result;
7894 }
7895 pai = pa_index(pa);
7896 /**
7897 * Check for preemption disablement and in that case use pvh_try_lock()
7898 * for the same reason we use pmap_try_lock() above.
7899 */
7900 if (__probable(preemption_enabled())) {
7901 locked_pvh = pvh_lock(pai);
7902 } else {
7903 locked_pvh = pvh_try_lock(pai);
7904 if (__improbable(!pvh_try_lock_success(&locked_pvh))) {
7905 pmap_unlock(pmap, PMAP_LOCK_SHARED);
7906 return result;
7907 }
7908 }
7909 assert(locked_pvh.pvh != 0);
7910 if (os_atomic_load(ptep, relaxed) == spte) {
7911 /*
7912 * Double-check the spte value, as we care about the AF bit.
7913 * It's also possible that pmap_page_protect() transitioned the
7914 * PTE to compressed/empty before we grabbed the PVH lock.
7915 */
7916 break;
7917 }
7918 pvh_unlock(&locked_pvh);
7919 }
7920 } else {
7921 pmap_unlock(pmap, PMAP_LOCK_SHARED);
7922 return result;
7923 }
7924
7925
7926 if (result == KERN_SUCCESS) {
7927 goto ff_cleanup;
7928 }
7929
7930 pp_attr_t attrs = os_atomic_load(&pp_attr_table[pai], relaxed);
7931 if ((attrs & PP_ATTR_REFFAULT) || ((fault_type & VM_PROT_WRITE) && (attrs & PP_ATTR_MODFAULT))) {
7932 /*
7933 * An attempted access will always clear ref/mod fault state, as
7934 * appropriate for the fault type. arm_clear_fast_fault will
7935 * update the associated PTEs for the page as appropriate; if
7936 * any PTEs are updated, we redrive the access. If the mapping
7937 * does not actually allow for the attempted access, the
7938 * following fault will (hopefully) fail to update any PTEs, and
7939 * thus cause arm_fast_fault to decide that it failed to handle
7940 * the fault.
7941 */
7942 pp_attr_t attrs_to_clear = 0;
7943 if (attrs & PP_ATTR_REFFAULT) {
7944 attrs_to_clear |= PP_ATTR_REFFAULT;
7945 }
7946 if ((fault_type & VM_PROT_WRITE) && (attrs & PP_ATTR_MODFAULT)) {
7947 attrs_to_clear |= PP_ATTR_MODFAULT;
7948 }
7949
7950 if (arm_clear_fast_fault((ppnum_t)atop(pa), fault_type, locked_pvh.pvh, PT_ENTRY_NULL, attrs_to_clear)) {
7951 /*
7952 * Should this preserve KERN_PROTECTION_FAILURE? The
7953 * cost of not doing so is a another fault in a case
7954 * that should already result in an exception.
7955 */
7956 result = KERN_SUCCESS;
7957 }
7958 }
7959
7960 /*
7961 * If the PTE already has sufficient permissions, we can report the fault as handled.
7962 * This may happen, for example, if multiple threads trigger roughly simultaneous faults
7963 * on mappings of the same page
7964 */
7965 if ((result == KERN_FAILURE) && (spte & ARM_PTE_AF)) {
7966 uintptr_t ap_ro, ap_rw, ap_x;
7967 if (pmap == kernel_pmap) {
7968 ap_ro = ARM_PTE_AP(AP_RONA);
7969 ap_rw = ARM_PTE_AP(AP_RWNA);
7970 ap_x = ARM_PTE_NX;
7971 } else {
7972 ap_ro = pt_attr_leaf_ro(pmap_get_pt_attr(pmap));
7973 ap_rw = pt_attr_leaf_rw(pmap_get_pt_attr(pmap));
7974 ap_x = pt_attr_leaf_x(pmap_get_pt_attr(pmap));
7975 }
7976 /*
7977 * NOTE: this doesn't currently handle user-XO mappings. Depending upon the
7978 * hardware they may be xPRR-protected, in which case they'll be handled
7979 * by the is_pte_xprr_protected() case above. Additionally, the exception
7980 * handling path currently does not call arm_fast_fault() without at least
7981 * VM_PROT_READ in fault_type.
7982 */
7983 if (((spte & ARM_PTE_APMASK) == ap_rw) ||
7984 (!(fault_type & VM_PROT_WRITE) && ((spte & ARM_PTE_APMASK) == ap_ro))) {
7985 if (!(fault_type & VM_PROT_EXECUTE) || ((spte & ARM_PTE_XMASK) == ap_x)) {
7986 result = KERN_SUCCESS;
7987 }
7988 }
7989 }
7990
7991 if ((result == KERN_FAILURE) && arm_clear_fast_fault((ppnum_t)atop(pa), fault_type, locked_pvh.pvh, ptep, 0)) {
7992 /*
7993 * A prior arm_clear_fast_fault() operation may have returned early due to
7994 * another pending PV list operation or an excessively large PV list.
7995 * Attempt a targeted fixup of the PTE that caused the fault to avoid repeatedly
7996 * taking a fault on the same mapping.
7997 */
7998 result = KERN_SUCCESS;
7999 }
8000
8001 ff_cleanup:
8002
8003 pvh_unlock(&locked_pvh);
8004 pmap_unlock(pmap, PMAP_LOCK_SHARED);
8005 return result;
8006 }
8007
8008 kern_return_t
8009 arm_fast_fault(
8010 pmap_t pmap,
8011 vm_map_address_t va,
8012 vm_prot_t fault_type,
8013 bool was_af_fault,
8014 __unused bool from_user)
8015 {
8016 kern_return_t result = KERN_FAILURE;
8017
8018 if (va < pmap->min || va >= pmap->max) {
8019 return result;
8020 }
8021
8022 PMAP_TRACE(3, PMAP_CODE(PMAP__FAST_FAULT) | DBG_FUNC_START,
8023 VM_KERNEL_ADDRHIDE(pmap), VM_KERNEL_ADDRHIDE(va), fault_type,
8024 from_user);
8025
8026
8027 result = arm_fast_fault_internal(pmap, va, fault_type, was_af_fault, from_user);
8028
8029 PMAP_TRACE(3, PMAP_CODE(PMAP__FAST_FAULT) | DBG_FUNC_END, result);
8030
8031 return result;
8032 }
8033
8034 void
8035 pmap_copy_page(
8036 ppnum_t psrc,
8037 ppnum_t pdst)
8038 {
8039 bcopy_phys((addr64_t) (ptoa(psrc)),
8040 (addr64_t) (ptoa(pdst)),
8041 PAGE_SIZE);
8042 }
8043
8044
8045 /*
8046 * pmap_copy_page copies the specified (machine independent) pages.
8047 */
8048 void
8049 pmap_copy_part_page(
8050 ppnum_t psrc,
8051 vm_offset_t src_offset,
8052 ppnum_t pdst,
8053 vm_offset_t dst_offset,
8054 vm_size_t len)
8055 {
8056 bcopy_phys((addr64_t) (ptoa(psrc) + src_offset),
8057 (addr64_t) (ptoa(pdst) + dst_offset),
8058 len);
8059 }
8060
8061
8062 /*
8063 * pmap_zero_page zeros the specified (machine independent) page.
8064 */
8065 void
8066 pmap_zero_page(
8067 ppnum_t pn)
8068 {
8069 assert(pn != vm_page_fictitious_addr);
8070 bzero_phys((addr64_t) ptoa(pn), PAGE_SIZE);
8071 }
8072
8073 /*
8074 * pmap_zero_part_page
8075 * zeros the specified (machine independent) part of a page.
8076 */
8077 void
8078 pmap_zero_part_page(
8079 ppnum_t pn,
8080 vm_offset_t offset,
8081 vm_size_t len)
8082 {
8083 assert(pn != vm_page_fictitious_addr);
8084 assert(offset + len <= PAGE_SIZE);
8085 bzero_phys((addr64_t) (ptoa(pn) + offset), len);
8086 }
8087
8088 void
8089 pmap_map_globals(
8090 void)
8091 {
8092 pt_entry_t pte;
8093
8094 pte = pa_to_pte(kvtophys_nofail((vm_offset_t)&lowGlo)) | AP_RONA | ARM_PTE_NX | ARM_PTE_PNX | ARM_PTE_AF | ARM_PTE_TYPE;
8095 #if __ARM_KERNEL_PROTECT__
8096 pte |= ARM_PTE_NG;
8097 #endif /* __ARM_KERNEL_PROTECT__ */
8098 pte |= ARM_PTE_ATTRINDX(CACHE_ATTRINDX_WRITEBACK);
8099 pte |= ARM_PTE_SH(SH_OUTER_MEMORY);
8100 sptm_map_page(kernel_pmap->ttep, LOWGLOBAL_ALIAS, pte);
8101
8102 #if KASAN
8103 kasan_notify_address(LOWGLOBAL_ALIAS, PAGE_SIZE);
8104 #endif
8105 }
8106
8107 vm_offset_t
8108 pmap_cpu_windows_copy_addr(int cpu_num, unsigned int index)
8109 {
8110 if (__improbable(index >= CPUWINDOWS_MAX)) {
8111 panic("%s: invalid index %u", __func__, index);
8112 }
8113 return (vm_offset_t)(CPUWINDOWS_BASE + (PAGE_SIZE * ((CPUWINDOWS_MAX * cpu_num) + index)));
8114 }
8115
8116 MARK_AS_PMAP_TEXT unsigned int
8117 pmap_map_cpu_windows_copy_internal(
8118 ppnum_t pn,
8119 vm_prot_t prot,
8120 unsigned int wimg_bits)
8121 {
8122 pt_entry_t *ptep = NULL, pte;
8123 pmap_cpu_data_t *pmap_cpu_data = pmap_get_cpu_data();
8124 unsigned int cpu_num;
8125 unsigned int cpu_window_index;
8126 vm_offset_t cpu_copywindow_vaddr = 0;
8127 bool need_strong_sync = false;
8128
8129 assert(get_preemption_level() > 0);
8130 cpu_num = pmap_cpu_data->cpu_number;
8131
8132 for (cpu_window_index = 0; cpu_window_index < CPUWINDOWS_MAX; cpu_window_index++) {
8133 cpu_copywindow_vaddr = pmap_cpu_windows_copy_addr(cpu_num, cpu_window_index);
8134 ptep = pmap_pte(kernel_pmap, cpu_copywindow_vaddr);
8135 assert(!pte_is_compressed(*ptep, ptep));
8136 if (*ptep == ARM_PTE_TYPE_FAULT) {
8137 break;
8138 }
8139 }
8140 if (__improbable(cpu_window_index == CPUWINDOWS_MAX)) {
8141 panic("%s: out of windows", __func__);
8142 }
8143
8144 pte = pa_to_pte(ptoa(pn)) | ARM_PTE_TYPE | ARM_PTE_AF | ARM_PTE_NX | ARM_PTE_PNX;
8145 #if __ARM_KERNEL_PROTECT__
8146 pte |= ARM_PTE_NG;
8147 #endif /* __ARM_KERNEL_PROTECT__ */
8148 pte |= wimg_to_pte(wimg_bits, ptoa(pn));
8149
8150 if (prot & VM_PROT_WRITE) {
8151 pte |= ARM_PTE_AP(AP_RWNA);
8152 } else {
8153 pte |= ARM_PTE_AP(AP_RONA);
8154 }
8155
8156 /*
8157 * It's expected to be safe for an interrupt handler to nest copy-window usage with the
8158 * active thread on a CPU, as long as a sufficient number of copy windows are available.
8159 * --If the interrupt handler executes before the active thread creates the per-CPU mapping,
8160 * or after the active thread completely removes the mapping, it may use the same mapping
8161 * but will finish execution and tear down the mapping without the thread needing to know.
8162 * --If the interrupt handler executes after the active thread creates the per-CPU mapping,
8163 * it will observe the valid mapping and use a different copy window.
8164 * --If the interrupt handler executes after the active thread clears the PTE in
8165 * pmap_unmap_cpu_windows_copy() but before the active thread flushes the TLB, the code
8166 * for computing cpu_window_index above will observe the PTE_INVALID_IN_FLIGHT token set
8167 * by the SPTM, and will select a different index.
8168 */
8169 const sptm_return_t sptm_status = sptm_map_page(kernel_pmap->ttep, cpu_copywindow_vaddr, pte);
8170 if (__improbable(sptm_status != SPTM_SUCCESS)) {
8171 panic("%s: failed to map CPU copy-window VA 0x%llx with SPTM status %d",
8172 __func__, (unsigned long long)cpu_copywindow_vaddr, sptm_status);
8173 }
8174
8175 /*
8176 * Clean up any pending strong TLB flush for the same window in a thread we may have
8177 * interrupted.
8178 */
8179 if (__improbable(pmap_cpu_data->copywindow_strong_sync[cpu_window_index])) {
8180 arm64_sync_tlb(true);
8181 }
8182 pmap_cpu_data->copywindow_strong_sync[cpu_window_index] = need_strong_sync;
8183
8184 return cpu_window_index;
8185 }
8186
8187 unsigned int
8188 pmap_map_cpu_windows_copy(
8189 ppnum_t pn,
8190 vm_prot_t prot,
8191 unsigned int wimg_bits)
8192 {
8193 return pmap_map_cpu_windows_copy_internal(pn, prot, wimg_bits);
8194 }
8195
8196 MARK_AS_PMAP_TEXT void
8197 pmap_unmap_cpu_windows_copy_internal(
8198 unsigned int index)
8199 {
8200 unsigned int cpu_num;
8201 vm_offset_t cpu_copywindow_vaddr = 0;
8202 pmap_cpu_data_t *pmap_cpu_data = pmap_get_cpu_data();
8203
8204 assert(index < CPUWINDOWS_MAX);
8205 assert(get_preemption_level() > 0);
8206
8207 cpu_num = pmap_cpu_data->cpu_number;
8208
8209 cpu_copywindow_vaddr = pmap_cpu_windows_copy_addr(cpu_num, index);
8210 /* Issue full-system DSB to ensure prior operations on the per-CPU window
8211 * (which are likely to have been on I/O memory) are complete before
8212 * tearing down the mapping. */
8213 __builtin_arm_dsb(DSB_SY);
8214 sptm_unmap_region(kernel_pmap->ttep, cpu_copywindow_vaddr, 1, 0);
8215 if (__improbable(pmap_cpu_data->copywindow_strong_sync[index])) {
8216 arm64_sync_tlb(true);
8217 pmap_cpu_data->copywindow_strong_sync[index] = false;
8218 }
8219 }
8220
8221 void
8222 pmap_unmap_cpu_windows_copy(
8223 unsigned int index)
8224 {
8225 return pmap_unmap_cpu_windows_copy_internal(index);
8226 }
8227
8228 /*
8229 * Indicate that a pmap is intended to be used as a nested pmap
8230 * within one or more larger address spaces. This must be set
8231 * before pmap_nest() is called with this pmap as the 'subordinate'.
8232 */
8233 MARK_AS_PMAP_TEXT void
8234 pmap_set_nested_internal(
8235 pmap_t pmap)
8236 {
8237 validate_pmap_mutable(pmap);
8238 const pt_attr_t * const pt_attr = pmap_get_pt_attr(pmap);
8239 if (__improbable(pmap->type != PMAP_TYPE_USER)) {
8240 panic("%s: attempt to nest unsupported pmap %p of type 0x%hhx",
8241 __func__, pmap, pmap->type);
8242 }
8243 pmap->type = PMAP_TYPE_NESTED;
8244 sptm_retype_params_t retype_params = {.raw = SPTM_RETYPE_PARAMS_NULL};
8245 retype_params.attr_idx = (pt_attr_page_size(pt_attr) == 4096) ? SPTM_PT_GEOMETRY_4K : SPTM_PT_GEOMETRY_16K;
8246 pmap_txm_acquire_exclusive_lock(pmap);
8247 sptm_retype(pmap->ttep, XNU_USER_ROOT_TABLE, XNU_SHARED_ROOT_TABLE, retype_params);
8248 pmap_txm_release_exclusive_lock(pmap);
8249 pmap_get_pt_ops(pmap)->free_id(pmap);
8250 }
8251
8252 void
8253 pmap_set_nested(
8254 pmap_t pmap)
8255 {
8256 pmap_set_nested_internal(pmap);
8257 }
8258
8259 bool
8260 pmap_is_nested(
8261 pmap_t pmap)
8262 {
8263 return pmap->type == PMAP_TYPE_NESTED;
8264 }
8265
8266 /*
8267 * pmap_trim_range(pmap, start, end)
8268 *
8269 * pmap = pmap to operate on
8270 * start = start of the range
8271 * end = end of the range
8272 *
8273 * Attempts to deallocate TTEs for the given range in the nested range.
8274 */
8275 MARK_AS_PMAP_TEXT static void
8276 pmap_trim_range(
8277 pmap_t pmap,
8278 addr64_t start,
8279 addr64_t end)
8280 {
8281 addr64_t cur;
8282 addr64_t nested_region_start;
8283 addr64_t nested_region_end;
8284 addr64_t adjusted_start;
8285 addr64_t adjusted_end;
8286 addr64_t adjust_offmask;
8287 tt_entry_t * tte_p;
8288 __unused const pt_attr_t * const pt_attr = pmap_get_pt_attr(pmap);
8289
8290 if (__improbable(end < start)) {
8291 panic("%s: invalid address range, "
8292 "pmap=%p, start=%p, end=%p",
8293 __func__,
8294 pmap, (void*)start, (void*)end);
8295 }
8296
8297 nested_region_start = pmap->nested_region_addr;
8298 nested_region_end = nested_region_start + pmap->nested_region_size;
8299
8300 if (__improbable((start < nested_region_start) || (end > nested_region_end))) {
8301 panic("%s: range outside nested region %p-%p, "
8302 "pmap=%p, start=%p, end=%p",
8303 __func__, (void *)nested_region_start, (void *)nested_region_end,
8304 pmap, (void*)start, (void*)end);
8305 }
8306
8307 /* Contract the range to TT page boundaries. */
8308 const uint64_t page_ratio = PAGE_SIZE / pt_attr_page_size(pt_attr);
8309
8310 adjust_offmask = pt_attr_leaf_table_offmask(pt_attr) * page_ratio;
8311 adjusted_start = ((start + adjust_offmask) & ~adjust_offmask);
8312 adjusted_end = end & ~adjust_offmask;
8313
8314 /* Iterate over the range, trying to remove TTEs. */
8315 for (cur = adjusted_start; (cur < adjusted_end) && (cur >= adjusted_start); cur += (pt_attr_twig_size(pt_attr) * page_ratio)) {
8316 pmap_lock(pmap, PMAP_LOCK_EXCLUSIVE);
8317
8318 tte_p = pmap_tte(pmap, cur);
8319
8320 if ((tte_p != NULL) && ((*tte_p & ARM_TTE_TYPE_MASK) == ARM_TTE_TYPE_TABLE)) {
8321 /* pmap_tte_deallocate()/pmap_tte_trim() will drop the pmap lock */
8322 if ((pmap->type == PMAP_TYPE_NESTED) && (sptm_get_page_table_refcnt(tte_to_pa(*tte_p)) == 0)) {
8323 /* Deallocate for the nested map. */
8324 pmap_tte_deallocate(pmap, cur, tte_p, pt_attr_twig_level(pt_attr));
8325 } else if (pmap->type == PMAP_TYPE_USER) {
8326 /**
8327 * Just remove for the parent map. If the leaf table pointed
8328 * to by the TTE being removed (owned by the nested pmap)
8329 * has any mappings, then this call will panic. This
8330 * enforces the policy that tables being trimmed must be
8331 * empty to prevent possible use-after-free attacks.
8332 */
8333 pmap_tte_trim(pmap, cur, tte_p);
8334 } else {
8335 panic("%s: Unsupported pmap type for nesting %p %d", __func__, pmap, pmap->type);
8336 }
8337 } else {
8338 pmap_unlock(pmap, PMAP_LOCK_EXCLUSIVE);
8339 }
8340 }
8341 }
8342
8343 /*
8344 * pmap_trim_internal(grand, subord, vstart, size)
8345 *
8346 * grand = pmap subord is nested in
8347 * subord = nested pmap
8348 * vstart = start of the used range in grand
8349 * size = size of the used range
8350 *
8351 * Attempts to trim the shared region page tables down to only cover the given
8352 * range in subord and grand.
8353 */
8354 MARK_AS_PMAP_TEXT void
8355 pmap_trim_internal(
8356 pmap_t grand,
8357 pmap_t subord,
8358 addr64_t vstart,
8359 uint64_t size)
8360 {
8361 addr64_t vend;
8362 addr64_t adjust_offmask;
8363
8364 if (__improbable(os_add_overflow(vstart, size, &vend))) {
8365 panic("%s: grand addr wraps around, "
8366 "grand=%p, subord=%p, vstart=%p, size=%#llx",
8367 __func__, grand, subord, (void*)vstart, size);
8368 }
8369
8370 validate_pmap_mutable(grand);
8371 validate_pmap(subord);
8372
8373 __unused const pt_attr_t * const pt_attr = pmap_get_pt_attr(grand);
8374
8375 pmap_lock(subord, PMAP_LOCK_EXCLUSIVE);
8376
8377 if (__improbable(subord->type != PMAP_TYPE_NESTED)) {
8378 panic("%s: subord is of non-nestable type 0x%hhx, "
8379 "grand=%p, subord=%p, vstart=%p, size=%#llx",
8380 __func__, subord->type, grand, subord, (void*)vstart, size);
8381 }
8382
8383 if (__improbable(grand->type != PMAP_TYPE_USER)) {
8384 panic("%s: grand is of unsupprted type 0x%hhx for nesting, "
8385 "grand=%p, subord=%p, vstart=%p, size=%#llx",
8386 __func__, grand->type, grand, subord, (void*)vstart, size);
8387 }
8388
8389 if (__improbable(grand->nested_pmap != subord)) {
8390 panic("%s: grand->nested != subord, "
8391 "grand=%p, subord=%p, vstart=%p, size=%#llx",
8392 __func__, grand, subord, (void*)vstart, size);
8393 }
8394
8395 if (__improbable((size != 0) &&
8396 ((vstart < grand->nested_region_addr) || (vend > (grand->nested_region_addr + grand->nested_region_size))))) {
8397 panic("%s: grand range not in nested region, "
8398 "grand=%p, subord=%p, vstart=%p, size=%#llx",
8399 __func__, grand, subord, (void*)vstart, size);
8400 }
8401
8402
8403 if (!grand->nested_has_no_bounds_ref) {
8404 assert(subord->nested_bounds_set);
8405
8406 if (!grand->nested_bounds_set) {
8407 /* Inherit the bounds from subord. */
8408 grand->nested_region_true_start = subord->nested_region_true_start;
8409 grand->nested_region_true_end = subord->nested_region_true_end;
8410 grand->nested_bounds_set = true;
8411 }
8412
8413 pmap_unlock(subord, PMAP_LOCK_EXCLUSIVE);
8414 return;
8415 }
8416
8417 if ((!subord->nested_bounds_set) && size) {
8418 const uint64_t page_ratio = PAGE_SIZE / pt_attr_page_size(pt_attr);
8419 adjust_offmask = pt_attr_leaf_table_offmask(pt_attr) * page_ratio;
8420
8421 subord->nested_region_true_start = vstart;
8422 subord->nested_region_true_end = vend;
8423 subord->nested_region_true_start &= ~adjust_offmask;
8424
8425 if (__improbable(os_add_overflow(subord->nested_region_true_end, adjust_offmask, &subord->nested_region_true_end))) {
8426 panic("%s: padded true end wraps around, "
8427 "grand=%p, subord=%p, vstart=%p, size=%#llx",
8428 __func__, grand, subord, (void*)vstart, size);
8429 }
8430
8431 subord->nested_region_true_end &= ~adjust_offmask;
8432 subord->nested_bounds_set = true;
8433 }
8434
8435 if (subord->nested_bounds_set) {
8436 /* Inherit the bounds from subord. */
8437 grand->nested_region_true_start = subord->nested_region_true_start;
8438 grand->nested_region_true_end = subord->nested_region_true_end;
8439 grand->nested_bounds_set = true;
8440
8441 /* If we know the bounds, we can trim the pmap. */
8442 grand->nested_has_no_bounds_ref = false;
8443 pmap_unlock(subord, PMAP_LOCK_EXCLUSIVE);
8444 } else {
8445 /* Don't trim if we don't know the bounds. */
8446 pmap_unlock(subord, PMAP_LOCK_EXCLUSIVE);
8447 return;
8448 }
8449
8450 /* Trim grand to only cover the given range. */
8451 pmap_trim_range(grand, grand->nested_region_addr, grand->nested_region_true_start);
8452 pmap_trim_range(grand, grand->nested_region_true_end, (grand->nested_region_addr + grand->nested_region_size));
8453
8454 /* Try to trim subord. */
8455 pmap_trim_subord(subord);
8456 }
8457
8458 MARK_AS_PMAP_TEXT static void
8459 pmap_trim_self(pmap_t pmap)
8460 {
8461 if (pmap->nested_has_no_bounds_ref && pmap->nested_pmap) {
8462 /* If we have a no bounds ref, we need to drop it. */
8463 pmap_lock(pmap->nested_pmap, PMAP_LOCK_SHARED);
8464 pmap->nested_has_no_bounds_ref = false;
8465 boolean_t nested_bounds_set = pmap->nested_pmap->nested_bounds_set;
8466 vm_map_offset_t nested_region_true_start = pmap->nested_pmap->nested_region_true_start;
8467 vm_map_offset_t nested_region_true_end = pmap->nested_pmap->nested_region_true_end;
8468 pmap_unlock(pmap->nested_pmap, PMAP_LOCK_SHARED);
8469
8470 if (nested_bounds_set) {
8471 pmap_trim_range(pmap, pmap->nested_region_addr, nested_region_true_start);
8472 pmap_trim_range(pmap, nested_region_true_end, (pmap->nested_region_addr + pmap->nested_region_size));
8473 }
8474 /*
8475 * Try trimming the nested pmap, in case we had the
8476 * last reference.
8477 */
8478 pmap_trim_subord(pmap->nested_pmap);
8479 }
8480 }
8481
8482 /*
8483 * pmap_trim_subord(grand, subord)
8484 *
8485 * grand = pmap that we have nested subord in
8486 * subord = nested pmap we are attempting to trim
8487 *
8488 * Trims subord if possible
8489 */
8490 MARK_AS_PMAP_TEXT static void
8491 pmap_trim_subord(pmap_t subord)
8492 {
8493 bool contract_subord = false;
8494
8495 pmap_lock(subord, PMAP_LOCK_EXCLUSIVE);
8496
8497 subord->nested_no_bounds_refcnt--;
8498
8499 if ((subord->nested_no_bounds_refcnt == 0) && (subord->nested_bounds_set)) {
8500 /* If this was the last no bounds reference, trim subord. */
8501 contract_subord = true;
8502 }
8503
8504 pmap_unlock(subord, PMAP_LOCK_EXCLUSIVE);
8505
8506 if (contract_subord) {
8507 pmap_trim_range(subord, subord->nested_region_addr, subord->nested_region_true_start);
8508 pmap_trim_range(subord, subord->nested_region_true_end, subord->nested_region_addr + subord->nested_region_size);
8509 }
8510 }
8511
8512 void
8513 pmap_trim(
8514 pmap_t grand,
8515 pmap_t subord,
8516 addr64_t vstart,
8517 uint64_t size)
8518 {
8519 pmap_trim_internal(grand, subord, vstart, size);
8520 }
8521
8522 #if HAS_APPLE_PAC
8523
8524 void *
8525 pmap_sign_user_ptr(void *value, ptrauth_key key, uint64_t discriminator, uint64_t jop_key)
8526 {
8527 void *res = NULL;
8528 const boolean_t current_intr_state = ml_set_interrupts_enabled(FALSE);
8529
8530 uint64_t saved_jop_state = ml_enable_user_jop_key(jop_key);
8531 __compiler_materialize_and_prevent_reordering_on(value);
8532 res = sptm_sign_user_pointer(value, key, discriminator, jop_key);
8533 __compiler_materialize_and_prevent_reordering_on(res);
8534 ml_disable_user_jop_key(jop_key, saved_jop_state);
8535
8536 ml_set_interrupts_enabled(current_intr_state);
8537
8538 return res;
8539 }
8540
8541 void *
8542 pmap_auth_user_ptr(void *value, ptrauth_key key, uint64_t discriminator, uint64_t jop_key)
8543 {
8544 void *res = NULL;
8545 const boolean_t current_intr_state = ml_set_interrupts_enabled(FALSE);
8546
8547 uint64_t saved_jop_state = ml_enable_user_jop_key(jop_key);
8548 __compiler_materialize_and_prevent_reordering_on(value);
8549 res = sptm_auth_user_pointer(value, key, discriminator, jop_key);
8550 __compiler_materialize_and_prevent_reordering_on(res);
8551 ml_disable_user_jop_key(jop_key, saved_jop_state);
8552
8553 ml_set_interrupts_enabled(current_intr_state);
8554
8555 return res;
8556 }
8557 #endif /* HAS_APPLE_PAC */
8558
8559 /*
8560 * kern_return_t pmap_nest(grand, subord, vstart, size)
8561 *
8562 * grand = the pmap that we will nest subord into
8563 * subord = the pmap that goes into the grand
8564 * vstart = start of range in pmap to be inserted
8565 * size = Size of nest area (up to 16TB)
8566 *
8567 * Inserts a pmap into another. This is used to implement shared segments.
8568 *
8569 */
8570
8571 /**
8572 * Embeds a range of mappings from one pmap ('subord') into another ('grand')
8573 * by inserting the twig-level TTEs from 'subord' directly into 'grand'.
8574 * This function operates in 3 main phases:
8575 * 1. Bookkeeping to ensure tracking structures for the nested region are set up.
8576 * 2. Expansion of subord to ensure the required leaf-level page table pages for
8577 * the mapping range are present in subord.
8578 * 3. Expansion of grand to ensure the required twig-level page table pages for
8579 * the mapping range are present in grand.
8580 * 4. Invoke sptm_nest_region() to copy the relevant TTEs from subord to grand.
8581 *
8582 * This function may return early due to pending AST_URGENT preemption; if so
8583 * it will indicate the need to be re-entered.
8584 *
8585 * @param grand pmap to insert the TTEs into. Must be a user pmap.
8586 * @param subord pmap from which to extract the TTEs. Must be a nested pmap.
8587 * @param vstart twig-aligned virtual address for the beginning of the nesting range
8588 * @param size twig-aligned size of the nesting range
8589 *
8590 * @return KERN_RESOURCE_SHORTAGE on allocation failure, KERN_SUCCESS otherwise
8591 */
8592 MARK_AS_PMAP_TEXT kern_return_t
8593 pmap_nest_internal(
8594 pmap_t grand,
8595 pmap_t subord,
8596 addr64_t vstart,
8597 uint64_t size)
8598 {
8599 kern_return_t kr = KERN_SUCCESS;
8600 vm_map_offset_t vaddr;
8601 tt_entry_t *stte_p;
8602 tt_entry_t *gtte_p;
8603 bitmap_t *nested_region_unnested_table_bitmap;
8604 int expand_options = 0;
8605 bool deref_subord = true;
8606
8607 addr64_t vend;
8608 if (__improbable(os_add_overflow(vstart, size, &vend))) {
8609 panic("%s: %p grand addr wraps around: 0x%llx + 0x%llx", __func__, grand, vstart, size);
8610 }
8611
8612 validate_pmap_mutable(grand);
8613 validate_pmap(subord);
8614 os_ref_retain_raw(&subord->ref_count, &pmap_refgrp);
8615
8616 const pt_attr_t * const pt_attr = pmap_get_pt_attr(grand);
8617 if (__improbable(pmap_get_pt_attr(subord) != pt_attr)) {
8618 panic("%s: attempt to nest pmap %p into pmap %p with mismatched attributes", __func__, subord, grand);
8619 }
8620
8621 if (__improbable(((size | vstart) &
8622 (pt_attr_leaf_table_offmask(pt_attr))) != 0x0ULL)) {
8623 panic("pmap_nest() pmap %p unaligned nesting request 0x%llx, 0x%llx",
8624 grand, vstart, size);
8625 }
8626
8627 if (__improbable(subord->type != PMAP_TYPE_NESTED)) {
8628 panic("%s: subordinate pmap %p is of non-nestable type 0x%hhx", __func__, subord, subord->type);
8629 }
8630
8631 if (__improbable(grand->type != PMAP_TYPE_USER)) {
8632 panic("%s: grand pmap %p is of unsupported type 0x%hhx for nesting", __func__, grand, grand->type);
8633 }
8634
8635 /**
8636 * Use an acquire barrier to ensure that subsequent loads of nested_region_* fields are not
8637 * speculated ahead of the load of nested_region_unnested_table_bitmap, so that if we observe a non-NULL
8638 * nested_region_unnested_table_bitmap then we can be sure the other fields have been initialized as well.
8639 */
8640 if (os_atomic_load(&subord->nested_region_unnested_table_bitmap, acquire) == NULL) {
8641 uint64_t nested_region_unnested_table_bits = size >> pt_attr_twig_shift(pt_attr);
8642
8643 if (__improbable((nested_region_unnested_table_bits > UINT_MAX))) {
8644 panic("%s: bitmap allocation size %llu will truncate, "
8645 "grand=%p, subord=%p, vstart=0x%llx, size=%llx",
8646 __func__, nested_region_unnested_table_bits,
8647 grand, subord, vstart, size);
8648 }
8649
8650 nested_region_unnested_table_bitmap = bitmap_alloc((uint) nested_region_unnested_table_bits);
8651
8652 pmap_lock(subord, PMAP_LOCK_EXCLUSIVE);
8653 if (subord->nested_region_unnested_table_bitmap == NULL) {
8654 subord->nested_region_addr = vstart;
8655 subord->nested_region_size = (mach_vm_offset_t) size;
8656 sptm_configure_shared_region(subord->ttep, vstart, size >> pt_attr->pta_page_shift);
8657
8658 /**
8659 * Ensure that the rest of the subord->nested_region_* fields are
8660 * initialized and visible before setting the nested_region_unnested_table_bitmap
8661 * field (which is used as the flag to say that the rest are initialized).
8662 */
8663 os_atomic_store(&subord->nested_region_unnested_table_bitmap, nested_region_unnested_table_bitmap, release);
8664 nested_region_unnested_table_bitmap = NULL;
8665 }
8666 pmap_unlock(subord, PMAP_LOCK_EXCLUSIVE);
8667 if (nested_region_unnested_table_bitmap != NULL) {
8668 bitmap_free(nested_region_unnested_table_bitmap, nested_region_unnested_table_bits);
8669 }
8670 }
8671
8672 assertf(subord->nested_region_addr == vstart, "%s: pmap %p nested region addr 0x%llx doesn't match vstart 0x%llx",
8673 __func__, subord, (unsigned long long)subord->nested_region_addr, (unsigned long long)vstart);
8674 assertf(subord->nested_region_size == size, "%s: pmap %p nested region size 0x%llx doesn't match size 0x%llx",
8675 __func__, subord, (unsigned long long)subord->nested_region_size, (unsigned long long)size);
8676
8677 pmap_lock(subord, PMAP_LOCK_EXCLUSIVE);
8678
8679 if (os_atomic_cmpxchg(&grand->nested_pmap, PMAP_NULL, subord, relaxed)) {
8680 /*
8681 * If this is grand's first nesting operation, keep the reference on subord.
8682 * It will be released by pmap_destroy_internal() when grand is destroyed.
8683 */
8684 deref_subord = false;
8685
8686 if (!subord->nested_bounds_set) {
8687 /*
8688 * We are nesting without the shared regions bounds
8689 * being known. We'll have to trim the pmap later.
8690 */
8691 grand->nested_has_no_bounds_ref = true;
8692 subord->nested_no_bounds_refcnt++;
8693 }
8694
8695 grand->nested_region_addr = vstart;
8696 grand->nested_region_size = (mach_vm_offset_t) size;
8697 } else {
8698 if (__improbable(grand->nested_pmap != subord)) {
8699 panic("pmap_nest() pmap %p has a nested pmap", grand);
8700 } else if (__improbable(grand->nested_region_addr > vstart)) {
8701 panic("pmap_nest() pmap %p : attempt to nest outside the nested region", grand);
8702 } else if ((grand->nested_region_addr + grand->nested_region_size) < vend) {
8703 grand->nested_region_size = (mach_vm_offset_t)(vstart - grand->nested_region_addr + size);
8704 }
8705 }
8706
8707 vaddr = vstart;
8708 if (vaddr < subord->nested_region_true_start) {
8709 vaddr = subord->nested_region_true_start;
8710 }
8711
8712 addr64_t true_end = vend;
8713 if (true_end > subord->nested_region_true_end) {
8714 true_end = subord->nested_region_true_end;
8715 }
8716
8717 while (vaddr < true_end) {
8718 stte_p = pmap_tte(subord, vaddr);
8719 if (stte_p == PT_ENTRY_NULL || *stte_p == ARM_TTE_EMPTY) {
8720 pmap_unlock(subord, PMAP_LOCK_EXCLUSIVE);
8721 kr = pmap_expand(subord, vaddr, expand_options, pt_attr_leaf_level(pt_attr));
8722
8723 if (kr != KERN_SUCCESS) {
8724 pmap_lock(grand, PMAP_LOCK_EXCLUSIVE);
8725 goto done;
8726 }
8727
8728 pmap_lock(subord, PMAP_LOCK_EXCLUSIVE);
8729 }
8730 vaddr += pt_attr_twig_size(pt_attr);
8731 }
8732
8733 /*
8734 * copy TTEs from subord pmap into grand pmap
8735 */
8736
8737 vaddr = (vm_map_offset_t) vstart;
8738 if (vaddr < subord->nested_region_true_start) {
8739 vaddr = subord->nested_region_true_start;
8740 }
8741
8742 pmap_unlock(subord, PMAP_LOCK_EXCLUSIVE);
8743 pmap_lock(grand, PMAP_LOCK_EXCLUSIVE);
8744
8745 while (vaddr < true_end) {
8746 gtte_p = pmap_tte(grand, vaddr);
8747 if (gtte_p == PT_ENTRY_NULL) {
8748 pmap_unlock(grand, PMAP_LOCK_EXCLUSIVE);
8749 kr = pmap_expand(grand, vaddr, expand_options, pt_attr_twig_level(pt_attr));
8750 pmap_lock(grand, PMAP_LOCK_EXCLUSIVE);
8751
8752 if (kr != KERN_SUCCESS) {
8753 goto done;
8754 }
8755 }
8756
8757 vaddr += pt_attr_twig_size(pt_attr);
8758 }
8759
8760 vaddr = (vm_map_offset_t) vstart;
8761
8762 /*
8763 * It is possible to have a preempted nest operation execute concurrently
8764 * with a trim operation that sets nested_region_true_start. In this case,
8765 * update the nesting bounds. This is useful both as a performance
8766 * optimization and to prevent an attempt to nest a just-trimmed TTE,
8767 * which will trigger an SPTM violation.
8768 * Note that pmap_trim() may concurrently update grand's bounds as we are
8769 * making these checks, but in that case pmap_trim_range() has not yet
8770 * been called on grand and will wait for us to drop grand's lock, so it
8771 * should see any TTEs we've nested here and clear them appropriately.
8772 */
8773 if (vaddr < subord->nested_region_true_start) {
8774 vaddr = subord->nested_region_true_start;
8775 }
8776 if (vaddr < grand->nested_region_true_start) {
8777 vaddr = grand->nested_region_true_start;
8778 }
8779 if (true_end > subord->nested_region_true_end) {
8780 true_end = subord->nested_region_true_end;
8781 }
8782 if (true_end > grand->nested_region_true_end) {
8783 true_end = grand->nested_region_true_end;
8784 }
8785
8786 while (vaddr < true_end) {
8787 /*
8788 * The SPTM requires the run of TTE updates to all reside within the same L2 page, so the region
8789 * we supply to the SPTM can't span multiple L1 TTEs.
8790 */
8791 vm_map_offset_t vlim = ((vaddr + pt_attr_ln_size(pt_attr, PMAP_TT_L1_LEVEL)) & ~pt_attr_ln_offmask(pt_attr, PMAP_TT_L1_LEVEL));
8792 if (vlim > true_end) {
8793 vlim = true_end;
8794 }
8795 pmap_txm_acquire_exclusive_lock(grand);
8796 pmap_txm_acquire_shared_lock(subord);
8797 sptm_nest_region(grand->ttep, subord->ttep, vaddr, (vlim - vaddr) >> pt_attr->pta_page_shift);
8798 pmap_txm_release_shared_lock(subord);
8799 pmap_txm_release_exclusive_lock(grand);
8800 vaddr = vlim;
8801 }
8802
8803 done:
8804 pmap_unlock(grand, PMAP_LOCK_EXCLUSIVE);
8805 if (deref_subord) {
8806 pmap_destroy_internal(subord);
8807 }
8808
8809 return kr;
8810 }
8811
8812 kern_return_t
8813 pmap_nest(
8814 pmap_t grand,
8815 pmap_t subord,
8816 addr64_t vstart,
8817 uint64_t size)
8818 {
8819 kern_return_t kr = KERN_SUCCESS;
8820
8821 PMAP_TRACE(2, PMAP_CODE(PMAP__NEST) | DBG_FUNC_START,
8822 VM_KERNEL_ADDRHIDE(grand), VM_KERNEL_ADDRHIDE(subord),
8823 VM_KERNEL_ADDRHIDE(vstart));
8824
8825 pmap_verify_preemptible();
8826 kr = pmap_nest_internal(grand, subord, vstart, size);
8827
8828 PMAP_TRACE(2, PMAP_CODE(PMAP__NEST) | DBG_FUNC_END, kr);
8829
8830 return kr;
8831 }
8832
8833 /*
8834 * kern_return_t pmap_unnest(grand, vaddr)
8835 *
8836 * grand = the pmap that will have the virtual range unnested
8837 * vaddr = start of range in pmap to be unnested
8838 * size = size of range in pmap to be unnested
8839 *
8840 */
8841
8842 kern_return_t
8843 pmap_unnest(
8844 pmap_t grand,
8845 addr64_t vaddr,
8846 uint64_t size)
8847 {
8848 return pmap_unnest_options(grand, vaddr, size, 0);
8849 }
8850
8851 /**
8852 * Undoes a prior pmap_nest() operation by removing a range of nesting mappings
8853 * from a top-level pmap ('grand'). The corresponding mappings in the nested
8854 * pmap will be marked non-global to avoid TLB conflicts with pmaps that may
8855 * still have the region nested. The mappings in 'grand' will be left empty
8856 * with the assumption that they will be demand-filled by subsequent access faults.
8857 *
8858 * This function operates in 2 main phases:
8859 * 1. Iteration over the nested pmap's mappings for the specified range to mark
8860 * them non-global.
8861 * 2. Calling the SPTM to clear the twig-level TTEs for the address range in grand.
8862 *
8863 * This function may return early due to pending AST_URGENT preemption; if so
8864 * it will indicate the need to be re-entered.
8865 *
8866 * @param grand pmap from which to unnest mappings
8867 * @param vaddr twig-aligned virtual address for the beginning of the nested range
8868 * @param size twig-aligned size of the nested range
8869 * @param option Extra control flags; may contain PMAP_UNNEST_CLEAN to indicate that
8870 * grand is being torn down and step 1) above is not needed.
8871 */
8872 MARK_AS_PMAP_TEXT void
8873 pmap_unnest_options_internal(
8874 pmap_t grand,
8875 addr64_t vaddr,
8876 uint64_t size,
8877 unsigned int option)
8878 {
8879 vm_map_offset_t start;
8880 vm_map_offset_t addr;
8881 unsigned int current_index;
8882 unsigned int start_index;
8883 unsigned int max_index;
8884
8885 addr64_t vend;
8886 addr64_t true_end;
8887 if (__improbable(os_add_overflow(vaddr, size, &vend))) {
8888 panic("%s: %p vaddr wraps around: 0x%llx + 0x%llx", __func__, grand, vaddr, size);
8889 }
8890
8891 validate_pmap_mutable(grand);
8892
8893 __unused const pt_attr_t * const pt_attr = pmap_get_pt_attr(grand);
8894
8895 if (__improbable(((size | vaddr) & pt_attr_twig_offmask(pt_attr)) != 0x0ULL)) {
8896 panic("%s: unaligned base address 0x%llx or size 0x%llx", __func__,
8897 (unsigned long long)vaddr, (unsigned long long)size);
8898 }
8899
8900 if (__improbable(grand->nested_pmap == NULL)) {
8901 panic("%s: %p has no nested pmap", __func__, grand);
8902 }
8903
8904 true_end = vend;
8905 if (true_end > grand->nested_pmap->nested_region_true_end) {
8906 true_end = grand->nested_pmap->nested_region_true_end;
8907 }
8908
8909 if ((option & PMAP_UNNEST_CLEAN) == 0) {
8910 if ((vaddr < grand->nested_region_addr) || (vend > (grand->nested_region_addr + grand->nested_region_size))) {
8911 panic("%s: %p: unnest request to not-fully-nested region [%p, %p)", __func__, grand, (void*)vaddr, (void*)vend);
8912 }
8913
8914 /*
8915 * SPTM TODO: I suspect we may be able to hold the nested pmap lock shared here.
8916 * We would need to use atomic_bitmap_set below where we currently use bitmap_test + bitmap_set.
8917 * The risk is that a concurrent pmap_enter() against the nested pmap could observe the relevant
8918 * bit in the nested region bitmap to be clear, but could then create the (global) mapping after
8919 * we've made our SPTM sweep below to set NG. In that case we could end up with a mix of global
8920 * and non-global mappings for the same VA region and thus a TLB conflict. I'm uncertain if the
8921 * VM would allow these operation to happen concurrently. Even if it does, we could still do
8922 * something fancier here such as waiting for concurrent pmap_enter() to drain after updating
8923 * the bitmap.
8924 */
8925 pmap_lock(grand->nested_pmap, PMAP_LOCK_EXCLUSIVE);
8926
8927 disable_preemption();
8928 pmap_sptm_percpu_data_t *sptm_pcpu = PERCPU_GET(pmap_sptm_percpu);
8929 unsigned int num_mappings = 0;
8930 start = vaddr;
8931 if (start < grand->nested_pmap->nested_region_true_start) {
8932 start = grand->nested_pmap->nested_region_true_start;
8933 }
8934 start_index = (unsigned int)((start - grand->nested_region_addr) >> pt_attr_twig_shift(pt_attr));
8935 max_index = (unsigned int)((true_end - grand->nested_region_addr) >> pt_attr_twig_shift(pt_attr));
8936
8937 for (current_index = start_index, addr = start; current_index < max_index; current_index++) {
8938 pt_entry_t *bpte, *cpte;
8939
8940 vm_map_offset_t vlim = (addr + pt_attr_twig_size(pt_attr)) & ~pt_attr_twig_offmask(pt_attr);
8941
8942 bpte = pmap_pte(grand->nested_pmap, addr);
8943
8944 if (!bitmap_test(grand->nested_pmap->nested_region_unnested_table_bitmap, current_index)) {
8945 /*
8946 * We've marked the 'twig' region as being unnested. Every mapping entered within
8947 * the nested pmap in this region will now be marked non-global.
8948 */
8949 bitmap_set(grand->nested_pmap->nested_region_unnested_table_bitmap, current_index);
8950 for (cpte = bpte; (bpte != NULL) && (addr < vlim); cpte += PAGE_RATIO) {
8951 pt_entry_t spte = os_atomic_load(cpte, relaxed);
8952
8953 if ((spte & ARM_PTE_TYPE_MASK) != ARM_PTE_TYPE_FAULT) {
8954 spte |= ARM_PTE_NG;
8955 }
8956
8957 addr += (pt_attr_page_size(pt_attr) * PAGE_RATIO);
8958
8959 sptm_pcpu->sptm_templates[num_mappings] = spte;
8960 ++num_mappings;
8961
8962 if (num_mappings == SPTM_MAPPING_LIMIT) {
8963 pmap_retype_epoch_enter();
8964 sptm_update_region(grand->nested_pmap->ttep, start, num_mappings,
8965 sptm_pcpu->sptm_templates_pa, SPTM_UPDATE_NG);
8966 pmap_retype_epoch_exit();
8967 enable_preemption();
8968 num_mappings = 0;
8969 start = addr;
8970 disable_preemption();
8971 sptm_pcpu = PERCPU_GET(pmap_sptm_percpu);
8972 }
8973 }
8974 }
8975 /**
8976 * The SPTM does not allow region updates to span multiple leaf page tables, so request
8977 * any remaining updates up to vlim before moving to the next page table page.
8978 */
8979 if (num_mappings != 0) {
8980 pmap_retype_epoch_enter();
8981 sptm_update_region(grand->nested_pmap->ttep, start, num_mappings,
8982 sptm_pcpu->sptm_templates_pa, SPTM_UPDATE_NG);
8983 pmap_retype_epoch_exit();
8984 enable_preemption();
8985 num_mappings = 0;
8986 disable_preemption();
8987 sptm_pcpu = PERCPU_GET(pmap_sptm_percpu);
8988 }
8989 addr = start = vlim;
8990 }
8991
8992 if (num_mappings != 0) {
8993 pmap_retype_epoch_enter();
8994 sptm_update_region(grand->nested_pmap->ttep, start, num_mappings,
8995 sptm_pcpu->sptm_templates_pa, SPTM_UPDATE_NG);
8996 pmap_retype_epoch_exit();
8997 }
8998
8999 enable_preemption();
9000 pmap_unlock(grand->nested_pmap, PMAP_LOCK_EXCLUSIVE);
9001 }
9002
9003 /*
9004 * invalidate all pdes for segment at vaddr in pmap grand
9005 */
9006 addr = vaddr;
9007
9008 pmap_lock(grand, PMAP_LOCK_EXCLUSIVE);
9009
9010 if (addr < grand->nested_pmap->nested_region_true_start) {
9011 addr = grand->nested_pmap->nested_region_true_start;
9012 }
9013
9014 if (true_end > grand->nested_pmap->nested_region_true_end) {
9015 true_end = grand->nested_pmap->nested_region_true_end;
9016 }
9017
9018 while (addr < true_end) {
9019 vm_map_offset_t vlim = ((addr + pt_attr_ln_size(pt_attr, PMAP_TT_L1_LEVEL)) & ~pt_attr_ln_offmask(pt_attr, PMAP_TT_L1_LEVEL));
9020 if (vlim > true_end) {
9021 vlim = true_end;
9022 }
9023 sptm_unnest_region(grand->ttep, grand->nested_pmap->ttep, addr, (vlim - addr) >> pt_attr->pta_page_shift);
9024 addr = vlim;
9025 }
9026
9027 pmap_unlock(grand, PMAP_LOCK_EXCLUSIVE);
9028 }
9029
9030 kern_return_t
9031 pmap_unnest_options(
9032 pmap_t grand,
9033 addr64_t vaddr,
9034 uint64_t size,
9035 unsigned int option)
9036 {
9037 PMAP_TRACE(2, PMAP_CODE(PMAP__UNNEST) | DBG_FUNC_START,
9038 VM_KERNEL_ADDRHIDE(grand), VM_KERNEL_ADDRHIDE(vaddr));
9039
9040 pmap_verify_preemptible();
9041 pmap_unnest_options_internal(grand, vaddr, size, option);
9042
9043 PMAP_TRACE(2, PMAP_CODE(PMAP__UNNEST) | DBG_FUNC_END, KERN_SUCCESS);
9044
9045 return KERN_SUCCESS;
9046 }
9047
9048 boolean_t
9049 pmap_adjust_unnest_parameters(
9050 __unused pmap_t p,
9051 __unused vm_map_offset_t *s,
9052 __unused vm_map_offset_t *e)
9053 {
9054 return TRUE; /* to get to log_unnest_badness()... */
9055 }
9056
9057 #if PMAP_FORK_NEST
9058 /**
9059 * Perform any necessary pre-nesting of the parent's shared region at fork()
9060 * time.
9061 *
9062 * @note This should only be called from vm_map_fork().
9063 *
9064 * @param old_pmap The pmap of the parent task.
9065 * @param new_pmap The pmap of the child task.
9066 * @param nesting_start An output parameter that is updated with the start
9067 * address of the range that was pre-nested
9068 * @param nesting_end An output parameter that is updated with the end
9069 * address of the range that was pre-nested
9070 *
9071 * @return KERN_SUCCESS if the pre-nesting was succesfully completed.
9072 * KERN_INVALID_ARGUMENT if the arguments were not valid.
9073 */
9074 kern_return_t
9075 pmap_fork_nest(
9076 pmap_t old_pmap,
9077 pmap_t new_pmap,
9078 vm_map_offset_t *nesting_start,
9079 vm_map_offset_t *nesting_end)
9080 {
9081 if (old_pmap == NULL || new_pmap == NULL) {
9082 return KERN_INVALID_ARGUMENT;
9083 }
9084 if (old_pmap->nested_pmap == NULL) {
9085 return KERN_SUCCESS;
9086 }
9087 pmap_nest(new_pmap,
9088 old_pmap->nested_pmap,
9089 old_pmap->nested_region_addr,
9090 old_pmap->nested_region_size);
9091 assertf(new_pmap->nested_pmap == old_pmap->nested_pmap &&
9092 new_pmap->nested_region_addr == old_pmap->nested_region_addr &&
9093 new_pmap->nested_region_size == old_pmap->nested_region_size,
9094 "nested new (%p,0x%llx,0x%llx) old (%p,0x%llx,0x%llx)",
9095 new_pmap->nested_pmap,
9096 new_pmap->nested_region_addr,
9097 new_pmap->nested_region_size,
9098 old_pmap->nested_pmap,
9099 old_pmap->nested_region_addr,
9100 old_pmap->nested_region_size);
9101 *nesting_start = old_pmap->nested_region_addr;
9102 *nesting_end = *nesting_start + old_pmap->nested_region_size;
9103 return KERN_SUCCESS;
9104 }
9105 #endif /* PMAP_FORK_NEST */
9106
9107 /*
9108 * disable no-execute capability on
9109 * the specified pmap
9110 */
9111 #if DEVELOPMENT || DEBUG
9112 void
9113 pmap_disable_NX(
9114 pmap_t pmap)
9115 {
9116 pmap->nx_enabled = FALSE;
9117 }
9118 #else
9119 void
9120 pmap_disable_NX(
9121 __unused pmap_t pmap)
9122 {
9123 }
9124 #endif
9125
9126 /*
9127 * flush a range of hardware TLB entries.
9128 * NOTE: assumes the smallest TLB entry in use will be for
9129 * an ARM small page (4K).
9130 */
9131
9132 #if __ARM_RANGE_TLBI__
9133 #define ARM64_RANGE_TLB_FLUSH_THRESHOLD 1
9134 #define ARM64_FULL_TLB_FLUSH_THRESHOLD ARM64_TLB_RANGE_MAX_PAGES
9135 #else
9136 #define ARM64_FULL_TLB_FLUSH_THRESHOLD 256
9137 #endif // __ARM_RANGE_TLBI__
9138
9139 static void
9140 flush_mmu_tlb_region_asid_async(
9141 vm_offset_t va,
9142 size_t length,
9143 pmap_t pmap,
9144 bool last_level_only __unused)
9145 {
9146 unsigned long pmap_page_shift = pt_attr_leaf_shift(pmap_get_pt_attr(pmap));
9147 const uint64_t pmap_page_size = 1ULL << pmap_page_shift;
9148 ppnum_t npages = (ppnum_t)(length >> pmap_page_shift);
9149 const uint16_t asid = PMAP_HWASID(pmap);
9150
9151 if (npages > ARM64_FULL_TLB_FLUSH_THRESHOLD) {
9152 boolean_t flush_all = FALSE;
9153
9154 if ((asid == 0) || (pmap->type == PMAP_TYPE_NESTED)) {
9155 flush_all = TRUE;
9156 }
9157 if (flush_all) {
9158 flush_mmu_tlb_async();
9159 } else {
9160 flush_mmu_tlb_asid_async((uint64_t)asid << TLBI_ASID_SHIFT, false);
9161 }
9162 return;
9163 }
9164 #if __ARM_RANGE_TLBI__
9165 if (npages > ARM64_RANGE_TLB_FLUSH_THRESHOLD) {
9166 va = generate_rtlbi_param(npages, asid, va, pmap_page_shift);
9167 if (pmap->type == PMAP_TYPE_NESTED) {
9168 flush_mmu_tlb_allrange_async(va, last_level_only, false);
9169 } else {
9170 flush_mmu_tlb_range_async(va, last_level_only, false);
9171 }
9172 return;
9173 }
9174 #endif
9175 vm_offset_t end = tlbi_asid(asid) | tlbi_addr(va + length);
9176 va = tlbi_asid(asid) | tlbi_addr(va);
9177
9178 if (pmap->type == PMAP_TYPE_NESTED) {
9179 flush_mmu_tlb_allentries_async(va, end, pmap_page_size, last_level_only, false);
9180 } else {
9181 flush_mmu_tlb_entries_async(va, end, pmap_page_size, last_level_only, false);
9182 }
9183 }
9184
9185 void
9186 flush_mmu_tlb_region(
9187 vm_offset_t va,
9188 unsigned length)
9189 {
9190 flush_mmu_tlb_region_asid_async(va, length, kernel_pmap, true);
9191 sync_tlb_flush();
9192 }
9193
9194 unsigned int
9195 pmap_cache_attributes(
9196 ppnum_t pn)
9197 {
9198 pmap_paddr_t paddr;
9199 unsigned int pai;
9200 unsigned int result;
9201 pp_attr_t pp_attr_current;
9202
9203 paddr = ptoa(pn);
9204
9205 assert(vm_last_phys > vm_first_phys); // Check that pmap has been bootstrapped
9206
9207 if (!pa_valid(paddr)) {
9208 pmap_io_range_t *io_rgn = pmap_find_io_attr(paddr);
9209 return (io_rgn == NULL) ? VM_WIMG_IO : io_rgn->wimg;
9210 }
9211
9212 result = VM_WIMG_DEFAULT;
9213
9214 pai = pa_index(paddr);
9215
9216 pp_attr_current = pp_attr_table[pai];
9217 if (pp_attr_current & PP_ATTR_WIMG_MASK) {
9218 result = pp_attr_current & PP_ATTR_WIMG_MASK;
9219 }
9220 return result;
9221 }
9222
9223 MARK_AS_PMAP_TEXT static void
9224 pmap_sync_wimg(ppnum_t pn, unsigned int wimg_bits_prev, unsigned int wimg_bits_new)
9225 {
9226 if ((wimg_bits_prev != wimg_bits_new)
9227 && ((wimg_bits_prev == VM_WIMG_COPYBACK)
9228 || ((wimg_bits_prev == VM_WIMG_INNERWBACK)
9229 && (wimg_bits_new != VM_WIMG_COPYBACK))
9230 || ((wimg_bits_prev == VM_WIMG_WTHRU)
9231 && ((wimg_bits_new != VM_WIMG_COPYBACK) || (wimg_bits_new != VM_WIMG_INNERWBACK))))) {
9232 pmap_sync_page_attributes_phys(pn);
9233 }
9234
9235 if ((wimg_bits_new == VM_WIMG_RT) && (wimg_bits_prev != VM_WIMG_RT)) {
9236 pmap_force_dcache_clean(phystokv(ptoa(pn)), PAGE_SIZE);
9237 }
9238 }
9239
9240 MARK_AS_PMAP_TEXT __unused void
9241 pmap_update_compressor_page_internal(ppnum_t pn, unsigned int prev_cacheattr, unsigned int new_cacheattr)
9242 {
9243 pmap_paddr_t paddr = ptoa(pn);
9244
9245 if (__improbable(!pa_valid(paddr))) {
9246 panic("%s called on non-managed page 0x%08x", __func__, pn);
9247 }
9248
9249 pmap_set_cache_attributes_internal(pn, new_cacheattr, false);
9250
9251 pmap_sync_wimg(pn, prev_cacheattr & VM_WIMG_MASK, new_cacheattr & VM_WIMG_MASK);
9252 }
9253
9254 void *
9255 pmap_map_compressor_page(ppnum_t pn)
9256 {
9257 unsigned int cacheattr = pmap_cache_attributes(pn) & VM_WIMG_MASK;
9258 if (cacheattr != VM_WIMG_DEFAULT) {
9259 pmap_update_compressor_page_internal(pn, cacheattr, VM_WIMG_DEFAULT);
9260 }
9261
9262 return (void*)phystokv(ptoa(pn));
9263 }
9264
9265 void
9266 pmap_unmap_compressor_page(ppnum_t pn __unused, void *kva __unused)
9267 {
9268 unsigned int cacheattr = pmap_cache_attributes(pn) & VM_WIMG_MASK;
9269 if (cacheattr != VM_WIMG_DEFAULT) {
9270 pmap_update_compressor_page_internal(pn, VM_WIMG_DEFAULT, cacheattr);
9271 }
9272 }
9273
9274 /**
9275 * Flushes TLB entries associated with the page specified by paddr, but do not
9276 * issue barriers yet.
9277 *
9278 * @param paddr The physical address to be flushed from TLB. Must be a managed address.
9279 */
9280 static void
9281 pmap_flush_tlb_for_paddr_async(pmap_paddr_t paddr)
9282 {
9283 /* Flush the physical aperture mappings. */
9284 const vm_offset_t kva = phystokv(paddr);
9285 flush_mmu_tlb_region_asid_async(kva, PAGE_SIZE, kernel_pmap, true);
9286
9287 /* Flush the mappings tracked in the ptes. */
9288 const unsigned int pai = pa_index(paddr);
9289 locked_pvh_t locked_pvh = pvh_lock(pai);
9290
9291 pt_entry_t *pte_p = PT_ENTRY_NULL;
9292 pv_entry_t *pve_p = PV_ENTRY_NULL;
9293
9294 if (pvh_test_type(locked_pvh.pvh, PVH_TYPE_PTEP)) {
9295 pte_p = pvh_ptep(locked_pvh.pvh);
9296 } else if (pvh_test_type(locked_pvh.pvh, PVH_TYPE_PVEP)) {
9297 pve_p = pvh_pve_list(locked_pvh.pvh);
9298 pte_p = PT_ENTRY_NULL;
9299 }
9300
9301 unsigned int nptes = 0;
9302 int pve_ptep_idx = 0;
9303 while ((pve_p != PV_ENTRY_NULL) || (pte_p != PT_ENTRY_NULL)) {
9304 if (pve_p != PV_ENTRY_NULL) {
9305 pte_p = pve_get_ptep(pve_p, pve_ptep_idx);
9306 if (pte_p == PT_ENTRY_NULL) {
9307 goto flush_tlb_skip_pte;
9308 }
9309 }
9310
9311 if (__improbable(nptes == SPTM_MAPPING_LIMIT)) {
9312 pvh_lock_enter_sleep_mode(&locked_pvh);
9313 }
9314 ++nptes;
9315 #ifdef PVH_FLAG_IOMMU
9316 if (pvh_ptep_is_iommu(pte_p)) {
9317 goto flush_tlb_skip_pte;
9318 }
9319 #endif /* PVH_FLAG_IOMMU */
9320 const pmap_t pmap = ptep_get_pmap(pte_p);
9321 const vm_map_address_t va = ptep_get_va(pte_p);
9322
9323 pmap_get_pt_ops(pmap)->flush_tlb_region_async(va, pt_attr_page_size(pmap_get_pt_attr(pmap)) * PAGE_RATIO, pmap, true);
9324
9325 flush_tlb_skip_pte:
9326 pte_p = PT_ENTRY_NULL;
9327 if ((pve_p != PV_ENTRY_NULL) && (++pve_ptep_idx == PTE_PER_PVE)) {
9328 pve_ptep_idx = 0;
9329 pve_p = pve_next(pve_p);
9330 }
9331 }
9332 pvh_unlock(&locked_pvh);
9333 }
9334
9335 /**
9336 * Updates the pp_attr_table entry indexed by pai with cacheattr atomically.
9337 *
9338 * @param pai The Physical Address Index of the entry.
9339 * @param cacheattr The new cache attribute.
9340 */
9341 MARK_AS_PMAP_TEXT static void
9342 pmap_update_pp_attr_wimg_bits_locked(unsigned int pai, unsigned int cacheattr)
9343 {
9344 pvh_assert_locked(pai);
9345
9346 pp_attr_t pp_attr_current, pp_attr_template;
9347 do {
9348 pp_attr_current = pp_attr_table[pai];
9349 pp_attr_template = (pp_attr_current & ~PP_ATTR_WIMG_MASK) | PP_ATTR_WIMG(cacheattr);
9350
9351 /**
9352 * WIMG bits should only be updated under the PVH lock, but we should do
9353 * this in a CAS loop to avoid losing simultaneous updates to other bits like refmod.
9354 */
9355 } while (!OSCompareAndSwap16(pp_attr_current, pp_attr_template, &pp_attr_table[pai]));
9356 }
9357
9358 /**
9359 * Structure for tracking where we are during the collection of mappings for batch
9360 * cache attribute updates.
9361 *
9362 * @note We need to track where in the per-cpu ops table we are filling the next mappings into,
9363 * because the collection routine can return with a not completely filled ops table when
9364 * it exhausts the PV list for a page. In such case, the remaining slots in the ops table
9365 * will be used for mappings of the next page.
9366 *
9367 * @note We also need to record where we are in the PV list, because the collection routine can
9368 * also return when the ops table is filled but it's still in the middle of the PV list.
9369 * Those remaining items in the PV list need to be handled by the next batch operation in
9370 * a new ops table.
9371 */
9372 typedef struct {
9373 /* Where we are in the sptm ops table. */
9374 unsigned int sptm_ops_index;
9375
9376 /**
9377 * The last collected physical address from the previous full ops array (and in turn, SPTM
9378 * call). This is used to know whether the SPTM call for the latest full ops table should
9379 * skip updating the PAPT mapping (seeing as the last call would have handled updating it).
9380 */
9381 pmap_paddr_t last_table_last_papt_pa;
9382
9383 /**
9384 * Where we are in the pv list.
9385 *
9386 * When ptep is non-null, there's only one mapping to the page and the ptep is the address
9387 * of it.
9388 *
9389 * When pvep is non-null, there's more than one mapping and the mappings are tracked by the
9390 * PV list.
9391 *
9392 * When they are both null, it indicates we are collecting for a new page and the collection
9393 * function will initialize them to be one of the two states above.
9394 *
9395 * It is undefined when they are both non-null.
9396 */
9397 pt_entry_t *ptep;
9398 pv_entry_t *pvep;
9399 unsigned int pve_ptep_idx;
9400 } pmap_sptm_update_cache_attr_ops_collect_state_t;
9401
9402 /**
9403 * Reports whether there is any pending ops in an sptm cache attr ops table.
9404 *
9405 * @param state A pmap_sptm_update_cache_attr_ops_collect_state_t structure.
9406 *
9407 * @return True if there's any outstanding cache attr op.
9408 * False otherwise.
9409 */
9410 static inline bool
9411 pmap_is_sptm_update_cache_attr_ops_pending(pmap_sptm_update_cache_attr_ops_collect_state_t state)
9412 {
9413 return state.sptm_ops_index > 0;
9414 }
9415
9416 /**
9417 * Struct for encoding the collection status into pmap_sptm_update_cache_attr_ops_collect()'s
9418 * return value indicating what kind of attention it needs.
9419 */
9420 typedef enum {
9421 OPS_COLLECT_NOTHING = 0x0,
9422
9423 /* The ops table is full, and the caller should commit the table to SPTM. */
9424 OPS_COLLECT_RETURN_FULL_TABLE = 0x1,
9425
9426 /**
9427 * The page has its mappings completely collected, and the caller should
9428 * pass in a new page next time.
9429 */
9430 OPS_COLLECT_RETURN_COMPLETED_PAGE = 0x2,
9431 } pmap_sptm_update_cache_attr_ops_collect_return_t;
9432
9433 /**
9434 * Collects mappings of a physical page into an SPTM ops table for cache attribute updates.
9435 *
9436 * @note This routine returns either when the ops table is full or the page represented by
9437 * pa has no more mapping to collect. The caller should call this routine again with
9438 * a fresh ops table, or a new page, or both, depending on the return code.
9439 *
9440 * @note The PVH lock needs to be held for pa.
9441 *
9442 * @param state Tracks the state of PV list traversal and SPTM ops table filling. It is used
9443 * by this routine to save the progress of the collection.
9444 * @param sptm_ops Pointer to the SPTM ops table.
9445 * @param pa The physical address whose mappings are to be collected.
9446 * @param attributes The new cache attributes.
9447 *
9448 * @return A pmap_sptm_update_cache_attr_ops_collect_return_t that encodes what the caller
9449 * should do before calling this routine again. See the inline comments around
9450 * pmap_sptm_update_cache_attr_ops_collect_return_t for details.
9451 */
9452 static pmap_sptm_update_cache_attr_ops_collect_return_t
9453 pmap_sptm_update_cache_attr_ops_collect(
9454 pmap_sptm_update_cache_attr_ops_collect_state_t *state,
9455 sptm_update_disjoint_multipage_op_t *sptm_ops,
9456 pmap_paddr_t pa,
9457 unsigned int attributes)
9458 {
9459 if (state == NULL || sptm_ops == NULL) {
9460 panic("%s: unexpected null arguments - state: %p, sptm_ops: %p", __func__, state, sptm_ops);
9461 }
9462
9463 PMAP_TRACE(2, PMAP_CODE(PMAP__COLLECT_CACHE_OPS) | DBG_FUNC_START, pa, attributes, state->sptm_ops_index);
9464
9465 /* Copy the states into local variables. */
9466 unsigned int sptm_ops_index = state->sptm_ops_index;
9467 pmap_paddr_t last_table_last_papt_pa = state->last_table_last_papt_pa;
9468 pv_entry_t *pvep = state->pvep;
9469 pt_entry_t *ptep = state->ptep;
9470 unsigned int pve_ptep_idx = state->pve_ptep_idx;
9471
9472 unsigned int pai = pa_index(pa);
9473
9474 /* We should at least have one free slot in the ops table. */
9475 assert(sptm_ops_index < SPTM_MAPPING_LIMIT);
9476
9477 /* The PVH lock for pa has to be locked. */
9478 pvh_assert_locked(pai);
9479
9480 /* If pvep and ptep are both null in the state, it's a new page. Initialize the states. */
9481 if (pvep == PV_ENTRY_NULL && ptep == PT_ENTRY_NULL) {
9482 const uintptr_t pvh = pai_to_pvh(pai);
9483 if (pvh_test_type(pvh, PVH_TYPE_PVEP)) {
9484 ptep = PT_ENTRY_NULL;
9485 pvep = pvh_pve_list(pvh);
9486 pve_ptep_idx = 0;
9487 } else if (pvh_test_type(pvh, PVH_TYPE_PTEP)) {
9488 ptep = pvh_ptep(pvh);
9489 pvep = PV_ENTRY_NULL;
9490 pve_ptep_idx = 0;
9491 }
9492 }
9493
9494 /**
9495 * The first entry filled in is always the PAPT header entry:
9496 *
9497 * 1) In the case of a fresh ops table, the first entry has to be a PAPT header.
9498 * 2) In the case of a fresh page, we need to insert a new PAPT header to request
9499 * SPTM to operate on a new page.
9500 *
9501 * Remember the index of the PAPT header here so that we can update the number
9502 * of mappings field later when we finish collecting.
9503 */
9504 const unsigned int papt_sptm_ops_index = sptm_ops_index;
9505 unsigned int num_mappings = 0;
9506
9507 /* Assemble the PTE template for the PAPT mapping. */
9508 const vm_address_t kva = phystokv(pa);
9509 const pt_entry_t *papt_ptep = pmap_pte(kernel_pmap, kva);
9510
9511 pt_entry_t template = os_atomic_load(papt_ptep, relaxed);
9512 template &= ~(ARM_PTE_ATTRINDXMASK | ARM_PTE_SHMASK);
9513 template |= wimg_to_pte(attributes, pa);
9514
9515 /* Fill in the PAPT header entry. */
9516 sptm_ops[papt_sptm_ops_index].per_paddr_header.paddr = pa;
9517 sptm_ops[papt_sptm_ops_index].per_paddr_header.papt_pte_template = template;
9518 sptm_ops[papt_sptm_ops_index].per_paddr_header.options = SPTM_UPDATE_SH | SPTM_UPDATE_MAIR | SPTM_UPDATE_DEFER_TLBI;
9519
9520 if ((papt_sptm_ops_index == 0) && (pa == last_table_last_papt_pa)) {
9521 /**
9522 * If the previous SPTM call was made with an ops table that already included
9523 * updating the PA of the page that this table starts with, then we can assume
9524 * that call already updated the PAPT and we can safely skip it in this
9525 * upcoming one.
9526 */
9527 sptm_ops[0].per_paddr_header.options |= SPTM_UPDATE_SKIP_PAPT;
9528 }
9529
9530 sptm_ops_index++;
9531
9532 /**
9533 * Main loop for collecting the mappings into the ops table. It terminates either
9534 * when the ops table is full or the PV list is exhausted.
9535 */
9536 while ((sptm_ops_index < SPTM_MAPPING_LIMIT) && (pvep != PV_ENTRY_NULL || ptep != PT_ENTRY_NULL)) {
9537 /**
9538 * Update ptep. There are really two cases here:
9539 *
9540 * 1) pvep is PV_ENTRY_NULL. In this case, ptep holds the pointer to
9541 * the only mapping to the page.
9542 * 2) pvep is not PV_ENTRY_NULL. In such case, ptep is updated accroding to
9543 * pvep and pve_ptep_idx.
9544 */
9545 if (pvep != PV_ENTRY_NULL) {
9546 ptep = pve_get_ptep(pvep, pve_ptep_idx);
9547
9548 /* This pve is empty, so skip to next one. */
9549 if (ptep == PT_ENTRY_NULL) {
9550 goto sucaoc_skip_pte;
9551 }
9552 }
9553
9554 #ifdef PVH_FLAG_IOMMU
9555 /* Skip IOMMU pteps. */
9556 if (pvh_ptep_is_iommu(ptep)) {
9557 goto sucaoc_skip_pte;
9558 }
9559 #endif
9560 /* Assemble the PTE template for the mapping. */
9561 const vm_address_t va = ptep_get_va(ptep);
9562 const pmap_t pmap = ptep_get_pmap(ptep);
9563
9564 template = os_atomic_load(ptep, relaxed);
9565 template &= ~(ARM_PTE_ATTRINDXMASK | ARM_PTE_SHMASK);
9566 template |= pmap_get_pt_ops(pmap)->wimg_to_pte(attributes, pa);
9567
9568 /* Fill into the ops table. */
9569 sptm_ops[sptm_ops_index].disjoint_op.root_pt_paddr = pmap->ttep;
9570 sptm_ops[sptm_ops_index].disjoint_op.vaddr = va;
9571 sptm_ops[sptm_ops_index].disjoint_op.pte_template = template;
9572
9573 /* Move the sptm ops table cursor. */
9574 sptm_ops_index++;
9575
9576 /* Increment the mappings counter. */
9577 num_mappings++;
9578
9579 sucaoc_skip_pte:
9580 /**
9581 * Reset ptep to PT_ENTRY_NULL to keep the loop precondition of either ptep
9582 * or pvep is nonnull (not both, not neither) true.
9583 */
9584 ptep = PT_ENTRY_NULL;
9585
9586 /* Advance to next pvep if we have exhausted the pteps in it. */
9587 if ((pvep != PV_ENTRY_NULL) && (++pve_ptep_idx == PTE_PER_PVE)) {
9588 pve_ptep_idx = 0;
9589 pvep = pve_next(pvep);
9590 }
9591 }
9592
9593 /* Update the PAPT header for the number of mappings. */
9594 sptm_ops[papt_sptm_ops_index].per_paddr_header.num_mappings = num_mappings;
9595
9596 const bool full_table = (sptm_ops_index >= SPTM_MAPPING_LIMIT);
9597 const bool collection_done_for_page = (pvep == PV_ENTRY_NULL && ptep == PT_ENTRY_NULL);
9598
9599 /**
9600 * The ops table is full, so the caller should now invoke the SPTM before calling
9601 * into this function again.
9602 */
9603 if (full_table) {
9604 /* Update last_table_last_papt_pa to be the pa collected in this call. */
9605 last_table_last_papt_pa = pa;
9606
9607 /* Reset sptm_ops_index. */
9608 sptm_ops_index = 0;
9609 }
9610
9611 /* Copy the updated collection states back to the parameter structure. */
9612 state->sptm_ops_index = sptm_ops_index;
9613 state->last_table_last_papt_pa = last_table_last_papt_pa;
9614 state->pvep = pvep;
9615 state->ptep = ptep;
9616 state->pve_ptep_idx = pve_ptep_idx;
9617
9618 /* Assemble the return value. */
9619 pmap_sptm_update_cache_attr_ops_collect_return_t retval = OPS_COLLECT_NOTHING;
9620
9621 if (full_table) {
9622 retval |= OPS_COLLECT_RETURN_FULL_TABLE;
9623 }
9624
9625 if (collection_done_for_page) {
9626 retval |= OPS_COLLECT_RETURN_COMPLETED_PAGE;
9627 }
9628
9629 PMAP_TRACE(2, PMAP_CODE(PMAP__COLLECT_CACHE_OPS) | DBG_FUNC_END, pa, attributes, sptm_ops_index);
9630
9631 return retval;
9632 }
9633
9634 /* At least one PAPT header plus one mapping. */
9635 static_assert(SPTM_MAPPING_LIMIT >= 2);
9636
9637 /**
9638 * Returns if a cache attribute is allowed (on managed pages).
9639 *
9640 * @param attributes A 32-bit value whose VM_WIMG_MASK bits represent the
9641 * cache attribute.
9642 *
9643 * @return True if the cache attribute is allowed on managed pages.
9644 * False otherwise.
9645 */
9646 static bool
9647 pmap_is_cache_attribute_allowed(unsigned int attributes)
9648 {
9649 if (pmap_panic_dev_wimg_on_managed) {
9650 switch (attributes & VM_WIMG_MASK) {
9651 /* supported on DRAM, but slow, so we disallow */
9652 case VM_WIMG_IO: // nGnRnE
9653 case VM_WIMG_POSTED: // nGnRE
9654
9655 /* unsupported on DRAM */
9656 case VM_WIMG_POSTED_REORDERED: // nGRE
9657 case VM_WIMG_POSTED_COMBINED_REORDERED: // GRE
9658 return false;
9659
9660 default:
9661 return true;
9662 }
9663 }
9664
9665 return true;
9666 }
9667
9668 /**
9669 * Batch updates the cache attributes of a list of pages in three passes.
9670 *
9671 * In pass one, the pp_attr_table and the pte are updated (by SPTM) for the pages in the list.
9672 * In pass two, TLB entries are flushed for each page in the list if necessary.
9673 * In pass three, caches are cleaned for each page in the list if necessary.
9674 *
9675 * @param page_list List of pages to be updated.
9676 * @param cacheattr The new cache attributes.
9677 * @param update_attr_table Whether the pp_attr_table should be updated. This is useful for compressor
9678 * pages where it's desired to keep the old WIMG bits.
9679 */
9680 void
9681 pmap_batch_set_cache_attributes_internal(
9682 const unified_page_list_t *page_list,
9683 unsigned int cacheattr,
9684 bool update_attr_table)
9685 {
9686 bool tlb_flush_pass_needed = false;
9687 bool rt_cache_flush_pass_needed = false;
9688 bool preemption_disabled = false;
9689
9690 PMAP_TRACE(2, PMAP_CODE(PMAP__BATCH_UPDATE_CACHING), page_list, cacheattr, 0xCECC0DE1);
9691
9692 pmap_sptm_percpu_data_t *sptm_pcpu = NULL;
9693 sptm_update_disjoint_multipage_op_t *sptm_ops = NULL;
9694
9695 pmap_sptm_update_cache_attr_ops_collect_state_t state = {0};
9696
9697 unified_page_list_iterator_t iter;
9698
9699 for (unified_page_list_iterator_init(page_list, &iter);
9700 !unified_page_list_iterator_end(&iter);
9701 unified_page_list_iterator_next(&iter)) {
9702 bool is_fictitious = false;
9703 const ppnum_t pn = unified_page_list_iterator_page(&iter, &is_fictitious);
9704 const pmap_paddr_t paddr = ptoa(pn);
9705
9706 /**
9707 * Skip if the page is not managed.
9708 *
9709 * We don't panic here because sometimes the user just blindly pass in
9710 * pages that are not managed. We need to handle that gracefully.
9711 */
9712 if (__improbable(!pa_valid(paddr) || is_fictitious)) {
9713 continue;
9714 }
9715
9716 const unsigned int pai = pa_index(paddr);
9717 locked_pvh_t locked_pvh = {.pvh = 0};
9718
9719 if (pmap_is_sptm_update_cache_attr_ops_pending(state)) {
9720 /**
9721 * If we're partway through processing a multi-page batched call,
9722 * preemption will already be disabled so we can't simply call
9723 * pvh_lock() which may block. Instead, we first try to acquire
9724 * the lock without waiting, which in most cases should succeed.
9725 * If it fails, we submit the pending batched operations to re-
9726 * enable preemption and then acquire the lock normally.
9727 */
9728 locked_pvh = pvh_try_lock(pai);
9729 if (__improbable(!pvh_try_lock_success(&locked_pvh))) {
9730 assert(preemption_disabled);
9731 const sptm_return_t sptm_ret = sptm_update_disjoint_multipage(sptm_pcpu->sptm_ops_pa, state.sptm_ops_index);
9732 pmap_retype_epoch_exit();
9733 enable_preemption();
9734 preemption_disabled = false;
9735 if (sptm_ret == SPTM_UPDATE_DELAYED_TLBI) {
9736 tlb_flush_pass_needed = true;
9737 }
9738 state.sptm_ops_index = 0;
9739 locked_pvh = pvh_lock(pai);
9740 }
9741 } else {
9742 locked_pvh = pvh_lock(pai);
9743 }
9744 assert(locked_pvh.pvh != 0);
9745
9746 const pp_attr_t pp_attr_current = pp_attr_table[pai];
9747
9748 unsigned int wimg_bits_prev = VM_WIMG_DEFAULT;
9749 if (pp_attr_current & PP_ATTR_WIMG_MASK) {
9750 wimg_bits_prev = pp_attr_current & PP_ATTR_WIMG_MASK;
9751 }
9752
9753 const pp_attr_t pp_attr_template = (pp_attr_current & ~PP_ATTR_WIMG_MASK) | PP_ATTR_WIMG(cacheattr);
9754
9755 unsigned int wimg_bits_new = VM_WIMG_DEFAULT;
9756 if (pp_attr_template & PP_ATTR_WIMG_MASK) {
9757 wimg_bits_new = pp_attr_template & PP_ATTR_WIMG_MASK;
9758 }
9759
9760 /**
9761 * When update_attr_table is false, we know that wimg_bits_prev read from pp_attr_table is not to be trusted,
9762 * and we should force update the cache attribute.
9763 */
9764 const bool force_update = !update_attr_table;
9765 /* Update the cache attributes in PTE and PP_ATTR table. */
9766 if ((wimg_bits_new != wimg_bits_prev) || force_update) {
9767 if (!pmap_is_cache_attribute_allowed(cacheattr)) {
9768 panic("%s: trying to use unsupported VM_WIMG type for managed page, VM_WIMG=%x, pn=%#x",
9769 __func__, cacheattr & VM_WIMG_MASK, pn);
9770 }
9771
9772 /* Update PP_ATTR_TABLE */
9773 if (update_attr_table) {
9774 pmap_update_pp_attr_wimg_bits_locked(pai, cacheattr);
9775 }
9776
9777 bool mapping_collection_done = false;
9778 bool pvh_lock_sleep_mode_needed = false;
9779 do {
9780 if (__improbable(pvh_lock_sleep_mode_needed)) {
9781 assert(!preemption_disabled);
9782 pvh_lock_enter_sleep_mode(&locked_pvh);
9783 pvh_lock_sleep_mode_needed = false;
9784 }
9785
9786 /* Disable preemption to use the per-CPU structure safely. */
9787 if (!preemption_disabled) {
9788 preemption_disabled = true;
9789 disable_preemption();
9790 /**
9791 * Enter the retype epoch while we gather the disjoint update arguments
9792 * and issue the SPTM call. Since this operation may cover multiple physical
9793 * pages, we may construct the argument array and invoke the SPTM without holding
9794 * all relevant PVH locks, we need to record that we are collecting and modifying
9795 * mapping state so that e.g. pmap_page_protect() does not attempt to retype the
9796 * underlying pages and pmap_remove() does not attempt to free the page tables
9797 * used for these mappings without first draining our epoch.
9798 */
9799 pmap_retype_epoch_enter();
9800
9801 sptm_pcpu = PERCPU_GET(pmap_sptm_percpu);
9802 sptm_ops = (sptm_update_disjoint_multipage_op_t *) sptm_pcpu->sptm_ops;
9803 }
9804
9805 /* The return value indicates if we should call into SPTM in this iteration. */
9806 pmap_sptm_update_cache_attr_ops_collect_return_t retval =
9807 pmap_sptm_update_cache_attr_ops_collect(&state, sptm_ops, paddr, cacheattr);
9808
9809 /* The collection routine should only return if it needs attention. */
9810 assert(retval != OPS_COLLECT_NOTHING);
9811
9812 /* Gather information for next step from the return value. */
9813 mapping_collection_done = retval & OPS_COLLECT_RETURN_COMPLETED_PAGE;
9814 const bool call_sptm = retval & OPS_COLLECT_RETURN_FULL_TABLE;
9815
9816 if (call_sptm) {
9817 /* Call into SPTM with this SPTM ops table. */
9818 sptm_return_t sptm_ret = sptm_update_disjoint_multipage(sptm_pcpu->sptm_ops_pa, SPTM_MAPPING_LIMIT);
9819 /**
9820 * We may be submitting the batch and exiting the epoch partway through
9821 * processing the PV list for a page. That's fine, because in that case we'll
9822 * hold the PV lock for that page, which will prevent mappings of that page from
9823 * being disconnected and will prevent the completion of pmap_remove() against
9824 * any of those mappings, thus also guaranteeing the relevant page table pages
9825 * can't be freed. The epoch still protects mappings for any prior page in
9826 * the batch, whose PV locks are no longer held.
9827 */
9828 pmap_retype_epoch_exit();
9829 /**
9830 * Balance out the explicit disable_preemption() made either at the beginning of
9831 * the function or on a prior iteration of the loop that placed the PVH lock in
9832 * sleep mode. Note that enable_preemption() decrements a per-thread counter,
9833 * so if we still happen to hold the PVH lock in spin mode preemption won't
9834 * actually be re-enabled until we switch the lock over to sleep mode on
9835 * the next iteration.
9836 */
9837 enable_preemption();
9838 preemption_disabled = false;
9839 pvh_lock_sleep_mode_needed = true;
9840
9841 if (sptm_ret == SPTM_UPDATE_DELAYED_TLBI) {
9842 tlb_flush_pass_needed = true;
9843 }
9844 }
9845
9846 /* We cannot be in a situation where we didn't call into SPTM while also having not finished walking the pv list. */
9847 assert(call_sptm || mapping_collection_done);
9848 } while (!mapping_collection_done);
9849
9850 /**
9851 * We could technically force the cache flush pass here when force_update is true, but
9852 * since the compressor mapping/unmapping path handles cache flushing itself, it's fine
9853 * leaving this as is.
9854 */
9855 if (wimg_bits_new == VM_WIMG_RT && wimg_bits_prev != VM_WIMG_RT) {
9856 rt_cache_flush_pass_needed = true;
9857 }
9858 }
9859
9860 pvh_unlock(&locked_pvh);
9861 }
9862
9863 if (pmap_is_sptm_update_cache_attr_ops_pending(state)) {
9864 assert(preemption_disabled);
9865 sptm_return_t sptm_ret = sptm_update_disjoint_multipage(sptm_pcpu->sptm_ops_pa, state.sptm_ops_index);
9866 pmap_retype_epoch_exit();
9867 if (sptm_ret == SPTM_UPDATE_DELAYED_TLBI) {
9868 tlb_flush_pass_needed = true;
9869 }
9870
9871 /**
9872 * This is the last sptm_update_cache_attr() call whatsoever, so it's
9873 * okay not to update the state variables.
9874 */
9875
9876 enable_preemption();
9877 } else if (preemption_disabled) {
9878 pmap_retype_epoch_exit();
9879 enable_preemption();
9880 }
9881
9882 if (tlb_flush_pass_needed) {
9883 /* Sync the PTE writes before potential TLB/Cache flushes. */
9884 FLUSH_PTE_STRONG();
9885
9886 /**
9887 * Pass 2: for each physical page and for each mapping, we need to flush
9888 * the TLB for it.
9889 */
9890 PMAP_TRACE(2, PMAP_CODE(PMAP__BATCH_UPDATE_CACHING), page_list, cacheattr, 0xCECC0DE2);
9891 for (unified_page_list_iterator_init(page_list, &iter);
9892 !unified_page_list_iterator_end(&iter);
9893 unified_page_list_iterator_next(&iter)) {
9894 bool is_fictitious = false;
9895 const ppnum_t pn = unified_page_list_iterator_page(&iter, &is_fictitious);
9896 const pmap_paddr_t paddr = ptoa(pn);
9897
9898 if (__improbable(!pa_valid(paddr) || is_fictitious)) {
9899 continue;
9900 }
9901
9902 pmap_flush_tlb_for_paddr_async(paddr);
9903 }
9904
9905 #if HAS_FEAT_XS
9906 /* With FEAT_XS, ordinary DSBs drain the prefetcher. */
9907 arm64_sync_tlb(false);
9908 #else
9909 /**
9910 * For targets that distinguish between mild and strong DSB, mild DSB
9911 * will not drain the prefetcher. This can lead to prefetch-driven
9912 * cache fills that defeat the uncacheable requirement of the RT memory type.
9913 * In those cases, strong DSB must instead be employed to drain the prefetcher.
9914 */
9915 arm64_sync_tlb((cacheattr & VM_WIMG_MASK) == VM_WIMG_RT);
9916 #endif
9917 }
9918
9919 if (rt_cache_flush_pass_needed) {
9920 /* Pass 3: Flush the cache if the page is recently set to RT */
9921 PMAP_TRACE(2, PMAP_CODE(PMAP__BATCH_UPDATE_CACHING), page_list, cacheattr, 0xCECC0DE3);
9922 /**
9923 * We disable preemption to ensure we are not preempted
9924 * in the state where DC by VA instructions remain enabled.
9925 */
9926 disable_preemption();
9927
9928 assert(get_preemption_level() > 0);
9929
9930 #if defined(APPLE_ARM64_ARCH_FAMILY) && !APPLEVIRTUALPLATFORM
9931 /**
9932 * On APPLEVIRTUALPLATFORM, HID register accesses cause a synchronous exception
9933 * and the host will handle cache maintenance for it. So we don't need to
9934 * worry about enabling the ops here for AVP.
9935 */
9936 enable_dc_mva_ops();
9937 #endif /* defined(APPLE_ARM64_ARCH_FAMILY) && !APPLEVIRTUALPLATFORM */
9938 /**
9939 * DMB should be sufficient to ensure prior accesses to the memory in question are
9940 * correctly ordered relative to the upcoming cache maintenance operations.
9941 */
9942 __builtin_arm_dmb(DMB_SY);
9943
9944 for (unified_page_list_iterator_init(page_list, &iter);
9945 !unified_page_list_iterator_end(&iter);) {
9946 bool is_fictitious = false;
9947 const ppnum_t pn = unified_page_list_iterator_page(&iter, &is_fictitious);
9948 const pmap_paddr_t paddr = ptoa(pn);
9949
9950 if (__improbable(!pa_valid(paddr) || is_fictitious)) {
9951 unified_page_list_iterator_next(&iter);
9952 continue;
9953 }
9954
9955 CleanPoC_DcacheRegion_Force_nopreempt_nohid_nobarrier(phystokv(paddr), PAGE_SIZE);
9956
9957 unified_page_list_iterator_next(&iter);
9958 if (__improbable(pmap_pending_preemption() && !unified_page_list_iterator_end(&iter))) {
9959 __builtin_arm_dsb(DSB_SY);
9960 #if defined(APPLE_ARM64_ARCH_FAMILY) && !APPLEVIRTUALPLATFORM
9961 disable_dc_mva_ops();
9962 #endif /* defined(APPLE_ARM64_ARCH_FAMILY) && !APPLEVIRTUALPLATFORM */
9963 enable_preemption();
9964 assert(preemption_enabled());
9965 disable_preemption();
9966 #if defined(APPLE_ARM64_ARCH_FAMILY) && !APPLEVIRTUALPLATFORM
9967 enable_dc_mva_ops();
9968 #endif /* defined(APPLE_ARM64_ARCH_FAMILY) && !APPLEVIRTUALPLATFORM */
9969 }
9970 }
9971
9972 /* Issue DSB to ensure cache maintenance is fully complete before subsequent accesses. */
9973 __builtin_arm_dsb(DSB_SY);
9974 #if defined(APPLE_ARM64_ARCH_FAMILY) && !APPLEVIRTUALPLATFORM
9975 disable_dc_mva_ops();
9976 #endif /* defined(APPLE_ARM64_ARCH_FAMILY) && !APPLEVIRTUALPLATFORM */
9977
9978 enable_preemption();
9979 }
9980
9981 PMAP_TRACE(2, PMAP_CODE(PMAP__BATCH_UPDATE_CACHING), page_list, cacheattr, 0xCECC0DE4);
9982 }
9983
9984 /**
9985 * Batch updates the cache attributes of a list of pages. This is a wrapper for
9986 * the ppl call on PPL-enabled platforms or the _internal helper on other platforms.
9987 *
9988 * @param page_list List of pages to be updated.
9989 * @param cacheattr The new cache attribute.
9990 */
9991 void
9992 pmap_batch_set_cache_attributes(
9993 const unified_page_list_t *page_list,
9994 unsigned int cacheattr)
9995 {
9996 PMAP_TRACE(2, PMAP_CODE(PMAP__BATCH_UPDATE_CACHING) | DBG_FUNC_START, page_list, cacheattr, 0xCECC0DE0);
9997
9998 /* Verify we are being called from a preemptible context. */
9999 pmap_verify_preemptible();
10000
10001 pmap_batch_set_cache_attributes_internal(page_list, cacheattr, true);
10002
10003 PMAP_TRACE(2, PMAP_CODE(PMAP__BATCH_UPDATE_CACHING) | DBG_FUNC_END, page_list, cacheattr, 0xCECC0DEF);
10004 }
10005
10006 MARK_AS_PMAP_TEXT void
10007 pmap_set_cache_attributes_internal(
10008 ppnum_t pn,
10009 unsigned int cacheattr,
10010 bool update_attr_table)
10011 {
10012 upl_page_info_t single_page_upl = { .phys_addr = pn };
10013 const unified_page_list_t page_list = {
10014 .upl = {.upl_info = &single_page_upl, .upl_size = 1},
10015 .type = UNIFIED_PAGE_LIST_TYPE_UPL_ARRAY,
10016 };
10017
10018 pmap_batch_set_cache_attributes_internal(&page_list, cacheattr, update_attr_table);
10019 }
10020
10021 void
10022 pmap_set_cache_attributes(
10023 ppnum_t pn,
10024 unsigned int cacheattr)
10025 {
10026 pmap_set_cache_attributes_internal(pn, cacheattr, true);
10027 }
10028
10029 void
10030 pmap_create_commpages(vm_map_address_t *kernel_data_addr, vm_map_address_t *kernel_text_addr,
10031 vm_map_address_t *kernel_ro_data_addr, vm_map_address_t *user_text_addr)
10032 {
10033 pmap_paddr_t data_pa = 0; // data address
10034 pmap_paddr_t ro_data_pa = 0; // kernel read-only data address
10035 pmap_paddr_t text_pa = 0; // text address
10036
10037 *kernel_data_addr = 0;
10038 *kernel_text_addr = 0;
10039 *user_text_addr = 0;
10040
10041 kern_return_t kr = pmap_page_alloc(&data_pa, PMAP_PAGE_ALLOCATE_NONE);
10042 assert(kr == KERN_SUCCESS);
10043
10044 kr = pmap_page_alloc(&ro_data_pa, PMAP_PAGE_ALLOCATE_NONE);
10045 assert(kr == KERN_SUCCESS);
10046
10047 #if CONFIG_ARM_PFZ
10048 kr = pmap_page_alloc(&text_pa, PMAP_PAGE_ALLOCATE_NONE);
10049 assert(kr == KERN_SUCCESS);
10050
10051 /**
10052 * User mapping of comm page text section for 64 bit mapping only
10053 *
10054 * We don't insert it into the 32 bit mapping because we don't want 32 bit
10055 * user processes to get this page mapped in, they should never call into
10056 * this page.
10057 *
10058 * The data comm page is in a pre-reserved L3 VA range and the text commpage
10059 * is slid in the same L3 as the data commpage. It is either outside the
10060 * max of user VA or is pre-reserved in vm_map_exec(). This means that
10061 * it is reserved and unavailable to mach VM for future mappings.
10062 */
10063 const int num_ptes = pt_attr_leaf_size(native_pt_attr) >> PTE_SHIFT;
10064
10065 do {
10066 const int text_leaf_index = random() % num_ptes;
10067
10068 /**
10069 * Generate a VA for the commpage text with the same root and twig index as data
10070 * comm page, but with new leaf index we've just generated.
10071 */
10072 commpage_text_user_va = (_COMM_PAGE64_BASE_ADDRESS & ~pt_attr_leaf_index_mask(native_pt_attr));
10073 commpage_text_user_va |= (text_leaf_index << pt_attr_leaf_shift(native_pt_attr));
10074 } while ((commpage_text_user_va == _COMM_PAGE64_BASE_ADDRESS) ||
10075 (commpage_text_user_va == _COMM_PAGE64_RO_ADDRESS)); // Try again if we collide (should be unlikely)
10076
10077 *user_text_addr = commpage_text_user_va;
10078 *kernel_text_addr = phystokv(text_pa);
10079 #endif
10080
10081 /* For manipulation in kernel, go straight to physical page */
10082 commpage_data_pa = data_pa;
10083 *kernel_data_addr = phystokv(data_pa);
10084 assert(commpage_ro_data_pa == 0);
10085 commpage_ro_data_pa = ro_data_pa;
10086 *kernel_ro_data_addr = phystokv(ro_data_pa);
10087 assert(commpage_text_pa == 0);
10088 commpage_text_pa = text_pa;
10089 }
10090
10091
10092 /*
10093 * Asserts to ensure that the TTEs we nest to map the shared page do not overlap
10094 * with user controlled TTEs for regions that aren't explicitly reserved by the
10095 * VM (e.g., _COMM_PAGE64_NESTING_START/_COMM_PAGE64_BASE_ADDRESS).
10096 */
10097 #if (ARM_PGSHIFT == 14)
10098 /**
10099 * Ensure that 64-bit devices with 32-bit userspace VAs (arm64_32) can nest the
10100 * commpage completely above the maximum 32-bit userspace VA.
10101 */
10102 static_assert((_COMM_PAGE32_BASE_ADDRESS & ~ARM_TT_L2_OFFMASK) >= VM_MAX_ADDRESS);
10103 static_assert(_COMM_PAGE64_NESTING_START == SPTM_ARM64_COMMPAGE_REGION_START);
10104 static_assert(_COMM_PAGE64_NESTING_SIZE == SPTM_ARM64_COMMPAGE_REGION_SIZE);
10105
10106 /**
10107 * Normally there'd be an assert to check that 64-bit devices with 64-bit
10108 * userspace VAs can nest the commpage completely above the maximum 64-bit
10109 * userpace VA, but that technically isn't true on macOS. On those systems, the
10110 * commpage lives within the userspace VA range, but is protected by the VM as
10111 * a reserved region (see vm_reserved_regions[] definition for more info).
10112 */
10113
10114 #elif (ARM_PGSHIFT == 12)
10115 /**
10116 * Ensure that 64-bit devices using 4K pages can nest the commpage completely
10117 * above the maximum userspace VA.
10118 */
10119 static_assert((_COMM_PAGE64_BASE_ADDRESS & ~ARM_TT_L1_OFFMASK) >= MACH_VM_MAX_ADDRESS);
10120 #else
10121 #error Nested shared page mapping is unsupported on this config
10122 #endif
10123
10124 MARK_AS_PMAP_TEXT kern_return_t
10125 pmap_insert_commpage_internal(
10126 pmap_t pmap)
10127 {
10128 kern_return_t kr = KERN_SUCCESS;
10129 vm_offset_t commpage_vaddr;
10130 pt_entry_t *ttep;
10131 pmap_paddr_t commpage_table = commpage_default_table;
10132
10133 /* Validate the pmap input before accessing its data. */
10134 validate_pmap_mutable(pmap);
10135
10136 const pt_attr_t * const pt_attr = pmap_get_pt_attr(pmap);
10137 const unsigned int commpage_level = pt_attr_commpage_level(pt_attr);
10138
10139 #if __ARM_MIXED_PAGE_SIZE__
10140 #if !__ARM_16K_PG__
10141 /* The following code assumes that commpage_pmap_default is a 16KB pmap. */
10142 #error "pmap_insert_commpage_internal requires a 16KB default kernel page size when __ARM_MIXED_PAGE_SIZE__ is enabled"
10143 #endif /* !__ARM_16K_PG__ */
10144
10145 /* Choose the correct shared page pmap to use. */
10146 const uint64_t pmap_page_size = pt_attr_page_size(pt_attr);
10147 if (pmap_page_size == 4096) {
10148 if (pmap_is_64bit(pmap)) {
10149 commpage_table = commpage_4k_table;
10150 } else {
10151 panic("32-bit commpage not currently supported for SPTM configurations");
10152 //commpage_table = commpage32_4k_table;
10153 }
10154 } else if (pmap_page_size != 16384) {
10155 panic("No commpage table exists for the wanted page size: %llu", pmap_page_size);
10156 } else
10157 #endif /* __ARM_MIXED_PAGE_SIZE__ */
10158 {
10159 if (pmap_is_64bit(pmap)) {
10160 commpage_table = commpage_default_table;
10161 } else {
10162 panic("32-bit commpage not currently supported for SPTM configurations");
10163 //commpage_table = commpage32_default_table;
10164 }
10165 }
10166
10167 #if _COMM_PAGE_AREA_LENGTH != PAGE_SIZE
10168 #error We assume a single page.
10169 #endif
10170
10171 if (pmap_is_64bit(pmap)) {
10172 commpage_vaddr = _COMM_PAGE64_BASE_ADDRESS;
10173 } else {
10174 commpage_vaddr = _COMM_PAGE32_BASE_ADDRESS;
10175 }
10176
10177
10178 pmap_lock(pmap, PMAP_LOCK_SHARED);
10179
10180 /*
10181 * For 4KB pages, we either "nest" at the level one page table (1GB) or level
10182 * two (2MB) depending on the address space layout. For 16KB pages, each level
10183 * one entry is 64GB, so we must go to the second level entry (32MB) in order
10184 * to "nest".
10185 *
10186 * Note: This is not "nesting" in the shared cache sense. This definition of
10187 * nesting just means inserting pointers to pre-allocated tables inside of
10188 * the passed in pmap to allow us to share page tables (which map the shared
10189 * page) for every task. This saves at least one page of memory per process
10190 * compared to creating new page tables in every process for mapping the
10191 * shared page.
10192 */
10193
10194 /**
10195 * Allocate the twig page tables if needed, and slam a pointer to the shared
10196 * page's tables into place.
10197 */
10198 while ((ttep = pmap_ttne(pmap, commpage_level, commpage_vaddr)) == TT_ENTRY_NULL) {
10199 pmap_unlock(pmap, PMAP_LOCK_SHARED);
10200
10201 kr = pmap_expand(pmap, commpage_vaddr, 0, commpage_level);
10202
10203 if (kr != KERN_SUCCESS) {
10204 panic("Failed to pmap_expand for commpage, pmap=%p", pmap);
10205 }
10206
10207 pmap_lock(pmap, PMAP_LOCK_SHARED);
10208 }
10209
10210 if (*ttep != ARM_PTE_EMPTY) {
10211 panic("%s: Found something mapped at the commpage address?!", __FUNCTION__);
10212 }
10213
10214 sptm_map_table(pmap->ttep, pt_attr_align_va(pt_attr, commpage_level, commpage_vaddr), (sptm_pt_level_t)commpage_level,
10215 (commpage_table & ARM_TTE_TABLE_MASK) | ARM_TTE_TYPE_TABLE | ARM_TTE_VALID);
10216
10217 pmap_unlock(pmap, PMAP_LOCK_SHARED);
10218
10219 return kr;
10220 }
10221
10222 static void
10223 pmap_unmap_commpage(
10224 pmap_t pmap)
10225 {
10226 pt_entry_t *ptep;
10227 vm_offset_t commpage_vaddr;
10228
10229 const pt_attr_t * const pt_attr = pmap_get_pt_attr(pmap);
10230 const unsigned int commpage_level = pt_attr_commpage_level(pt_attr);
10231 __assert_only pmap_paddr_t commpage_pa = commpage_data_pa;
10232
10233 if (pmap_is_64bit(pmap)) {
10234 commpage_vaddr = _COMM_PAGE64_BASE_ADDRESS;
10235 } else {
10236 commpage_vaddr = _COMM_PAGE32_BASE_ADDRESS;
10237 }
10238
10239
10240 ptep = pmap_pte(pmap, commpage_vaddr);
10241
10242 if (ptep == NULL) {
10243 return;
10244 }
10245
10246 /* It had better be mapped to the shared page. */
10247 if (pte_to_pa(*ptep) != commpage_pa) {
10248 panic("%s: non-commpage PA 0x%llx mapped at VA 0x%llx in pmap %p; expected 0x%llx",
10249 __func__, (unsigned long long)pte_to_pa(*ptep), (unsigned long long)commpage_vaddr,
10250 pmap, (unsigned long long)commpage_pa);
10251 }
10252
10253 sptm_unmap_table(pmap->ttep, pt_attr_align_va(pt_attr, commpage_level, commpage_vaddr), (sptm_pt_level_t)commpage_level);
10254 }
10255
10256 void
10257 pmap_insert_commpage(
10258 pmap_t pmap)
10259 {
10260 pmap_insert_commpage_internal(pmap);
10261 }
10262
10263 static boolean_t
10264 pmap_is_64bit(
10265 pmap_t pmap)
10266 {
10267 return pmap->is_64bit;
10268 }
10269
10270 bool
10271 pmap_is_exotic(
10272 pmap_t pmap __unused)
10273 {
10274 return false;
10275 }
10276
10277
10278 /* ARMTODO -- an implementation that accounts for
10279 * holes in the physical map, if any.
10280 */
10281 boolean_t
10282 pmap_valid_page(
10283 ppnum_t pn)
10284 {
10285 return pa_valid(ptoa(pn));
10286 }
10287
10288 boolean_t
10289 pmap_bootloader_page(
10290 ppnum_t pn)
10291 {
10292 pmap_paddr_t paddr = ptoa(pn);
10293
10294 if (pa_valid(paddr)) {
10295 return FALSE;
10296 }
10297 pmap_io_range_t *io_rgn = pmap_find_io_attr(paddr);
10298 return (io_rgn != NULL) && (io_rgn->wimg & PMAP_IO_RANGE_CARVEOUT);
10299 }
10300
10301 MARK_AS_PMAP_TEXT boolean_t
10302 pmap_is_empty_internal(
10303 pmap_t pmap,
10304 vm_map_offset_t va_start,
10305 vm_map_offset_t va_end)
10306 {
10307 vm_map_offset_t block_start, block_end;
10308 tt_entry_t *tte_p;
10309
10310 if (pmap == NULL) {
10311 return TRUE;
10312 }
10313
10314 validate_pmap(pmap);
10315
10316 __unused const pt_attr_t * const pt_attr = pmap_get_pt_attr(pmap);
10317 unsigned int initial_not_in_kdp = not_in_kdp;
10318
10319 if ((pmap != kernel_pmap) && (initial_not_in_kdp)) {
10320 pmap_lock(pmap, PMAP_LOCK_SHARED);
10321 }
10322
10323
10324 /* TODO: This will be faster if we increment ttep at each level. */
10325 block_start = va_start;
10326
10327 while (block_start < va_end) {
10328 pt_entry_t *bpte_p, *epte_p;
10329 pt_entry_t *pte_p;
10330
10331 block_end = (block_start + pt_attr_twig_size(pt_attr)) & ~pt_attr_twig_offmask(pt_attr);
10332 if (block_end > va_end) {
10333 block_end = va_end;
10334 }
10335
10336 tte_p = pmap_tte(pmap, block_start);
10337 if ((tte_p != PT_ENTRY_NULL)
10338 && ((*tte_p & ARM_TTE_TYPE_MASK) == ARM_TTE_TYPE_TABLE)) {
10339 pte_p = (pt_entry_t *) ttetokv(*tte_p);
10340 bpte_p = &pte_p[pte_index(pt_attr, block_start)];
10341 epte_p = &pte_p[pte_index(pt_attr, block_end)];
10342
10343 for (pte_p = bpte_p; pte_p < epte_p; pte_p++) {
10344 if (*pte_p != ARM_PTE_EMPTY) {
10345 if ((pmap != kernel_pmap) && (initial_not_in_kdp)) {
10346 pmap_unlock(pmap, PMAP_LOCK_SHARED);
10347 }
10348 return FALSE;
10349 }
10350 }
10351 }
10352 block_start = block_end;
10353 }
10354
10355 if ((pmap != kernel_pmap) && (initial_not_in_kdp)) {
10356 pmap_unlock(pmap, PMAP_LOCK_SHARED);
10357 }
10358
10359 return TRUE;
10360 }
10361
10362 boolean_t
10363 pmap_is_empty(
10364 pmap_t pmap,
10365 vm_map_offset_t va_start,
10366 vm_map_offset_t va_end)
10367 {
10368 return pmap_is_empty_internal(pmap, va_start, va_end);
10369 }
10370
10371 vm_map_offset_t
10372 pmap_max_offset(
10373 boolean_t is64,
10374 unsigned int option)
10375 {
10376 return (is64) ? pmap_max_64bit_offset(option) : pmap_max_32bit_offset(option);
10377 }
10378
10379 vm_map_offset_t
10380 pmap_max_64bit_offset(
10381 __unused unsigned int option)
10382 {
10383 vm_map_offset_t max_offset_ret = 0;
10384
10385 const vm_map_offset_t min_max_offset = ARM64_MIN_MAX_ADDRESS; // end of shared region + 512MB for various purposes
10386 if (option == ARM_PMAP_MAX_OFFSET_DEFAULT) {
10387 max_offset_ret = arm64_pmap_max_offset_default;
10388 } else if (option == ARM_PMAP_MAX_OFFSET_MIN) {
10389 max_offset_ret = min_max_offset;
10390 } else if (option == ARM_PMAP_MAX_OFFSET_MAX) {
10391 max_offset_ret = MACH_VM_MAX_ADDRESS;
10392 } else if (option == ARM_PMAP_MAX_OFFSET_DEVICE) {
10393 if (arm64_pmap_max_offset_default) {
10394 max_offset_ret = arm64_pmap_max_offset_default;
10395 } else if (max_mem > 0xC0000000) {
10396 // devices with > 3GB of memory
10397 max_offset_ret = ARM64_MAX_OFFSET_DEVICE_LARGE;
10398 } else if (max_mem > 0x40000000) {
10399 // devices with > 1GB and <= 3GB of memory
10400 max_offset_ret = ARM64_MAX_OFFSET_DEVICE_SMALL;
10401 } else {
10402 // devices with <= 1 GB of memory
10403 max_offset_ret = min_max_offset;
10404 }
10405 } else if (option == ARM_PMAP_MAX_OFFSET_JUMBO) {
10406 if (arm64_pmap_max_offset_default) {
10407 // Allow the boot-arg to override jumbo size
10408 max_offset_ret = arm64_pmap_max_offset_default;
10409 } else {
10410 max_offset_ret = MACH_VM_JUMBO_ADDRESS; // Max offset is 64GB for pmaps with special "jumbo" blessing
10411 }
10412 #if XNU_TARGET_OS_IOS && EXTENDED_USER_VA_SUPPORT
10413 } else if (option == ARM_PMAP_MAX_OFFSET_EXTRA_JUMBO) {
10414 max_offset_ret = MACH_VM_MAX_ADDRESS;
10415 #endif /* XNU_TARGET_OS_IOS && EXTENDED_USER_VA_SUPPORT */
10416 } else {
10417 panic("pmap_max_64bit_offset illegal option 0x%x", option);
10418 }
10419
10420 assert(max_offset_ret <= MACH_VM_MAX_ADDRESS);
10421 if (option != ARM_PMAP_MAX_OFFSET_DEFAULT) {
10422 assert(max_offset_ret >= min_max_offset);
10423 }
10424
10425 return max_offset_ret;
10426 }
10427
10428 vm_map_offset_t
10429 pmap_max_32bit_offset(
10430 unsigned int option)
10431 {
10432 vm_map_offset_t max_offset_ret = 0;
10433
10434 if (option == ARM_PMAP_MAX_OFFSET_DEFAULT) {
10435 max_offset_ret = arm_pmap_max_offset_default;
10436 } else if (option == ARM_PMAP_MAX_OFFSET_MIN) {
10437 max_offset_ret = VM_MAX_ADDRESS;
10438 } else if (option == ARM_PMAP_MAX_OFFSET_MAX) {
10439 max_offset_ret = VM_MAX_ADDRESS;
10440 } else if (option == ARM_PMAP_MAX_OFFSET_DEVICE) {
10441 if (arm_pmap_max_offset_default) {
10442 max_offset_ret = arm_pmap_max_offset_default;
10443 } else if (max_mem > 0x20000000) {
10444 max_offset_ret = VM_MAX_ADDRESS;
10445 } else {
10446 max_offset_ret = VM_MAX_ADDRESS;
10447 }
10448 } else if (option == ARM_PMAP_MAX_OFFSET_JUMBO) {
10449 max_offset_ret = VM_MAX_ADDRESS;
10450 } else {
10451 panic("pmap_max_32bit_offset illegal option 0x%x", option);
10452 }
10453
10454 assert(max_offset_ret <= MACH_VM_MAX_ADDRESS);
10455 return max_offset_ret;
10456 }
10457
10458 #if CONFIG_DTRACE
10459 /*
10460 * Constrain DTrace copyin/copyout actions
10461 */
10462 extern kern_return_t dtrace_copyio_preflight(addr64_t);
10463 extern kern_return_t dtrace_copyio_postflight(addr64_t);
10464
10465 kern_return_t
10466 dtrace_copyio_preflight(
10467 __unused addr64_t va)
10468 {
10469 if (current_map() == kernel_map) {
10470 return KERN_FAILURE;
10471 } else {
10472 return KERN_SUCCESS;
10473 }
10474 }
10475
10476 kern_return_t
10477 dtrace_copyio_postflight(
10478 __unused addr64_t va)
10479 {
10480 return KERN_SUCCESS;
10481 }
10482 #endif /* CONFIG_DTRACE */
10483
10484
10485 void
10486 pmap_flush_context_init(__unused pmap_flush_context *pfc)
10487 {
10488 }
10489
10490
10491 void
10492 pmap_flush(
10493 __unused pmap_flush_context *cpus_to_flush)
10494 {
10495 /* not implemented yet */
10496 return;
10497 }
10498
10499 /**
10500 * Perform basic validation checks on the destination only and
10501 * corresponding offset/sizes prior to writing to a read only allocation.
10502 *
10503 * @note Should be called before writing to an allocation from the read
10504 * only allocator.
10505 *
10506 * @param zid The ID of the zone the allocation belongs to.
10507 * @param va VA of element being modified (destination).
10508 * @param offset Offset being written to, in the element.
10509 * @param new_data_size Size of modification.
10510 *
10511 */
10512
10513 MARK_AS_PMAP_TEXT static void
10514 pmap_ro_zone_validate_element_dst(
10515 zone_id_t zid,
10516 vm_offset_t va,
10517 vm_offset_t offset,
10518 vm_size_t new_data_size)
10519 {
10520 if (__improbable((zid < ZONE_ID__FIRST_RO) || (zid > ZONE_ID__LAST_RO))) {
10521 panic("%s: ZoneID %u outside RO range %u - %u", __func__, zid,
10522 ZONE_ID__FIRST_RO, ZONE_ID__LAST_RO);
10523 }
10524
10525 vm_size_t elem_size = zone_ro_size_params[zid].z_elem_size;
10526
10527 /* Check element is from correct zone and properly aligned */
10528 zone_require_ro(zid, elem_size, (void*)va);
10529
10530 if (__improbable(new_data_size > (elem_size - offset))) {
10531 panic("%s: New data size %lu too large for elem size %lu at addr %p",
10532 __func__, (uintptr_t)new_data_size, (uintptr_t)elem_size, (void*)va);
10533 }
10534 if (__improbable(offset >= elem_size)) {
10535 panic("%s: Offset %lu too large for elem size %lu at addr %p",
10536 __func__, (uintptr_t)offset, (uintptr_t)elem_size, (void*)va);
10537 }
10538 }
10539
10540
10541 /**
10542 * Perform basic validation checks on the source, destination and
10543 * corresponding offset/sizes prior to writing to a read only allocation.
10544 *
10545 * @note Should be called before writing to an allocation from the read
10546 * only allocator.
10547 *
10548 * @param zid The ID of the zone the allocation belongs to.
10549 * @param va VA of element being modified (destination).
10550 * @param offset Offset being written to, in the element.
10551 * @param new_data Pointer to new data (source).
10552 * @param new_data_size Size of modification.
10553 *
10554 */
10555
10556 MARK_AS_PMAP_TEXT static void
10557 pmap_ro_zone_validate_element(
10558 zone_id_t zid,
10559 vm_offset_t va,
10560 vm_offset_t offset,
10561 const vm_offset_t new_data,
10562 vm_size_t new_data_size)
10563 {
10564 vm_offset_t sum = 0;
10565
10566 if (__improbable(os_add_overflow(new_data, new_data_size, &sum))) {
10567 panic("%s: Integer addition overflow %p + %lu = %lu",
10568 __func__, (void*)new_data, (uintptr_t)new_data_size, (uintptr_t)sum);
10569 }
10570
10571 pmap_ro_zone_validate_element_dst(zid, va, offset, new_data_size);
10572 }
10573
10574 /**
10575 * Function to configure RO zone access permissions for a forthcoming write operation.
10576 */
10577 static void
10578 pmap_ro_zone_prepare_write(void)
10579 {
10580 }
10581
10582 /**
10583 * Function to indicate that a preceding RO zone write operation is complete.
10584 */
10585 static void
10586 pmap_ro_zone_complete_write(void)
10587 {
10588 }
10589
10590 /**
10591 * Function to align an address or size to the required RO zone mapping alignment.
10592 *
10593 * For the SPTM the RO zone region must be aligned on a twig boundary so that at least
10594 * the last-level kernel pagetable can be of the appropriate SPTM RO zone table type,
10595 * which allows the SPTM to enforce RO zone mapping permission restrictions.
10596 *
10597 * @param value the address or size to be aligned.
10598 *
10599 * @return the aligned value
10600 */
10601 vm_offset_t
10602 pmap_ro_zone_align(vm_offset_t value)
10603 {
10604 const pt_attr_t * const pt_attr = pmap_get_pt_attr(kernel_pmap);
10605 return PMAP_ALIGN(value, pt_attr_twig_size(pt_attr));
10606 }
10607
10608 /**
10609 * Function to copy kauth_cred from new_data to kv.
10610 * Function defined in "kern_prot.c"
10611 *
10612 * @note Will be removed upon completion of
10613 * <rdar://problem/72635194> Compiler PAC support for memcpy.
10614 *
10615 * @param kv Address to copy new data to.
10616 * @param new_data Pointer to new data.
10617 *
10618 */
10619
10620 extern void
10621 kauth_cred_copy(const uintptr_t kv, const uintptr_t new_data);
10622
10623 /**
10624 * Zalloc-specific memcpy that writes through the physical aperture
10625 * and ensures the element being modified is from a read-only zone.
10626 *
10627 * @note Designed to work only with the zone allocator's read-only submap.
10628 *
10629 * @param zid The ID of the zone to allocate from.
10630 * @param va VA of element to be modified.
10631 * @param offset Offset from element.
10632 * @param new_data Pointer to new data.
10633 * @param new_data_size Size of modification.
10634 *
10635 */
10636
10637 void
10638 pmap_ro_zone_memcpy(
10639 zone_id_t zid,
10640 vm_offset_t va,
10641 vm_offset_t offset,
10642 const vm_offset_t new_data,
10643 vm_size_t new_data_size)
10644 {
10645 pmap_ro_zone_memcpy_internal(zid, va, offset, new_data, new_data_size);
10646 }
10647
10648 MARK_AS_PMAP_TEXT void
10649 pmap_ro_zone_memcpy_internal(
10650 zone_id_t zid,
10651 vm_offset_t va,
10652 vm_offset_t offset,
10653 const vm_offset_t new_data,
10654 vm_size_t new_data_size)
10655 {
10656 if (!new_data || new_data_size == 0) {
10657 return;
10658 }
10659
10660 const pmap_paddr_t pa = kvtophys_nofail(va + offset);
10661 const bool istate = ml_set_interrupts_enabled(FALSE);
10662 pmap_ro_zone_validate_element(zid, va, offset, new_data, new_data_size);
10663 pmap_ro_zone_prepare_write();
10664 memcpy((void*)phystokv(pa), (void*)new_data, new_data_size);
10665 pmap_ro_zone_complete_write();
10666 ml_set_interrupts_enabled(istate);
10667 }
10668
10669 /**
10670 * Zalloc-specific function to atomically mutate fields of an element that
10671 * belongs to a read-only zone, via the physcial aperture.
10672 *
10673 * @note Designed to work only with the zone allocator's read-only submap.
10674 *
10675 * @param zid The ID of the zone the element belongs to.
10676 * @param va VA of element to be modified.
10677 * @param offset Offset in element.
10678 * @param op Atomic operation to perform.
10679 * @param value Mutation value.
10680 *
10681 */
10682
10683 uint64_t
10684 pmap_ro_zone_atomic_op(
10685 zone_id_t zid,
10686 vm_offset_t va,
10687 vm_offset_t offset,
10688 zro_atomic_op_t op,
10689 uint64_t value)
10690 {
10691 return pmap_ro_zone_atomic_op_internal(zid, va, offset, op, value);
10692 }
10693
10694 MARK_AS_PMAP_TEXT uint64_t
10695 pmap_ro_zone_atomic_op_internal(
10696 zone_id_t zid,
10697 vm_offset_t va,
10698 vm_offset_t offset,
10699 zro_atomic_op_t op,
10700 uint64_t value)
10701 {
10702 const pmap_paddr_t pa = kvtophys_nofail(va + offset);
10703 vm_size_t value_size = op & 0xf;
10704 const boolean_t istate = ml_set_interrupts_enabled(FALSE);
10705
10706 pmap_ro_zone_validate_element_dst(zid, va, offset, value_size);
10707 pmap_ro_zone_prepare_write();
10708 value = __zalloc_ro_mut_atomic(phystokv(pa), op, value);
10709 pmap_ro_zone_complete_write();
10710 ml_set_interrupts_enabled(istate);
10711
10712 return value;
10713 }
10714
10715 /**
10716 * bzero for allocations from read only zones, that writes through the
10717 * physical aperture.
10718 *
10719 * @note This is called by the zfree path of all allocations from read
10720 * only zones.
10721 *
10722 * @param zid The ID of the zone the allocation belongs to.
10723 * @param va VA of element to be zeroed.
10724 * @param offset Offset in the element.
10725 * @param size Size of allocation.
10726 *
10727 */
10728
10729 void
10730 pmap_ro_zone_bzero(
10731 zone_id_t zid,
10732 vm_offset_t va,
10733 vm_offset_t offset,
10734 vm_size_t size)
10735 {
10736 pmap_ro_zone_bzero_internal(zid, va, offset, size);
10737 }
10738
10739 MARK_AS_PMAP_TEXT void
10740 pmap_ro_zone_bzero_internal(
10741 zone_id_t zid,
10742 vm_offset_t va,
10743 vm_offset_t offset,
10744 vm_size_t size)
10745 {
10746 const pmap_paddr_t pa = kvtophys_nofail(va + offset);
10747 const boolean_t istate = ml_set_interrupts_enabled(FALSE);
10748 pmap_ro_zone_validate_element(zid, va, offset, 0, size);
10749 pmap_ro_zone_prepare_write();
10750 bzero((void*)phystokv(pa), size);
10751 pmap_ro_zone_complete_write();
10752 ml_set_interrupts_enabled(istate);
10753 }
10754
10755 #define PMAP_RESIDENT_INVALID ((mach_vm_size_t)-1)
10756
10757 MARK_AS_PMAP_TEXT mach_vm_size_t
10758 pmap_query_resident_internal(
10759 pmap_t pmap,
10760 vm_map_address_t start,
10761 vm_map_address_t end,
10762 mach_vm_size_t *compressed_bytes_p)
10763 {
10764 mach_vm_size_t resident_bytes = 0;
10765 mach_vm_size_t compressed_bytes = 0;
10766
10767 pt_entry_t *bpte, *epte;
10768 pt_entry_t *pte_p;
10769 tt_entry_t *tte_p;
10770
10771 if (pmap == NULL) {
10772 return PMAP_RESIDENT_INVALID;
10773 }
10774
10775 validate_pmap(pmap);
10776
10777 const pt_attr_t * const pt_attr = pmap_get_pt_attr(pmap);
10778
10779 /* Ensure that this request is valid, and addresses exactly one TTE. */
10780 if (__improbable((start % pt_attr_page_size(pt_attr)) ||
10781 (end % pt_attr_page_size(pt_attr)))) {
10782 panic("%s: address range %p, %p not page-aligned to 0x%llx", __func__, (void*)start, (void*)end, pt_attr_page_size(pt_attr));
10783 }
10784
10785 if (__improbable((end < start) || (end > ((start + pt_attr_twig_size(pt_attr)) & ~pt_attr_twig_offmask(pt_attr))))) {
10786 panic("%s: invalid address range %p, %p", __func__, (void*)start, (void*)end);
10787 }
10788
10789 pmap_lock(pmap, PMAP_LOCK_SHARED);
10790 tte_p = pmap_tte(pmap, start);
10791 if (tte_p == (tt_entry_t *) NULL) {
10792 pmap_unlock(pmap, PMAP_LOCK_SHARED);
10793 return PMAP_RESIDENT_INVALID;
10794 }
10795 if ((*tte_p & ARM_TTE_TYPE_MASK) == ARM_TTE_TYPE_TABLE) {
10796 pte_p = (pt_entry_t *) ttetokv(*tte_p);
10797 bpte = &pte_p[pte_index(pt_attr, start)];
10798 epte = &pte_p[pte_index(pt_attr, end)];
10799
10800 for (; bpte < epte; bpte++) {
10801 if (pte_is_compressed(*bpte, bpte)) {
10802 compressed_bytes += pt_attr_page_size(pt_attr);
10803 } else if (pa_valid(pte_to_pa(*bpte))) {
10804 resident_bytes += pt_attr_page_size(pt_attr);
10805 }
10806 }
10807 }
10808 pmap_unlock(pmap, PMAP_LOCK_SHARED);
10809
10810 if (compressed_bytes_p) {
10811 *compressed_bytes_p += compressed_bytes;
10812 }
10813
10814 return resident_bytes;
10815 }
10816
10817 mach_vm_size_t
10818 pmap_query_resident(
10819 pmap_t pmap,
10820 vm_map_address_t start,
10821 vm_map_address_t end,
10822 mach_vm_size_t *compressed_bytes_p)
10823 {
10824 mach_vm_size_t total_resident_bytes;
10825 mach_vm_size_t compressed_bytes;
10826 vm_map_address_t va;
10827
10828
10829 if (pmap == PMAP_NULL) {
10830 if (compressed_bytes_p) {
10831 *compressed_bytes_p = 0;
10832 }
10833 return 0;
10834 }
10835
10836 __unused const pt_attr_t * const pt_attr = pmap_get_pt_attr(pmap);
10837
10838 total_resident_bytes = 0;
10839 compressed_bytes = 0;
10840
10841 PMAP_TRACE(3, PMAP_CODE(PMAP__QUERY_RESIDENT) | DBG_FUNC_START,
10842 VM_KERNEL_ADDRHIDE(pmap), VM_KERNEL_ADDRHIDE(start),
10843 VM_KERNEL_ADDRHIDE(end));
10844
10845 va = start;
10846 while (va < end) {
10847 vm_map_address_t l;
10848 mach_vm_size_t resident_bytes;
10849
10850 l = ((va + pt_attr_twig_size(pt_attr)) & ~pt_attr_twig_offmask(pt_attr));
10851
10852 if (l > end) {
10853 l = end;
10854 }
10855 resident_bytes = pmap_query_resident_internal(pmap, va, l, compressed_bytes_p);
10856 if (resident_bytes == PMAP_RESIDENT_INVALID) {
10857 break;
10858 }
10859
10860 total_resident_bytes += resident_bytes;
10861
10862 va = l;
10863 }
10864
10865 if (compressed_bytes_p) {
10866 *compressed_bytes_p = compressed_bytes;
10867 }
10868
10869 PMAP_TRACE(3, PMAP_CODE(PMAP__QUERY_RESIDENT) | DBG_FUNC_END,
10870 total_resident_bytes);
10871
10872 return total_resident_bytes;
10873 }
10874
10875 #if MACH_ASSERT
10876 static void
10877 pmap_check_ledgers(
10878 pmap_t pmap)
10879 {
10880 int pid;
10881 char *procname;
10882
10883 if (pmap->pmap_pid == 0 || pmap->pmap_pid == -1) {
10884 /*
10885 * This pmap was not or is no longer fully associated
10886 * with a task (e.g. the old pmap after a fork()/exec() or
10887 * spawn()). Its "ledger" still points at a task that is
10888 * now using a different (and active) address space, so
10889 * we can't check that all the pmap ledgers are balanced here.
10890 *
10891 * If the "pid" is set, that means that we went through
10892 * pmap_set_process() in task_terminate_internal(), so
10893 * this task's ledger should not have been re-used and
10894 * all the pmap ledgers should be back to 0.
10895 */
10896 return;
10897 }
10898
10899 pid = pmap->pmap_pid;
10900 procname = pmap->pmap_procname;
10901
10902 vm_map_pmap_check_ledgers(pmap, pmap->ledger, pid, procname);
10903 }
10904 #endif /* MACH_ASSERT */
10905
10906 void
10907 pmap_advise_pagezero_range(__unused pmap_t p, __unused uint64_t a)
10908 {
10909 }
10910
10911 /**
10912 * The minimum shared region nesting size is used by the VM to determine when to
10913 * break up large mappings to nested regions. The smallest size that these
10914 * mappings can be broken into is determined by what page table level those
10915 * regions are being nested in at and the size of the page tables.
10916 *
10917 * For instance, if a nested region is nesting at L2 for a process utilizing
10918 * 16KB page tables, then the minimum nesting size would be 32MB (size of an L2
10919 * block entry).
10920 *
10921 * @param pmap The target pmap to determine the block size based on whether it's
10922 * using 16KB or 4KB page tables.
10923 */
10924 uint64_t
10925 pmap_shared_region_size_min(__unused pmap_t pmap)
10926 {
10927 const pt_attr_t * const pt_attr = pmap_get_pt_attr(pmap);
10928
10929 /**
10930 * We always nest the shared region at L2 (32MB for 16KB pages, 8MB for
10931 * 4KB pages). This means that a target pmap will contain L2 entries that
10932 * point to shared L3 page tables in the shared region pmap.
10933 */
10934 const uint64_t page_ratio = PAGE_SIZE / pt_attr_page_size(pt_attr);
10935 return pt_attr_twig_size(pt_attr) * page_ratio;
10936 }
10937
10938 boolean_t
10939 pmap_enforces_execute_only(
10940 pmap_t pmap)
10941 {
10942 return pmap != kernel_pmap;
10943 }
10944
10945 MARK_AS_PMAP_TEXT void
10946 pmap_set_vm_map_cs_enforced_internal(
10947 pmap_t pmap,
10948 bool new_value)
10949 {
10950 validate_pmap_mutable(pmap);
10951 pmap->pmap_vm_map_cs_enforced = new_value;
10952 }
10953
10954 void
10955 pmap_set_vm_map_cs_enforced(
10956 pmap_t pmap,
10957 bool new_value)
10958 {
10959 pmap_set_vm_map_cs_enforced_internal(pmap, new_value);
10960 }
10961
10962 extern int cs_process_enforcement_enable;
10963 bool
10964 pmap_get_vm_map_cs_enforced(
10965 pmap_t pmap)
10966 {
10967 if (cs_process_enforcement_enable) {
10968 return true;
10969 }
10970 return pmap->pmap_vm_map_cs_enforced;
10971 }
10972
10973 MARK_AS_PMAP_TEXT void
10974 pmap_set_jit_entitled_internal(
10975 __unused pmap_t pmap)
10976 {
10977 }
10978
10979 void
10980 pmap_set_jit_entitled(
10981 pmap_t pmap)
10982 {
10983 pmap_set_jit_entitled_internal(pmap);
10984 }
10985
10986 bool
10987 pmap_get_jit_entitled(
10988 __unused pmap_t pmap)
10989 {
10990 return false;
10991 }
10992
10993 MARK_AS_PMAP_TEXT void
10994 pmap_set_tpro_internal(
10995 __unused pmap_t pmap)
10996 {
10997 return;
10998 }
10999
11000 void
11001 pmap_set_tpro(
11002 pmap_t pmap)
11003 {
11004 pmap_set_tpro_internal(pmap);
11005 }
11006
11007 bool
11008 pmap_get_tpro(
11009 __unused pmap_t pmap)
11010 {
11011 return false;
11012 }
11013
11014 uint64_t pmap_query_page_info_retries MARK_AS_PMAP_DATA;
11015
11016 MARK_AS_PMAP_TEXT kern_return_t
11017 pmap_query_page_info_internal(
11018 pmap_t pmap,
11019 vm_map_offset_t va,
11020 int *disp_p)
11021 {
11022 pmap_paddr_t pa;
11023 int disp;
11024 unsigned int pai;
11025 pt_entry_t *pte_p;
11026 pv_entry_t *pve_p;
11027
11028 if (pmap == PMAP_NULL || pmap == kernel_pmap) {
11029 *disp_p = 0;
11030 return KERN_INVALID_ARGUMENT;
11031 }
11032
11033 validate_pmap(pmap);
11034 pmap_lock(pmap, PMAP_LOCK_SHARED);
11035
11036 try_again:
11037 disp = 0;
11038
11039 pte_p = pmap_pte(pmap, va);
11040 if (pte_p == PT_ENTRY_NULL) {
11041 goto done;
11042 }
11043
11044 const pt_entry_t pte = os_atomic_load(pte_p, relaxed);
11045 pa = pte_to_pa(pte);
11046 if (pa == 0) {
11047 if (pte_is_compressed(pte, pte_p)) {
11048 disp |= PMAP_QUERY_PAGE_COMPRESSED;
11049 if (pte & ARM_PTE_COMPRESSED_ALT) {
11050 disp |= PMAP_QUERY_PAGE_COMPRESSED_ALTACCT;
11051 }
11052 }
11053 } else {
11054 disp |= PMAP_QUERY_PAGE_PRESENT;
11055 pai = pa_index(pa);
11056 if (!pa_valid(pa)) {
11057 goto done;
11058 }
11059 locked_pvh_t locked_pvh = pvh_lock(pai);
11060 if (__improbable(pte != os_atomic_load(pte_p, relaxed))) {
11061 /* something changed: try again */
11062 pvh_unlock(&locked_pvh);
11063 pmap_query_page_info_retries++;
11064 goto try_again;
11065 }
11066 pve_p = PV_ENTRY_NULL;
11067 int pve_ptep_idx = 0;
11068 if (pvh_test_type(locked_pvh.pvh, PVH_TYPE_PVEP)) {
11069 unsigned int npves = 0;
11070 pve_p = pvh_pve_list(locked_pvh.pvh);
11071 while (pve_p != PV_ENTRY_NULL &&
11072 (pve_ptep_idx = pve_find_ptep_index(pve_p, pte_p)) == -1) {
11073 if (__improbable(npves == (SPTM_MAPPING_LIMIT / PTE_PER_PVE))) {
11074 pvh_lock_enter_sleep_mode(&locked_pvh);
11075 }
11076 pve_p = pve_next(pve_p);
11077 npves++;
11078 }
11079 }
11080
11081 if (ppattr_pve_is_altacct(pai, pve_p, pve_ptep_idx)) {
11082 disp |= PMAP_QUERY_PAGE_ALTACCT;
11083 } else if (ppattr_test_reusable(pai)) {
11084 disp |= PMAP_QUERY_PAGE_REUSABLE;
11085 } else if (ppattr_pve_is_internal(pai, pve_p, pve_ptep_idx)) {
11086 disp |= PMAP_QUERY_PAGE_INTERNAL;
11087 }
11088 pvh_unlock(&locked_pvh);
11089 }
11090
11091 done:
11092 pmap_unlock(pmap, PMAP_LOCK_SHARED);
11093 *disp_p = disp;
11094 return KERN_SUCCESS;
11095 }
11096
11097 kern_return_t
11098 pmap_query_page_info(
11099 pmap_t pmap,
11100 vm_map_offset_t va,
11101 int *disp_p)
11102 {
11103 return pmap_query_page_info_internal(pmap, va, disp_p);
11104 }
11105
11106
11107
11108 uint32_t
11109 pmap_user_va_bits(pmap_t pmap __unused)
11110 {
11111 #if __ARM_MIXED_PAGE_SIZE__
11112 uint64_t tcr_value = pmap_get_pt_attr(pmap)->pta_tcr_value;
11113 return 64 - ((tcr_value >> TCR_T0SZ_SHIFT) & TCR_TSZ_MASK);
11114 #else
11115 return 64 - T0SZ_BOOT;
11116 #endif
11117 }
11118
11119 uint32_t
11120 pmap_kernel_va_bits(void)
11121 {
11122 return 64 - T1SZ_BOOT;
11123 }
11124
11125 static vm_map_size_t
11126 pmap_user_va_size(pmap_t pmap)
11127 {
11128 return 1ULL << pmap_user_va_bits(pmap);
11129 }
11130
11131
11132 bool
11133 pmap_in_ppl(void)
11134 {
11135 return false;
11136 }
11137
11138 MARK_AS_PMAP_TEXT void
11139 pmap_footprint_suspend_internal(
11140 vm_map_t map,
11141 boolean_t suspend)
11142 {
11143 #if DEVELOPMENT || DEBUG
11144 if (suspend) {
11145 current_thread()->pmap_footprint_suspended = TRUE;
11146 map->pmap->footprint_was_suspended = TRUE;
11147 } else {
11148 current_thread()->pmap_footprint_suspended = FALSE;
11149 }
11150 #else /* DEVELOPMENT || DEBUG */
11151 (void) map;
11152 (void) suspend;
11153 #endif /* DEVELOPMENT || DEBUG */
11154 }
11155
11156 void
11157 pmap_footprint_suspend(
11158 vm_map_t map,
11159 boolean_t suspend)
11160 {
11161 pmap_footprint_suspend_internal(map, suspend);
11162 }
11163
11164 void
11165 pmap_nop(pmap_t pmap)
11166 {
11167 validate_pmap_mutable(pmap);
11168 }
11169
11170 pmap_t
11171 pmap_txm_kernel_pmap(void)
11172 {
11173 return kernel_pmap;
11174 }
11175
11176 TXMAddressSpace_t*
11177 pmap_txm_addr_space(const pmap_t pmap)
11178 {
11179 if (pmap) {
11180 return pmap->txm_addr_space;
11181 }
11182
11183 /*
11184 * When the passed in PMAP is NULL, it means the caller wishes to operate
11185 * on the current_pmap(). We could resolve and return that, but it is actually
11186 * safer to return NULL since these TXM interfaces also accept NULL inputs
11187 * which causes TXM to resolve to the current_pmap() equivalent internally.
11188 */
11189 return NULL;
11190 }
11191
11192 void
11193 pmap_txm_set_addr_space(
11194 pmap_t pmap,
11195 TXMAddressSpace_t *txm_addr_space)
11196 {
11197 assert(pmap != NULL);
11198
11199 if (pmap->txm_addr_space && txm_addr_space) {
11200 /* Attempted to overwrite the address space in the PMAP */
11201 panic("attempted ovewrite of TXM address space: %p | %p | %p",
11202 pmap, pmap->txm_addr_space, txm_addr_space);
11203 } else if (!pmap->txm_addr_space && !txm_addr_space) {
11204 /* This should never happen */
11205 panic("attempted NULL overwrite of TXM address space: %p", pmap);
11206 }
11207
11208 pmap->txm_addr_space = txm_addr_space;
11209 }
11210
11211 void
11212 pmap_txm_set_trust_level(
11213 pmap_t pmap,
11214 CSTrust_t trust_level)
11215 {
11216 assert(pmap != NULL);
11217
11218 CSTrust_t current_trust = pmap->txm_trust_level;
11219 if (current_trust != kCSTrustUntrusted) {
11220 panic("attempted to overwrite TXM trust on the pmap: %p", pmap);
11221 }
11222
11223 pmap->txm_trust_level = trust_level;
11224 }
11225
11226 kern_return_t
11227 pmap_txm_get_trust_level_kdp(
11228 pmap_t pmap,
11229 CSTrust_t *trust_level)
11230 {
11231 if (pmap == NULL) {
11232 return KERN_INVALID_ARGUMENT;
11233 } else if (ml_validate_nofault((vm_offset_t)pmap, sizeof(*pmap)) == false) {
11234 return KERN_INVALID_ARGUMENT;
11235 }
11236
11237 if (trust_level != NULL) {
11238 *trust_level = pmap->txm_trust_level;
11239 }
11240 return KERN_SUCCESS;
11241 }
11242
11243 kern_return_t
11244 pmap_txm_get_jit_address_range_kdp(
11245 pmap_t pmap,
11246 uintptr_t *jit_region_start,
11247 uintptr_t *jit_region_end)
11248 {
11249 if (ml_validate_nofault((vm_offset_t)pmap, sizeof(*pmap)) == false) {
11250 return KERN_INVALID_ARGUMENT;
11251 }
11252 TXMAddressSpace_t *txm_addr_space = pmap_txm_addr_space(pmap);
11253 if (NULL == txm_addr_space) {
11254 return KERN_INVALID_ARGUMENT;
11255 }
11256 if (ml_validate_nofault((vm_offset_t)txm_addr_space, sizeof(*txm_addr_space)) == false) {
11257 return KERN_INVALID_ARGUMENT;
11258 }
11259 /**
11260 * It's a bit gross that we're dereferencing what is supposed to be an abstract type.
11261 * If we were running in the TXM, we would always perform additional checks on txm_addr_space,
11262 * but this isn't necessary here, since we are running in the kernel and only using the results for
11263 * diagnostic purposes, rather than any policy enforcement.
11264 */
11265 if (txm_addr_space->jitRegion) {
11266 if (ml_validate_nofault((vm_offset_t)txm_addr_space->jitRegion, sizeof(txm_addr_space->jitRegion)) == false) {
11267 return KERN_INVALID_ARGUMENT;
11268 }
11269 if (txm_addr_space->jitRegion->addr && txm_addr_space->jitRegion->addrEnd) {
11270 *jit_region_start = txm_addr_space->jitRegion->addr;
11271 *jit_region_end = txm_addr_space->jitRegion->addrEnd;
11272 return KERN_SUCCESS;
11273 }
11274 }
11275 return KERN_NOT_FOUND;
11276 }
11277
11278 static pmap_t
11279 _pmap_txm_resolve_pmap(pmap_t pmap)
11280 {
11281 if (pmap == NULL) {
11282 pmap = current_pmap();
11283 if (pmap == kernel_pmap) {
11284 return NULL;
11285 }
11286 }
11287
11288 return pmap;
11289 }
11290
11291 void
11292 pmap_txm_acquire_shared_lock(pmap_t pmap)
11293 {
11294 pmap = _pmap_txm_resolve_pmap(pmap);
11295 if (!pmap) {
11296 return;
11297 }
11298
11299 lck_rw_lock_shared(&pmap->txm_lck);
11300 }
11301
11302 void
11303 pmap_txm_release_shared_lock(pmap_t pmap)
11304 {
11305 pmap = _pmap_txm_resolve_pmap(pmap);
11306 if (!pmap) {
11307 return;
11308 }
11309
11310 lck_rw_unlock_shared(&pmap->txm_lck);
11311 }
11312
11313 void
11314 pmap_txm_acquire_exclusive_lock(pmap_t pmap)
11315 {
11316 pmap = _pmap_txm_resolve_pmap(pmap);
11317 if (!pmap) {
11318 return;
11319 }
11320
11321 lck_rw_lock_exclusive(&pmap->txm_lck);
11322 }
11323
11324 void
11325 pmap_txm_release_exclusive_lock(pmap_t pmap)
11326 {
11327 pmap = _pmap_txm_resolve_pmap(pmap);
11328 if (!pmap) {
11329 return;
11330 }
11331
11332 lck_rw_unlock_exclusive(&pmap->txm_lck);
11333 }
11334
11335 static void
11336 _pmap_txm_transfer_page(const pmap_paddr_t addr)
11337 {
11338 sptm_retype_params_t retype_params = {
11339 .raw = SPTM_RETYPE_PARAMS_NULL
11340 };
11341
11342 /* Retype through the SPTM */
11343 sptm_retype(addr, XNU_DEFAULT, TXM_DEFAULT, retype_params);
11344 }
11345
11346 /**
11347 * Prepare a page for retyping to TXM_DEFAULT by clearing its
11348 * internal flags.
11349 *
11350 * @param pa Physical address of the page.
11351 */
11352 static inline void
11353 _pmap_txm_retype_prepare(const pmap_paddr_t pa)
11354 {
11355 const sptm_retype_params_t retype_params = {
11356 .raw = SPTM_RETYPE_PARAMS_NULL
11357 };
11358
11359 /**
11360 * SPTM allows XNU_DEFAULT pages to request deferral of TLB flushing
11361 * when their PTE is updated, which is an important performance
11362 * optimization. However, this also allows an attacker controlled
11363 * XNU to exploit a read reference with a stale write-enabled PTE in
11364 * TLB. This is fine as long as the page is not retyped and the damage
11365 * will be contained within XNU domain. However, when such a page needs
11366 * to be retyped, SPTM has to make sure there's no outstanding
11367 * reference, or there's no history of deferring TLBIs. Internally,
11368 * SPTM maintains a flag tracking past deferred TLBIs that only gets
11369 * cleared on retyping with no outstanding reference. Therefore, we
11370 * do a dummy retype to XNU_DEFAULT itself to clear the internal flag,
11371 * before we actually transfer this page to TXM domain. To make sure
11372 * SPTM won't throw a violation, all the mappings to the page have to
11373 * be removed before calling this.
11374 */
11375 sptm_retype(pa, XNU_DEFAULT, XNU_DEFAULT, retype_params);
11376 }
11377
11378 /**
11379 * Transfer an XNU owned page to TXM domain.
11380 *
11381 * @param addr Kernel virtual address of the page. It has to be page size
11382 * aligned.
11383 */
11384 void
11385 pmap_txm_transfer_page(const vm_address_t addr)
11386 {
11387 assert((addr & PAGE_MASK) == 0);
11388
11389 const pmap_paddr_t pa = kvtophys_nofail(addr);
11390 const unsigned int pai = pa_index(pa);
11391
11392 /* Lock the PVH lock to prevent concurrent updates to the mappings during the self retype below. */
11393 locked_pvh_t locked_pvh = pvh_lock(pai);
11394
11395 /* Disconnect the mapping to assure SPTM of no pending TLBI. */
11396 pmap_page_protect_options_with_flush_range((ppnum_t)atop(pa), VM_PROT_NONE,
11397 PMAP_OPTIONS_PPO_PENDING_RETYPE, &locked_pvh, NULL);
11398
11399 /* Self retype to clear the SPTM internal flags tracking delayed TLBIs for revoked writes. */
11400 _pmap_txm_retype_prepare(pa);
11401
11402 pvh_unlock(&locked_pvh);
11403
11404 /* XNU needs to hold an RO reference to the page despite the ownership being transferred to TXM. */
11405 pmap_enter_addr(kernel_pmap, addr, pa, VM_PROT_READ, VM_PROT_NONE, 0, true, PMAP_MAPPING_TYPE_INFER);
11406
11407 /* Finally, retype the page to TXM_DEFAULT. */
11408 _pmap_txm_transfer_page(pa);
11409 }
11410
11411 struct vm_object txm_vm_object_storage VM_PAGE_PACKED_ALIGNED;
11412 SECURITY_READ_ONLY_LATE(vm_object_t) txm_vm_object = &txm_vm_object_storage;
11413
11414 _Static_assert(sizeof(vm_map_address_t) == sizeof(pmap_paddr_t),
11415 "sizeof(vm_map_address_t) != sizeof(pmap_paddr_t)");
11416
11417 vm_map_address_t
11418 pmap_txm_allocate_page(void)
11419 {
11420 pmap_paddr_t phys_addr = 0;
11421 vm_page_t page = VM_PAGE_NULL;
11422 boolean_t thread_vm_privileged = false;
11423
11424 /* We are allowed to allocate privileged memory */
11425 thread_vm_privileged = set_vm_privilege(true);
11426
11427 /* Allocate a page from the VM free list */
11428 while ((page = vm_page_grab()) == VM_PAGE_NULL) {
11429 VM_PAGE_WAIT();
11430 }
11431
11432 /* Wire all of the pages allocated for TXM */
11433 vm_page_lock_queues();
11434 vm_page_wire(page, VM_KERN_MEMORY_SECURITY, TRUE);
11435 vm_page_unlock_queues();
11436
11437 phys_addr = (pmap_paddr_t)ptoa(VM_PAGE_GET_PHYS_PAGE(page));
11438 if (phys_addr == 0) {
11439 panic("invalid VM page allocated for TXM: %llu", phys_addr);
11440 }
11441
11442 /* Add the physical page to the TXM VM object */
11443 vm_object_lock(txm_vm_object);
11444 vm_page_insert_wired(
11445 page,
11446 txm_vm_object,
11447 phys_addr - gPhysBase,
11448 VM_KERN_MEMORY_SECURITY);
11449 vm_object_unlock(txm_vm_object);
11450
11451 /* Reset thread privilege */
11452 set_vm_privilege(thread_vm_privileged);
11453
11454 /* Retype the page */
11455 _pmap_txm_transfer_page(phys_addr);
11456
11457 return phys_addr;
11458 }
11459
11460 int
11461 pmap_cs_configuration(void)
11462 {
11463 code_signing_config_t config = 0;
11464
11465 /* Compute the code signing configuration */
11466 code_signing_configuration(NULL, &config);
11467
11468 return (int)config;
11469 }
11470
11471 bool
11472 pmap_performs_stage2_translations(
11473 __unused pmap_t pmap)
11474 {
11475 return false;
11476 }
11477
11478 bool
11479 pmap_has_iofilter_protected_write(void)
11480 {
11481 #if HAS_GUARDED_IO_FILTER
11482 return true;
11483 #else
11484 return false;
11485 #endif
11486 }
11487
11488 #if HAS_GUARDED_IO_FILTER
11489
11490 void
11491 pmap_iofilter_protected_write(__unused vm_address_t addr, __unused uint64_t value, __unused uint64_t width)
11492 {
11493 /**
11494 * Even though this is done from EL1/2 for an address potentially owned by Guarded
11495 * Mode, we should be fine as mmu_kvtop uses "at s1e1r" checking for read access
11496 * only.
11497 */
11498 const pmap_paddr_t pa = mmu_kvtop(addr);
11499
11500 if (!pa) {
11501 panic("%s: addr 0x%016llx doesn't have a valid kernel mapping", __func__, (uint64_t) addr);
11502 }
11503
11504 const sptm_frame_type_t frame_type = sptm_get_frame_type(pa);
11505 if (frame_type == XNU_PROTECTED_IO) {
11506 sptm_iofilter_protected_write(pa, value, width);
11507 } else {
11508 /* Mappings is valid but not specified by I/O filter. However, we still try
11509 * accessing the address from kernel mode. This allows addresses that are not
11510 * owned by SPTM to be accessed by this interface.
11511 */
11512 switch (width) {
11513 case 1:
11514 *(volatile uint8_t *)addr = (uint8_t) value;
11515 break;
11516 case 2:
11517 *(volatile uint16_t *)addr = (uint16_t) value;
11518 break;
11519 case 4:
11520 *(volatile uint32_t *)addr = (uint32_t) value;
11521 break;
11522 case 8:
11523 *(volatile uint64_t *)addr = (uint64_t) value;
11524 break;
11525 default:
11526 panic("%s: width %llu not supported", __func__, width);
11527 }
11528 }
11529 }
11530
11531 #else /* HAS_GUARDED_IO_FILTER */
11532
11533 __attribute__((__noreturn__))
11534 void
11535 pmap_iofilter_protected_write(__unused vm_address_t addr, __unused uint64_t value, __unused uint64_t width)
11536 {
11537 panic("%s called on an unsupported platform.", __FUNCTION__);
11538 }
11539
11540 #endif /* HAS_GUARDED_IO_FILTER */
11541
11542 void * __attribute__((noreturn))
11543 pmap_claim_reserved_ppl_page(void)
11544 {
11545 panic("%s: function not supported in this environment", __FUNCTION__);
11546 }
11547
11548 void __attribute__((noreturn))
11549 pmap_free_reserved_ppl_page(void __unused *kva)
11550 {
11551 panic("%s: function not supported in this environment", __FUNCTION__);
11552 }
11553
11554 bool
11555 pmap_lookup_in_loaded_trust_caches(__unused const uint8_t cdhash[CS_CDHASH_LEN])
11556 {
11557 kern_return_t kr = query_trust_cache(
11558 kTCQueryTypeLoadable,
11559 cdhash,
11560 NULL);
11561
11562 if (kr == KERN_SUCCESS) {
11563 return true;
11564 }
11565 return false;
11566 }
11567
11568 uint32_t
11569 pmap_lookup_in_static_trust_cache(__unused const uint8_t cdhash[CS_CDHASH_LEN])
11570 {
11571 TrustCacheQueryToken_t query_token = {0};
11572 kern_return_t kr = KERN_NOT_FOUND;
11573 uint64_t flags = 0;
11574 uint8_t hash_type = 0;
11575
11576 kr = query_trust_cache(
11577 kTCQueryTypeStatic,
11578 cdhash,
11579 &query_token);
11580
11581 if (kr == KERN_SUCCESS) {
11582 amfi->TrustCache.queryGetFlags(&query_token, &flags);
11583 amfi->TrustCache.queryGetHashType(&query_token, &hash_type);
11584
11585 return (TC_LOOKUP_FOUND << TC_LOOKUP_RESULT_SHIFT) |
11586 (hash_type << TC_LOOKUP_HASH_TYPE_SHIFT) |
11587 ((uint8_t)flags << TC_LOOKUP_FLAGS_SHIFT);
11588 }
11589
11590 return 0;
11591 }
11592
11593 #if DEVELOPMENT || DEBUG
11594
11595 struct page_table_dump_header {
11596 uint64_t pa;
11597 uint64_t num_entries;
11598 uint64_t start_va;
11599 uint64_t end_va;
11600 };
11601
11602 static kern_return_t
11603 pmap_dump_page_tables_recurse(pmap_t pmap,
11604 const tt_entry_t *ttp,
11605 unsigned int cur_level,
11606 unsigned int level_mask,
11607 uint64_t start_va,
11608 void *buf_start,
11609 void *buf_end,
11610 size_t *bytes_copied)
11611 {
11612 const pt_attr_t * const pt_attr = pmap_get_pt_attr(pmap);
11613 uint64_t num_entries = pt_attr_page_size(pt_attr) / sizeof(*ttp);
11614
11615 uint64_t size = pt_attr->pta_level_info[cur_level].size;
11616 uint64_t valid_mask = pt_attr->pta_level_info[cur_level].valid_mask;
11617 uint64_t type_mask = pt_attr->pta_level_info[cur_level].type_mask;
11618 uint64_t type_block = pt_attr->pta_level_info[cur_level].type_block;
11619
11620 void *bufp = (uint8_t*)buf_start + *bytes_copied;
11621
11622 if (cur_level == pt_attr_root_level(pt_attr)) {
11623 start_va &= ~(pt_attr->pta_level_info[cur_level].offmask);
11624 num_entries = pmap_root_alloc_size(pmap) / sizeof(tt_entry_t);
11625 }
11626
11627 uint64_t tt_size = num_entries * sizeof(tt_entry_t);
11628 const tt_entry_t *tt_end = &ttp[num_entries];
11629
11630 if (((vm_offset_t)buf_end - (vm_offset_t)bufp) < (tt_size + sizeof(struct page_table_dump_header))) {
11631 return KERN_INSUFFICIENT_BUFFER_SIZE;
11632 }
11633
11634 if (level_mask & (1U << cur_level)) {
11635 struct page_table_dump_header *header = (struct page_table_dump_header*)bufp;
11636 header->pa = kvtophys_nofail((vm_offset_t)ttp);
11637 header->num_entries = num_entries;
11638 header->start_va = start_va;
11639 header->end_va = start_va + (num_entries * size);
11640
11641 bcopy(ttp, (uint8_t*)bufp + sizeof(*header), tt_size);
11642 *bytes_copied = *bytes_copied + sizeof(*header) + tt_size;
11643 }
11644 uint64_t current_va = start_va;
11645
11646 for (const tt_entry_t *ttep = ttp; ttep < tt_end; ttep++, current_va += size) {
11647 tt_entry_t tte = *ttep;
11648
11649 if (!(tte & valid_mask)) {
11650 continue;
11651 }
11652
11653 if ((tte & type_mask) == type_block) {
11654 continue;
11655 } else {
11656 if (cur_level >= pt_attr_leaf_level(pt_attr)) {
11657 panic("%s: corrupt entry %#llx at %p, "
11658 "ttp=%p, cur_level=%u, bufp=%p, buf_end=%p",
11659 __FUNCTION__, tte, ttep,
11660 ttp, cur_level, bufp, buf_end);
11661 }
11662
11663 const tt_entry_t *next_tt = (const tt_entry_t*)phystokv(tte & ARM_TTE_TABLE_MASK);
11664
11665 kern_return_t recurse_result = pmap_dump_page_tables_recurse(pmap, next_tt, cur_level + 1,
11666 level_mask, current_va, buf_start, buf_end, bytes_copied);
11667
11668 if (recurse_result != KERN_SUCCESS) {
11669 return recurse_result;
11670 }
11671 }
11672 }
11673
11674 return KERN_SUCCESS;
11675 }
11676
11677 kern_return_t
11678 pmap_dump_page_tables(pmap_t pmap, void *bufp, void *buf_end, unsigned int level_mask, size_t *bytes_copied)
11679 {
11680 if (not_in_kdp) {
11681 panic("pmap_dump_page_tables must only be called from kernel debugger context");
11682 }
11683 return pmap_dump_page_tables_recurse(pmap, pmap->tte, pt_attr_root_level(pmap_get_pt_attr(pmap)),
11684 level_mask, pmap->min, bufp, buf_end, bytes_copied);
11685 }
11686
11687 #else /* DEVELOPMENT || DEBUG */
11688
11689 kern_return_t
11690 pmap_dump_page_tables(pmap_t pmap __unused, void *bufp __unused, void *buf_end __unused,
11691 unsigned int level_mask __unused, size_t *bytes_copied __unused)
11692 {
11693 return KERN_NOT_SUPPORTED;
11694 }
11695 #endif /* !(DEVELOPMENT || DEBUG) */
11696
11697
11698 #ifdef CONFIG_XNUPOST
11699 static volatile bool pmap_test_took_fault = false;
11700
11701 static bool
11702 pmap_test_fault_handler(arm_saved_state_t * state)
11703 {
11704 bool retval = false;
11705 uint64_t esr = get_saved_state_esr(state);
11706 esr_exception_class_t class = ESR_EC(esr);
11707 fault_status_t fsc = ISS_IA_FSC(ESR_ISS(esr));
11708
11709 if ((class == ESR_EC_DABORT_EL1) &&
11710 ((fsc == FSC_PERMISSION_FAULT_L3) || (fsc == FSC_ACCESS_FLAG_FAULT_L3))) {
11711 pmap_test_took_fault = true;
11712 /* return to the instruction immediately after the call to NX page */
11713 set_saved_state_pc(state, get_saved_state_pc(state) + 4);
11714 retval = true;
11715 }
11716
11717 return retval;
11718 }
11719
11720 // Disable KASAN instrumentation, as the test pmap's TTBR0 space will not be in the shadow map
11721 static NOKASAN bool
11722 pmap_test_access(pmap_t pmap, vm_map_address_t va, bool should_fault, bool is_write)
11723 {
11724 pmap_t old_pmap = NULL;
11725
11726 pmap_test_took_fault = false;
11727
11728 /*
11729 * We're potentially switching pmaps without using the normal thread
11730 * mechanism; disable interrupts and preemption to avoid any unexpected
11731 * memory accesses.
11732 */
11733 const boolean_t old_int_state = ml_set_interrupts_enabled(FALSE);
11734 mp_disable_preemption();
11735
11736 if (pmap != NULL) {
11737 old_pmap = current_pmap();
11738 pmap_switch(pmap);
11739
11740 /* Disable PAN; pmap shouldn't be the kernel pmap. */
11741 #if __ARM_PAN_AVAILABLE__
11742 __builtin_arm_wsr("pan", 0);
11743 #endif /* __ARM_PAN_AVAILABLE__ */
11744 }
11745
11746 ml_expect_fault_begin(pmap_test_fault_handler, va);
11747
11748 if (is_write) {
11749 *((volatile uint64_t*)(va)) = 0xdec0de;
11750 } else {
11751 volatile uint64_t tmp = *((volatile uint64_t*)(va));
11752 (void)tmp;
11753 }
11754
11755 /* Save the fault bool, and undo the gross stuff we did. */
11756 bool took_fault = pmap_test_took_fault;
11757 ml_expect_fault_end();
11758
11759 if (pmap != NULL) {
11760 #if __ARM_PAN_AVAILABLE__
11761 __builtin_arm_wsr("pan", 1);
11762 #endif /* __ARM_PAN_AVAILABLE__ */
11763
11764 pmap_switch(old_pmap);
11765 }
11766
11767 mp_enable_preemption();
11768 ml_set_interrupts_enabled(old_int_state);
11769 bool retval = (took_fault == should_fault);
11770 return retval;
11771 }
11772
11773 static bool
11774 pmap_test_read(pmap_t pmap, vm_map_address_t va, bool should_fault)
11775 {
11776 bool retval = pmap_test_access(pmap, va, should_fault, false);
11777
11778 if (!retval) {
11779 T_FAIL("%s: %s, "
11780 "pmap=%p, va=%p, should_fault=%u",
11781 __func__, should_fault ? "did not fault" : "faulted",
11782 pmap, (void*)va, (unsigned)should_fault);
11783 }
11784
11785 return retval;
11786 }
11787
11788 static bool
11789 pmap_test_write(pmap_t pmap, vm_map_address_t va, bool should_fault)
11790 {
11791 bool retval = pmap_test_access(pmap, va, should_fault, true);
11792
11793 if (!retval) {
11794 T_FAIL("%s: %s, "
11795 "pmap=%p, va=%p, should_fault=%u",
11796 __func__, should_fault ? "did not fault" : "faulted",
11797 pmap, (void*)va, (unsigned)should_fault);
11798 }
11799
11800 return retval;
11801 }
11802
11803 static bool
11804 pmap_test_check_refmod(pmap_paddr_t pa, unsigned int should_be_set)
11805 {
11806 unsigned int should_be_clear = (~should_be_set) & (VM_MEM_REFERENCED | VM_MEM_MODIFIED);
11807 unsigned int bits = pmap_get_refmod((ppnum_t)atop(pa));
11808
11809 bool retval = (((bits & should_be_set) == should_be_set) && ((bits & should_be_clear) == 0));
11810
11811 if (!retval) {
11812 T_FAIL("%s: bits=%u, "
11813 "pa=%p, should_be_set=%u",
11814 __func__, bits,
11815 (void*)pa, should_be_set);
11816 }
11817
11818 return retval;
11819 }
11820
11821 static __attribute__((noinline)) bool
11822 pmap_test_read_write(pmap_t pmap, vm_map_address_t va, bool allow_read, bool allow_write)
11823 {
11824 bool retval = (pmap_test_read(pmap, va, !allow_read) | pmap_test_write(pmap, va, !allow_write));
11825 return retval;
11826 }
11827
11828 static int
11829 pmap_test_test_config(unsigned int flags)
11830 {
11831 T_LOG("running pmap_test_test_config flags=0x%X", flags);
11832 unsigned int map_count = 0;
11833 unsigned long page_ratio = 0;
11834 pmap_t pmap = pmap_create_options(NULL, 0, flags);
11835
11836 if (!pmap) {
11837 panic("Failed to allocate pmap");
11838 }
11839
11840 __unused const pt_attr_t * const pt_attr = pmap_get_pt_attr(pmap);
11841 uintptr_t native_page_size = pt_attr_page_size(native_pt_attr);
11842 uintptr_t pmap_page_size = pt_attr_page_size(pt_attr);
11843 uintptr_t pmap_twig_size = pt_attr_twig_size(pt_attr);
11844
11845 if (pmap_page_size <= native_page_size) {
11846 page_ratio = native_page_size / pmap_page_size;
11847 } else {
11848 /*
11849 * We claim to support a page_ratio of less than 1, which is
11850 * not currently supported by the pmap layer; panic.
11851 */
11852 panic("%s: page_ratio < 1, native_page_size=%lu, pmap_page_size=%lu"
11853 "flags=%u",
11854 __func__, native_page_size, pmap_page_size,
11855 flags);
11856 }
11857
11858 if (PAGE_RATIO > 1) {
11859 /*
11860 * The kernel is deliberately pretending to have 16KB pages.
11861 * The pmap layer has code that supports this, so pretend the
11862 * page size is larger than it is.
11863 */
11864 pmap_page_size = PAGE_SIZE;
11865 native_page_size = PAGE_SIZE;
11866 }
11867
11868 /*
11869 * Get two pages from the VM; one to be mapped wired, and one to be
11870 * mapped nonwired.
11871 */
11872 vm_page_t unwired_vm_page = vm_page_grab();
11873 vm_page_t wired_vm_page = vm_page_grab();
11874
11875 if ((unwired_vm_page == VM_PAGE_NULL) || (wired_vm_page == VM_PAGE_NULL)) {
11876 panic("Failed to grab VM pages");
11877 }
11878
11879 ppnum_t pn = VM_PAGE_GET_PHYS_PAGE(unwired_vm_page);
11880 ppnum_t wired_pn = VM_PAGE_GET_PHYS_PAGE(wired_vm_page);
11881
11882 pmap_paddr_t pa = ptoa(pn);
11883 pmap_paddr_t wired_pa = ptoa(wired_pn);
11884
11885 /*
11886 * We'll start mappings at the second twig TT. This keeps us from only
11887 * using the first entry in each TT, which would trivially be address
11888 * 0; one of the things we will need to test is retrieving the VA for
11889 * a given PTE.
11890 */
11891 vm_map_address_t va_base = pmap_twig_size;
11892 vm_map_address_t wired_va_base = ((2 * pmap_twig_size) - pmap_page_size);
11893
11894 if (wired_va_base < (va_base + (page_ratio * pmap_page_size))) {
11895 /*
11896 * Not exactly a functional failure, but this test relies on
11897 * there being a spare PTE slot we can use to pin the TT.
11898 */
11899 panic("Cannot pin translation table");
11900 }
11901
11902 /*
11903 * Create the wired mapping; this will prevent the pmap layer from
11904 * reclaiming our test TTs, which would interfere with this test
11905 * ("interfere" -> "make it panic").
11906 */
11907 pmap_enter_addr(pmap, wired_va_base, wired_pa, VM_PROT_READ, VM_PROT_READ, 0, true, PMAP_MAPPING_TYPE_INFER);
11908
11909 T_LOG("Validate that kernel cannot write to SPTM memory.");
11910 pt_entry_t * ptep = pmap_pte(pmap, va_base);
11911 pmap_test_write(NULL, (vm_map_address_t)ptep, true);
11912
11913 /*
11914 * Create read-only mappings of the nonwired page; if the pmap does
11915 * not use the same page size as the kernel, create multiple mappings
11916 * so that the kernel page is fully mapped.
11917 */
11918 for (map_count = 0; map_count < page_ratio; map_count++) {
11919 pmap_enter_addr(pmap, va_base + (pmap_page_size * map_count), pa + (pmap_page_size * (map_count)),
11920 VM_PROT_READ, VM_PROT_READ, 0, false, PMAP_MAPPING_TYPE_INFER);
11921 }
11922
11923 /* Validate that all the PTEs have the expected PA and VA. */
11924 for (map_count = 0; map_count < page_ratio; map_count++) {
11925 ptep = pmap_pte(pmap, va_base + (pmap_page_size * map_count));
11926
11927 if (pte_to_pa(*ptep) != (pa + (pmap_page_size * map_count))) {
11928 T_FAIL("Unexpected pa=%p, expected %p, map_count=%u",
11929 (void*)pte_to_pa(*ptep), (void*)(pa + (pmap_page_size * map_count)), map_count);
11930 }
11931
11932 if (ptep_get_va(ptep) != (va_base + (pmap_page_size * map_count))) {
11933 T_FAIL("Unexpected va=%p, expected %p, map_count=%u",
11934 (void*)ptep_get_va(ptep), (void*)(va_base + (pmap_page_size * map_count)), map_count);
11935 }
11936 }
11937
11938 T_LOG("Validate that reads to our mapping do not fault.");
11939 pmap_test_read(pmap, va_base, false);
11940
11941 T_LOG("Validate that writes to our mapping fault.");
11942 pmap_test_write(pmap, va_base, true);
11943
11944 T_LOG("Make the first mapping writable.");
11945 pmap_enter_addr(pmap, va_base, pa, VM_PROT_READ | VM_PROT_WRITE, VM_PROT_READ | VM_PROT_WRITE, 0, false, PMAP_MAPPING_TYPE_INFER);
11946
11947 T_LOG("Validate that writes to our mapping do not fault.");
11948 pmap_test_write(pmap, va_base, false);
11949
11950 /*
11951 * For page ratios of greater than 1: validate that writes to the other
11952 * mappings still fault. Remove the mappings afterwards (we're done
11953 * with page ratio testing).
11954 */
11955 for (map_count = 1; map_count < page_ratio; map_count++) {
11956 pmap_test_write(pmap, va_base + (pmap_page_size * map_count), true);
11957 pmap_remove(pmap, va_base + (pmap_page_size * map_count), va_base + (pmap_page_size * map_count) + pmap_page_size);
11958 }
11959
11960 /* Remove remaining mapping */
11961 pmap_remove(pmap, va_base, va_base + pmap_page_size);
11962
11963 T_LOG("Make the first mapping execute-only");
11964 pmap_enter_addr(pmap, va_base, pa, VM_PROT_EXECUTE, VM_PROT_EXECUTE, 0, false, PMAP_MAPPING_TYPE_INFER);
11965
11966
11967 T_LOG("Validate that reads to our mapping do not fault.");
11968 pmap_test_read(pmap, va_base, false);
11969
11970 T_LOG("Validate that reads to our mapping do not fault.");
11971 pmap_test_read(pmap, va_base, false);
11972
11973 T_LOG("Validate that writes to our mapping fault.");
11974 pmap_test_write(pmap, va_base, true);
11975
11976 pmap_remove(pmap, va_base, va_base + pmap_page_size);
11977
11978 T_LOG("Mark the page unreferenced and unmodified.");
11979 pmap_clear_refmod(pn, VM_MEM_MODIFIED | VM_MEM_REFERENCED);
11980 pmap_test_check_refmod(pa, 0);
11981
11982 /*
11983 * Begin testing the ref/mod state machine. Re-enter the mapping with
11984 * different protection/fault_type settings, and confirm that the
11985 * ref/mod state matches our expectations at each step.
11986 */
11987 T_LOG("!ref/!mod: read, no fault. Expect ref/!mod");
11988 pmap_enter_addr(pmap, va_base, pa, VM_PROT_READ, VM_PROT_NONE, 0, false, PMAP_MAPPING_TYPE_INFER);
11989 pmap_test_check_refmod(pa, VM_MEM_REFERENCED);
11990
11991 T_LOG("!ref/!mod: read, read fault. Expect ref/!mod");
11992 pmap_clear_refmod(pn, VM_MEM_MODIFIED | VM_MEM_REFERENCED);
11993 pmap_enter_addr(pmap, va_base, pa, VM_PROT_READ, VM_PROT_READ, 0, false, PMAP_MAPPING_TYPE_INFER);
11994 pmap_test_check_refmod(pa, VM_MEM_REFERENCED);
11995
11996 T_LOG("!ref/!mod: rw, read fault. Expect ref/!mod");
11997 pmap_clear_refmod(pn, VM_MEM_MODIFIED | VM_MEM_REFERENCED);
11998 pmap_enter_addr(pmap, va_base, pa, VM_PROT_READ | VM_PROT_WRITE, VM_PROT_NONE, 0, false, PMAP_MAPPING_TYPE_INFER);
11999 pmap_test_check_refmod(pa, VM_MEM_REFERENCED);
12000
12001 T_LOG("ref/!mod: rw, read fault. Expect ref/!mod");
12002 pmap_enter_addr(pmap, va_base, pa, VM_PROT_READ | VM_PROT_WRITE, VM_PROT_READ, 0, false, PMAP_MAPPING_TYPE_INFER);
12003 pmap_test_check_refmod(pa, VM_MEM_REFERENCED);
12004
12005 T_LOG("!ref/!mod: rw, rw fault. Expect ref/mod");
12006 pmap_clear_refmod(pn, VM_MEM_MODIFIED | VM_MEM_REFERENCED);
12007 pmap_enter_addr(pmap, va_base, pa, VM_PROT_READ | VM_PROT_WRITE, VM_PROT_READ | VM_PROT_WRITE, 0, false, PMAP_MAPPING_TYPE_INFER);
12008 pmap_test_check_refmod(pa, VM_MEM_REFERENCED | VM_MEM_MODIFIED);
12009
12010 /*
12011 * Shared memory testing; we'll have two mappings; one read-only,
12012 * one read-write.
12013 */
12014 vm_map_address_t rw_base = va_base;
12015 vm_map_address_t ro_base = va_base + pmap_page_size;
12016
12017 pmap_enter_addr(pmap, rw_base, pa, VM_PROT_READ | VM_PROT_WRITE, VM_PROT_READ | VM_PROT_WRITE, 0, false, PMAP_MAPPING_TYPE_INFER);
12018 pmap_enter_addr(pmap, ro_base, pa, VM_PROT_READ, VM_PROT_READ, 0, false, PMAP_MAPPING_TYPE_INFER);
12019
12020 /*
12021 * Test that we take faults as expected for unreferenced/unmodified
12022 * pages. Also test the arm_fast_fault interface, to ensure that
12023 * mapping permissions change as expected.
12024 */
12025 T_LOG("!ref/!mod: expect no access");
12026 pmap_clear_refmod(pn, VM_MEM_MODIFIED | VM_MEM_REFERENCED);
12027 pmap_test_read_write(pmap, ro_base, false, false);
12028 pmap_test_read_write(pmap, rw_base, false, false);
12029
12030 T_LOG("Read fault; expect !ref/!mod -> ref/!mod, read access");
12031 arm_fast_fault(pmap, rw_base, VM_PROT_READ, false, false);
12032 pmap_test_check_refmod(pa, VM_MEM_REFERENCED);
12033 pmap_test_read_write(pmap, ro_base, true, false);
12034 pmap_test_read_write(pmap, rw_base, true, false);
12035
12036 T_LOG("Write fault; expect ref/!mod -> ref/mod, read and write access");
12037 arm_fast_fault(pmap, rw_base, VM_PROT_READ | VM_PROT_WRITE, false, false);
12038 pmap_test_check_refmod(pa, VM_MEM_REFERENCED | VM_MEM_MODIFIED);
12039 pmap_test_read_write(pmap, ro_base, true, false);
12040 pmap_test_read_write(pmap, rw_base, true, true);
12041
12042 T_LOG("Write fault; expect !ref/!mod -> ref/mod, read and write access");
12043 pmap_clear_refmod(pn, VM_MEM_MODIFIED | VM_MEM_REFERENCED);
12044 arm_fast_fault(pmap, rw_base, VM_PROT_READ | VM_PROT_WRITE, false, false);
12045 pmap_test_check_refmod(pa, VM_MEM_REFERENCED | VM_MEM_MODIFIED);
12046 pmap_test_read_write(pmap, ro_base, true, false);
12047 pmap_test_read_write(pmap, rw_base, true, true);
12048
12049 T_LOG("RW protect both mappings; should not change protections.");
12050 pmap_protect(pmap, ro_base, ro_base + pmap_page_size, VM_PROT_READ | VM_PROT_WRITE);
12051 pmap_protect(pmap, rw_base, rw_base + pmap_page_size, VM_PROT_READ | VM_PROT_WRITE);
12052 pmap_test_read_write(pmap, ro_base, true, false);
12053 pmap_test_read_write(pmap, rw_base, true, true);
12054
12055 T_LOG("Read protect both mappings; RW mapping should become RO.");
12056 pmap_protect(pmap, ro_base, ro_base + pmap_page_size, VM_PROT_READ);
12057 pmap_protect(pmap, rw_base, rw_base + pmap_page_size, VM_PROT_READ);
12058 pmap_test_read_write(pmap, ro_base, true, false);
12059 pmap_test_read_write(pmap, rw_base, true, false);
12060
12061 T_LOG("RW protect the page; mappings should not change protections.");
12062 pmap_enter_addr(pmap, rw_base, pa, VM_PROT_READ | VM_PROT_WRITE, VM_PROT_READ | VM_PROT_WRITE, 0, false, PMAP_MAPPING_TYPE_INFER);
12063 pmap_page_protect(pn, VM_PROT_ALL);
12064 pmap_test_read_write(pmap, ro_base, true, false);
12065 pmap_test_read_write(pmap, rw_base, true, true);
12066
12067 T_LOG("Read protect the page; RW mapping should become RO.");
12068 pmap_page_protect(pn, VM_PROT_READ);
12069 pmap_test_read_write(pmap, ro_base, true, false);
12070 pmap_test_read_write(pmap, rw_base, true, false);
12071
12072 T_LOG("Validate that disconnect removes all known mappings of the page.");
12073 pmap_disconnect(pn);
12074 if (!pmap_verify_free(pn)) {
12075 T_FAIL("Page still has mappings");
12076 }
12077
12078 #if defined(ARM_LARGE_MEMORY)
12079 #define PMAP_TEST_LARGE_MEMORY_VA 64 * (1ULL << 40) /* 64 TB */
12080
12081 T_LOG("Create new wired mapping in the extended address space enabled by ARM_LARGE_MEMORY.");
12082 pmap_enter_addr(pmap, PMAP_TEST_LARGE_MEMORY_VA, wired_pa, VM_PROT_READ | VM_PROT_WRITE, VM_PROT_READ | VM_PROT_WRITE, 0, true, PMAP_MAPPING_TYPE_INFER);
12083 pmap_test_read_write(pmap, PMAP_TEST_LARGE_MEMORY_VA, true, true);
12084 pmap_remove(pmap, PMAP_TEST_LARGE_MEMORY_VA, PMAP_TEST_LARGE_MEMORY_VA + pmap_page_size);
12085 #endif /* ARM_LARGE_MEMORY */
12086
12087 T_LOG("Remove the wired mapping, so we can tear down the test map.");
12088 pmap_remove(pmap, wired_va_base, wired_va_base + pmap_page_size);
12089 pmap_destroy(pmap);
12090
12091 T_LOG("Release the pages back to the VM.");
12092 vm_page_lock_queues();
12093 vm_page_free(unwired_vm_page);
12094 vm_page_free(wired_vm_page);
12095 vm_page_unlock_queues();
12096
12097 T_LOG("Testing successful!");
12098 return 0;
12099 }
12100
12101 kern_return_t
12102 pmap_test(void)
12103 {
12104 T_LOG("Starting pmap_tests");
12105 int flags = 0;
12106 flags |= PMAP_CREATE_64BIT;
12107
12108 #if __ARM_MIXED_PAGE_SIZE__ && !CONFIG_SPTM
12109 T_LOG("Testing VM_PAGE_SIZE_4KB");
12110 pmap_test_test_config(flags | PMAP_CREATE_FORCE_4K_PAGES);
12111 T_LOG("Testing VM_PAGE_SIZE_16KB");
12112 pmap_test_test_config(flags);
12113 #else /* __ARM_MIXED_PAGE_SIZE__ */
12114 pmap_test_test_config(flags);
12115 #endif /* __ARM_MIXED_PAGE_SIZE__ */
12116
12117 T_PASS("completed pmap_test successfully");
12118 return KERN_SUCCESS;
12119 }
12120 #endif /* CONFIG_XNUPOST */
12121
12122 /*
12123 * The following function should never make it to RELEASE code, since
12124 * it provides a way to get the PPL to modify text pages.
12125 */
12126 #if DEVELOPMENT || DEBUG
12127
12128 /**
12129 * Forcibly overwrite executable text with an illegal instruction.
12130 *
12131 * @note Only used for xnu unit testing.
12132 *
12133 * @param pa The physical address to corrupt.
12134 *
12135 * @return KERN_SUCCESS on success.
12136 */
12137 kern_return_t
12138 pmap_test_text_corruption(pmap_paddr_t pa __unused)
12139 {
12140 /*
12141 * SPTM TODO: implement an SPTM version of this.
12142 * The physical apertue is owned by the SPTM and text
12143 * pages have RO physical aperture mappings.
12144 */
12145 return KERN_SUCCESS;
12146 }
12147
12148 #endif /* DEVELOPMENT || DEBUG */
12149
12150