xref: /xnu-11215/osfmk/arm64/sptm/pmap/pmap.c (revision 4f1223e8)
1 /*
2  * Copyright (c) 2011-2022 Apple Inc. All rights reserved.
3  *
4  * @APPLE_OSREFERENCE_LICENSE_HEADER_START@
5  *
6  * This file contains Original Code and/or Modifications of Original Code
7  * as defined in and that are subject to the Apple Public Source License
8  * Version 2.0 (the 'License'). You may not use this file except in
9  * compliance with the License. The rights granted to you under the License
10  * may not be used to create, or enable the creation or redistribution of,
11  * unlawful or unlicensed copies of an Apple operating system, or to
12  * circumvent, violate, or enable the circumvention or violation of, any
13  * terms of an Apple operating system software license agreement.
14  *
15  * Please obtain a copy of the License at
16  * http://www.opensource.apple.com/apsl/ and read it before using this file.
17  *
18  * The Original Code and all software distributed under the License are
19  * distributed on an 'AS IS' basis, WITHOUT WARRANTY OF ANY KIND, EITHER
20  * EXPRESS OR IMPLIED, AND APPLE HEREBY DISCLAIMS ALL SUCH WARRANTIES,
21  * INCLUDING WITHOUT LIMITATION, ANY WARRANTIES OF MERCHANTABILITY,
22  * FITNESS FOR A PARTICULAR PURPOSE, QUIET ENJOYMENT OR NON-INFRINGEMENT.
23  * Please see the License for the specific language governing rights and
24  * limitations under the License.
25  *
26  * @APPLE_OSREFERENCE_LICENSE_HEADER_END@
27  */
28 #include <string.h>
29 #include <stdlib.h>
30 #include <mach_assert.h>
31 #include <mach_ldebug.h>
32 
33 #include <mach/shared_region.h>
34 #include <mach/vm_param.h>
35 #include <mach/vm_prot.h>
36 #include <mach/vm_map.h>
37 #include <mach/machine/vm_param.h>
38 #include <mach/machine/vm_types.h>
39 
40 #include <mach/boolean.h>
41 #include <kern/bits.h>
42 #include <kern/ecc.h>
43 #include <kern/thread.h>
44 #include <kern/sched.h>
45 #include <kern/zalloc.h>
46 #include <kern/zalloc_internal.h>
47 #include <kern/kalloc.h>
48 #include <kern/spl.h>
49 #include <kern/startup.h>
50 #include <kern/trustcache.h>
51 
52 #include <os/overflow.h>
53 
54 #include <vm/pmap.h>
55 #include <vm/pmap_cs.h>
56 #include <vm/vm_map_xnu.h>
57 #include <vm/vm_kern.h>
58 #include <vm/vm_protos.h>
59 #include <vm/vm_object_internal.h>
60 #include <vm/vm_page_internal.h>
61 #include <vm/vm_pageout.h>
62 #include <vm/cpm_internal.h>
63 
64 
65 #include <libkern/section_keywords.h>
66 #include <sys/errno.h>
67 
68 #include <libkern/amfi/amfi.h>
69 #include <sys/trusted_execution_monitor.h>
70 #include <sys/trust_caches.h>
71 #include <sys/code_signing.h>
72 
73 #include <machine/atomic.h>
74 #include <machine/thread.h>
75 #include <machine/lowglobals.h>
76 
77 #include <arm/caches_internal.h>
78 #include <arm/cpu_data.h>
79 #include <arm/cpu_data_internal.h>
80 #include <arm/cpu_capabilities.h>
81 #include <arm/cpu_number.h>
82 #include <arm/machine_cpu.h>
83 #include <arm/misc_protos.h>
84 #include <arm/trap_internal.h>
85 #include <arm64/sptm/pmap/pmap_internal.h>
86 #include <arm64/sptm/sptm.h>
87 
88 #include <arm64/proc_reg.h>
89 #include <pexpert/arm64/boot.h>
90 #include <arm64/ppl/uat.h>
91 #if defined(KERNEL_INTEGRITY_KTRR) || defined(KERNEL_INTEGRITY_CTRR)
92 #include <arm64/amcc_rorgn.h>
93 #endif // defined(KERNEL_INTEGRITY_KTRR) || defined(KERNEL_INTEGRITY_CTRR)
94 
95 #include <pexpert/device_tree.h>
96 
97 #include <san/kasan.h>
98 #include <sys/cdefs.h>
99 
100 #if defined(HAS_APPLE_PAC)
101 #include <ptrauth.h>
102 #endif
103 
104 #ifdef CONFIG_XNUPOST
105 #include <tests/xnupost.h>
106 #endif
107 
108 
109 #if HIBERNATION
110 #include <IOKit/IOHibernatePrivate.h>
111 #endif /* HIBERNATION */
112 
113 #define PMAP_ROOT_ALLOC_SIZE (ARM_PGBYTES)
114 
115 #define ARRAY_LEN(x) (sizeof (x) / sizeof (x[0]))
116 
117 
118 /**
119  * Per-CPU data used to do setup and post-processing for SPTM calls.
120  * On the setup side, this structure is used to store parameters for batched SPTM operations.
121  * These parameters may be large (upwards of 1K), and given that SPTM calls are generally
122  * issued from preemption-disabled contexts anyway, it's better to store them in per-CPU
123  * data rather than the local stack.
124  * On the post-processing side, this structure exposes a pointer to the SPTM's per-CPU array
125  * of 'prev_ptes', that is the prior value encountered in each PTE at the time of the SPTM's
126  * atomic update of that PTE.
127  */
128 pmap_sptm_percpu_data_t PERCPU_DATA(pmap_sptm_percpu);
129 
130 /**
131  * Reference group for global tracking of all outstanding pmap references.
132  */
133 os_refgrp_decl(static, pmap_refgrp, "pmap", NULL);
134 
135 /* Boot-arg to enable/disable the use of XNU_KERNEL_RESTRICTED type in SPTM. */
136 TUNABLE(bool, use_xnu_restricted, "xnu_restricted", true);
137 
138 extern u_int32_t random(void); /* from <libkern/libkern.h> */
139 
140 static bool alloc_asid(pmap_t pmap);
141 static void free_asid(pmap_t pmap);
142 static void flush_mmu_tlb_region_asid_async(vm_offset_t va, size_t length, pmap_t pmap, bool last_level_only);
143 static pt_entry_t wimg_to_pte(unsigned int wimg, pmap_paddr_t pa);
144 
145 const struct page_table_ops native_pt_ops =
146 {
147 	.alloc_id = alloc_asid,
148 	.free_id = free_asid,
149 	.flush_tlb_region_async = flush_mmu_tlb_region_asid_async,
150 	.wimg_to_pte = wimg_to_pte,
151 };
152 
153 const struct page_table_level_info pmap_table_level_info_16k[] =
154 {
155 	[0] = {
156 		.size       = ARM_16K_TT_L0_SIZE,
157 		.offmask    = ARM_16K_TT_L0_OFFMASK,
158 		.shift      = ARM_16K_TT_L0_SHIFT,
159 		.index_mask = ARM_16K_TT_L0_INDEX_MASK,
160 		.valid_mask = ARM_TTE_VALID,
161 		.type_mask  = ARM_TTE_TYPE_MASK,
162 		.type_block = ARM_TTE_TYPE_BLOCK
163 	},
164 	[1] = {
165 		.size       = ARM_16K_TT_L1_SIZE,
166 		.offmask    = ARM_16K_TT_L1_OFFMASK,
167 		.shift      = ARM_16K_TT_L1_SHIFT,
168 		.index_mask = ARM_16K_TT_L1_INDEX_MASK,
169 		.valid_mask = ARM_TTE_VALID,
170 		.type_mask  = ARM_TTE_TYPE_MASK,
171 		.type_block = ARM_TTE_TYPE_BLOCK
172 	},
173 	[2] = {
174 		.size       = ARM_16K_TT_L2_SIZE,
175 		.offmask    = ARM_16K_TT_L2_OFFMASK,
176 		.shift      = ARM_16K_TT_L2_SHIFT,
177 		.index_mask = ARM_16K_TT_L2_INDEX_MASK,
178 		.valid_mask = ARM_TTE_VALID,
179 		.type_mask  = ARM_TTE_TYPE_MASK,
180 		.type_block = ARM_TTE_TYPE_BLOCK
181 	},
182 	[3] = {
183 		.size       = ARM_16K_TT_L3_SIZE,
184 		.offmask    = ARM_16K_TT_L3_OFFMASK,
185 		.shift      = ARM_16K_TT_L3_SHIFT,
186 		.index_mask = ARM_16K_TT_L3_INDEX_MASK,
187 		.valid_mask = ARM_PTE_TYPE_VALID,
188 		.type_mask  = ARM_PTE_TYPE_MASK,
189 		.type_block = ARM_TTE_TYPE_L3BLOCK
190 	}
191 };
192 
193 const struct page_table_level_info pmap_table_level_info_4k[] =
194 {
195 	[0] = {
196 		.size       = ARM_4K_TT_L0_SIZE,
197 		.offmask    = ARM_4K_TT_L0_OFFMASK,
198 		.shift      = ARM_4K_TT_L0_SHIFT,
199 		.index_mask = ARM_4K_TT_L0_INDEX_MASK,
200 		.valid_mask = ARM_TTE_VALID,
201 		.type_mask  = ARM_TTE_TYPE_MASK,
202 		.type_block = ARM_TTE_TYPE_BLOCK
203 	},
204 	[1] = {
205 		.size       = ARM_4K_TT_L1_SIZE,
206 		.offmask    = ARM_4K_TT_L1_OFFMASK,
207 		.shift      = ARM_4K_TT_L1_SHIFT,
208 		.index_mask = ARM_4K_TT_L1_INDEX_MASK,
209 		.valid_mask = ARM_TTE_VALID,
210 		.type_mask  = ARM_TTE_TYPE_MASK,
211 		.type_block = ARM_TTE_TYPE_BLOCK
212 	},
213 	[2] = {
214 		.size       = ARM_4K_TT_L2_SIZE,
215 		.offmask    = ARM_4K_TT_L2_OFFMASK,
216 		.shift      = ARM_4K_TT_L2_SHIFT,
217 		.index_mask = ARM_4K_TT_L2_INDEX_MASK,
218 		.valid_mask = ARM_TTE_VALID,
219 		.type_mask  = ARM_TTE_TYPE_MASK,
220 		.type_block = ARM_TTE_TYPE_BLOCK
221 	},
222 	[3] = {
223 		.size       = ARM_4K_TT_L3_SIZE,
224 		.offmask    = ARM_4K_TT_L3_OFFMASK,
225 		.shift      = ARM_4K_TT_L3_SHIFT,
226 		.index_mask = ARM_4K_TT_L3_INDEX_MASK,
227 		.valid_mask = ARM_PTE_TYPE_VALID,
228 		.type_mask  = ARM_PTE_TYPE_MASK,
229 		.type_block = ARM_TTE_TYPE_L3BLOCK
230 	}
231 };
232 
233 const struct page_table_level_info pmap_table_level_info_4k_stage2[] =
234 {
235 	[0] = { /* Unused */
236 		.size       = ARM_4K_TT_L0_SIZE,
237 		.offmask    = ARM_4K_TT_L0_OFFMASK,
238 		.shift      = ARM_4K_TT_L0_SHIFT,
239 		.index_mask = ARM_4K_TT_L0_INDEX_MASK,
240 		.valid_mask = ARM_TTE_VALID,
241 		.type_mask  = ARM_TTE_TYPE_MASK,
242 		.type_block = ARM_TTE_TYPE_BLOCK
243 	},
244 	[1] = { /* Concatenated, so index mask is larger than normal */
245 		.size       = ARM_4K_TT_L1_SIZE,
246 		.offmask    = ARM_4K_TT_L1_OFFMASK,
247 		.shift      = ARM_4K_TT_L1_SHIFT,
248 #ifdef ARM_4K_TT_L1_40_BIT_CONCATENATED_INDEX_MASK
249 		.index_mask = ARM_4K_TT_L1_40_BIT_CONCATENATED_INDEX_MASK,
250 #else
251 		.index_mask = ARM_4K_TT_L1_INDEX_MASK,
252 #endif
253 		.valid_mask = ARM_TTE_VALID,
254 		.type_mask  = ARM_TTE_TYPE_MASK,
255 		.type_block = ARM_TTE_TYPE_BLOCK
256 	},
257 	[2] = {
258 		.size       = ARM_4K_TT_L2_SIZE,
259 		.offmask    = ARM_4K_TT_L2_OFFMASK,
260 		.shift      = ARM_4K_TT_L2_SHIFT,
261 		.index_mask = ARM_4K_TT_L2_INDEX_MASK,
262 		.valid_mask = ARM_TTE_VALID,
263 		.type_mask  = ARM_TTE_TYPE_MASK,
264 		.type_block = ARM_TTE_TYPE_BLOCK
265 	},
266 	[3] = {
267 		.size       = ARM_4K_TT_L3_SIZE,
268 		.offmask    = ARM_4K_TT_L3_OFFMASK,
269 		.shift      = ARM_4K_TT_L3_SHIFT,
270 		.index_mask = ARM_4K_TT_L3_INDEX_MASK,
271 		.valid_mask = ARM_PTE_TYPE_VALID,
272 		.type_mask  = ARM_PTE_TYPE_MASK,
273 		.type_block = ARM_TTE_TYPE_L3BLOCK
274 	}
275 };
276 
277 const struct page_table_attr pmap_pt_attr_4k = {
278 	.pta_level_info = pmap_table_level_info_4k,
279 	.pta_root_level = (T0SZ_BOOT - 16) / 9,
280 #if __ARM_MIXED_PAGE_SIZE__
281 	.pta_commpage_level = PMAP_TT_L2_LEVEL,
282 #else /* __ARM_MIXED_PAGE_SIZE__ */
283 #if __ARM_16K_PG__
284 	.pta_commpage_level = PMAP_TT_L2_LEVEL,
285 #else /* __ARM_16K_PG__ */
286 	.pta_commpage_level = PMAP_TT_L1_LEVEL,
287 #endif /* __ARM_16K_PG__ */
288 #endif /* __ARM_MIXED_PAGE_SIZE__ */
289 	.pta_max_level  = PMAP_TT_L3_LEVEL,
290 	.pta_ops = &native_pt_ops,
291 	.ap_ro = ARM_PTE_AP(AP_RORO),
292 	.ap_rw = ARM_PTE_AP(AP_RWRW),
293 	.ap_rona = ARM_PTE_AP(AP_RONA),
294 	.ap_rwna = ARM_PTE_AP(AP_RWNA),
295 	.ap_xn = ARM_PTE_PNX | ARM_PTE_NX,
296 	.ap_x = ARM_PTE_PNX,
297 #if __ARM_MIXED_PAGE_SIZE__
298 	.pta_tcr_value  = TCR_EL1_4KB,
299 #endif /* __ARM_MIXED_PAGE_SIZE__ */
300 	.pta_page_size  = 4096,
301 	.pta_page_shift = 12,
302 	.geometry_id = SPTM_PT_GEOMETRY_4K,
303 };
304 
305 const struct page_table_attr pmap_pt_attr_16k = {
306 	.pta_level_info = pmap_table_level_info_16k,
307 	.pta_root_level = PMAP_TT_L1_LEVEL,
308 	.pta_commpage_level = PMAP_TT_L2_LEVEL,
309 	.pta_max_level  = PMAP_TT_L3_LEVEL,
310 	.pta_ops = &native_pt_ops,
311 	.ap_ro = ARM_PTE_AP(AP_RORO),
312 	.ap_rw = ARM_PTE_AP(AP_RWRW),
313 	.ap_rona = ARM_PTE_AP(AP_RONA),
314 	.ap_rwna = ARM_PTE_AP(AP_RWNA),
315 	.ap_xn = ARM_PTE_PNX | ARM_PTE_NX,
316 	.ap_x = ARM_PTE_PNX,
317 #if __ARM_MIXED_PAGE_SIZE__
318 	.pta_tcr_value  = TCR_EL1_16KB,
319 #endif /* __ARM_MIXED_PAGE_SIZE__ */
320 	.pta_page_size  = 16384,
321 	.pta_page_shift = 14,
322 	.geometry_id = SPTM_PT_GEOMETRY_16K,
323 };
324 
325 #if __ARM_16K_PG__
326 const struct page_table_attr * const native_pt_attr = &pmap_pt_attr_16k;
327 #else /* !__ARM_16K_PG__ */
328 const struct page_table_attr * const native_pt_attr = &pmap_pt_attr_4k;
329 #endif /* !__ARM_16K_PG__ */
330 
331 
332 #if DEVELOPMENT || DEBUG
333 int vm_footprint_suspend_allowed = 1;
334 
335 extern int pmap_ledgers_panic;
336 extern int pmap_ledgers_panic_leeway;
337 
338 #endif /* DEVELOPMENT || DEBUG */
339 
340 #if DEVELOPMENT || DEBUG
341 #define PMAP_FOOTPRINT_SUSPENDED(pmap) \
342 	(current_thread()->pmap_footprint_suspended)
343 #else /* DEVELOPMENT || DEBUG */
344 #define PMAP_FOOTPRINT_SUSPENDED(pmap) (FALSE)
345 #endif /* DEVELOPMENT || DEBUG */
346 
347 #define PMAP_TT_ALLOCATE_NOWAIT         0x1
348 
349 
350 /* Keeps track of whether the pmap has been bootstrapped */
351 SECURITY_READ_ONLY_LATE(bool) pmap_bootstrapped = false;
352 
353 /*
354  * Represents a tlb range that will be flushed before returning from the pmap.
355  * Used by phys_attribute_clear_range to defer flushing pages in this range until
356  * the end of the operation, and to accumulate batched operations for submission
357  * to the SPTM as a performance optimization.
358  */
359 typedef struct pmap_tlb_flush_range {
360 	/* Address space in which the flush region resides */
361 	pmap_t ptfr_pmap;
362 
363 	/* Page-aligned beginning of the flush region */
364 	vm_map_address_t ptfr_start;
365 
366 	/* Page-aligned non-inclusive end of the flush region */
367 	vm_map_address_t ptfr_end;
368 
369 	/**
370 	 * Address of current PTE position in ptfr_pmap's [ptfr_start, ptfr_end) region.
371 	 * This is meant to be set up by the caller of pmap_page_protect_options_with_flush_range()
372 	 * or arm_force_fast_fault_with_flush_range(), and used by those functions to determine
373 	 * when a given mapping can be added to the SPTM's per-CPU region templates array vs.
374 	 * the more complex task of adding it to the disjoint ops array.
375 	 */
376 	pt_entry_t *current_ptep;
377 
378 	/**
379 	 * Starting VA for any not-yet-submitted per-CPU region templates.  This is meant to be
380 	 * set up by the caller of pmap_page_protect_options_with_flush_range() or
381 	 * arm_force_fast_fault_with_flush_range() and used by pmap_multipage_op_submit_region()
382 	 * when issuing the SPTM call to purge any pending region ops.
383 	 */
384 	vm_map_address_t pending_region_start;
385 
386 	/**
387 	 * Number of entries in the per-CPU SPTM region templates array which have not
388 	 * yet been submitted to the SPTM.
389 	 */
390 	unsigned int pending_region_entries;
391 
392 	/**
393 	 * Indicates whether at least one region entry was added to the per-CPU region ops
394 	 * array since the last time this field was checked.  Intended to be cleared by the
395 	 * caller.
396 	 */
397 	bool region_entry_added;
398 
399 	/**
400 	 * Marker for the current paddr "header" entry in the per-CPU SPTM disjoint ops array.
401 	 * This field is intended to be modified only by pmap_multipage_op_submit_disjoint()
402 	 * and pmap_multipage_op_add_page(), and should be treated as opaque by callers
403 	 * of those functions.
404 	 */
405 	sptm_update_disjoint_multipage_op_t *current_header;
406 
407 	/**
408 	 * Position in the per-CPU SPTM ops array of the first ordinary
409 	 * sptm_disjoint_op_t entry following [current_header].  This is the starting
410 	 * point at which mappings should be inserted for the page described by
411 	 * [current_header].
412 	 */
413 	unsigned int current_header_first_mapping_index;
414 
415 	/**
416 	 * Number of entries in the per-CPU SPTM disjoint ops array, including paddr headers,
417 	 * which have not yet been submitted to the SPTM.
418 	 */
419 	unsigned int pending_disjoint_entries;
420 
421 	/**
422 	 * This field is used by the preemption check interval logic on the
423 	 * phys_attribute_clear_range() path to determine when sufficient
424 	 * forward progress has been made to check for and (if necessary)
425 	 * handle pending preemption.
426 	 */
427 	unsigned int processed_entries;
428 
429 	/**
430 	 * Indicates whether the top-level caller needs to flush the TLB for
431 	 * the region in [ptfr_pmap] described by [ptfr_start, ptfr_end).
432 	 * This will be set if the SPTM indicates that it needed to alter
433 	 * any valid mapping within this region and SPTM_UPDATE_DEFER_TLBI
434 	 * was passed to the relevant SPTM call(s).
435 	 */
436 	bool ptfr_flush_needed;
437 } pmap_tlb_flush_range_t;
438 
439 
440 
441 /* Virtual memory region for early allocation */
442 #define VREGION1_HIGH_WINDOW    (PE_EARLY_BOOT_VA)
443 #define VREGION1_START          ((VM_MAX_KERNEL_ADDRESS & CPUWINDOWS_BASE_MASK) - VREGION1_HIGH_WINDOW)
444 #define VREGION1_SIZE           (trunc_page(VM_MAX_KERNEL_ADDRESS - (VREGION1_START)))
445 
446 extern uint8_t bootstrap_pagetables[];
447 
448 extern unsigned int not_in_kdp;
449 
450 extern vm_offset_t first_avail;
451 
452 extern vm_offset_t     virtual_space_start;     /* Next available kernel VA */
453 extern vm_offset_t     virtual_space_end;       /* End of kernel address space */
454 extern vm_offset_t     static_memory_end;
455 
456 extern const vm_map_address_t physmap_base;
457 extern const vm_map_address_t physmap_end;
458 
459 extern int maxproc, hard_maxproc;
460 
461 extern bool sdsb_io_rgns_present;
462 
463 vm_address_t MARK_AS_PMAP_DATA image4_slab = 0;
464 vm_address_t MARK_AS_PMAP_DATA image4_late_slab = 0;
465 
466 /* The number of address bits one TTBR can cover. */
467 #define PGTABLE_ADDR_BITS (64ULL - T0SZ_BOOT)
468 
469 /*
470  * The bounds on our TTBRs.  These are for sanity checking that
471  * an address is accessible by a TTBR before we attempt to map it.
472  */
473 
474 /* The level of the root of a page table. */
475 const uint64_t arm64_root_pgtable_level = (3 - ((PGTABLE_ADDR_BITS - 1 - ARM_PGSHIFT) / (ARM_PGSHIFT - TTE_SHIFT)));
476 
477 /* The number of entries in the root TT of a page table. */
478 const uint64_t arm64_root_pgtable_num_ttes = (2 << ((PGTABLE_ADDR_BITS - 1 - ARM_PGSHIFT) % (ARM_PGSHIFT - TTE_SHIFT)));
479 
480 struct pmap     kernel_pmap_store MARK_AS_PMAP_DATA;
481 const pmap_t    kernel_pmap = &kernel_pmap_store;
482 
483 static SECURITY_READ_ONLY_LATE(zone_t) pmap_zone;  /* zone of pmap structures */
484 
485 MARK_AS_PMAP_DATA SIMPLE_LOCK_DECLARE(pmaps_lock, 0);
486 queue_head_t    map_pmap_list MARK_AS_PMAP_DATA;
487 
488 typedef struct tt_free_entry {
489 	struct tt_free_entry    *next;
490 } tt_free_entry_t;
491 
492 unsigned int    inuse_user_ttepages_count MARK_AS_PMAP_DATA = 0; /* non-root, non-leaf user pagetable pages, in units of PAGE_SIZE */
493 unsigned int    inuse_user_ptepages_count MARK_AS_PMAP_DATA = 0; /* leaf user pagetable pages, in units of PAGE_SIZE */
494 unsigned int    inuse_user_tteroot_count MARK_AS_PMAP_DATA = 0;  /* root user pagetables, in units of PMAP_ROOT_ALLOC_SIZE */
495 unsigned int    inuse_kernel_ttepages_count MARK_AS_PMAP_DATA = 0; /* non-root, non-leaf kernel pagetable pages, in units of PAGE_SIZE */
496 unsigned int    inuse_kernel_ptepages_count MARK_AS_PMAP_DATA = 0; /* leaf kernel pagetable pages, in units of PAGE_SIZE */
497 unsigned int    inuse_kernel_tteroot_count MARK_AS_PMAP_DATA = 0; /* root kernel pagetables, in units of PMAP_ROOT_ALLOC_SIZE */
498 _Atomic unsigned int inuse_iommu_pages_count[SPTM_IOMMUS_N_IDS] = {0}; /* number of active pages for each IOMMU class */
499 
500 SECURITY_READ_ONLY_LATE(tt_entry_t *) invalid_tte  = 0;
501 SECURITY_READ_ONLY_LATE(pmap_paddr_t) invalid_ttep = 0;
502 
503 SECURITY_READ_ONLY_LATE(tt_entry_t *) cpu_tte  = 0;                     /* set by arm_vm_init() - keep out of bss */
504 SECURITY_READ_ONLY_LATE(pmap_paddr_t) cpu_ttep = 0;                     /* set by arm_vm_init() - phys tte addr */
505 
506 /* Lock group used for all pmap object locks. */
507 lck_grp_t pmap_lck_grp MARK_AS_PMAP_DATA;
508 
509 #if DEVELOPMENT || DEBUG
510 int nx_enabled = 1;                                     /* enable no-execute protection */
511 int allow_data_exec  = 0;                               /* No apps may execute data */
512 int allow_stack_exec = 0;                               /* No apps may execute from the stack */
513 unsigned long pmap_asid_flushes MARK_AS_PMAP_DATA = 0;
514 unsigned long pmap_asid_hits MARK_AS_PMAP_DATA = 0;
515 unsigned long pmap_asid_misses MARK_AS_PMAP_DATA = 0;
516 unsigned long pmap_speculation_restrictions MARK_AS_PMAP_DATA = 0;
517 #else /* DEVELOPMENT || DEBUG */
518 const int nx_enabled = 1;                                       /* enable no-execute protection */
519 const int allow_data_exec  = 0;                         /* No apps may execute data */
520 const int allow_stack_exec = 0;                         /* No apps may execute from the stack */
521 #endif /* DEVELOPMENT || DEBUG */
522 
523 
524 #if MACH_ASSERT
525 static void pmap_check_ledgers(pmap_t pmap);
526 #else
527 static inline void
pmap_check_ledgers(__unused pmap_t pmap)528 pmap_check_ledgers(__unused pmap_t pmap)
529 {
530 }
531 #endif /* MACH_ASSERT */
532 
533 SIMPLE_LOCK_DECLARE(phys_backup_lock, 0);
534 
535 SECURITY_READ_ONLY_LATE(pmap_paddr_t)   vm_first_phys = (pmap_paddr_t) 0;
536 SECURITY_READ_ONLY_LATE(pmap_paddr_t)   vm_last_phys = (pmap_paddr_t) 0;
537 
538 SECURITY_READ_ONLY_LATE(boolean_t)      pmap_initialized = FALSE;       /* Has pmap_init completed? */
539 
540 SECURITY_READ_ONLY_LATE(vm_map_offset_t) arm_pmap_max_offset_default  = 0x0;
541 
542 /* end of shared region + 512MB for various purposes */
543 #define ARM64_MIN_MAX_ADDRESS (SHARED_REGION_BASE_ARM64 + SHARED_REGION_SIZE_ARM64 + 0x20000000)
544 _Static_assert((ARM64_MIN_MAX_ADDRESS > SHARED_REGION_BASE_ARM64) && (ARM64_MIN_MAX_ADDRESS <= MACH_VM_MAX_ADDRESS),
545     "Minimum address space size outside allowable range");
546 
547 // Max offset is 15.375GB for devices with "large" memory config
548 #define ARM64_MAX_OFFSET_DEVICE_LARGE (ARM64_MIN_MAX_ADDRESS + 0x138000000)
549 // Max offset is 11.375GB for devices with "small" memory config
550 #define ARM64_MAX_OFFSET_DEVICE_SMALL (ARM64_MIN_MAX_ADDRESS + 0x38000000)
551 
552 
553 _Static_assert((ARM64_MAX_OFFSET_DEVICE_LARGE > ARM64_MIN_MAX_ADDRESS) && (ARM64_MAX_OFFSET_DEVICE_LARGE <= MACH_VM_MAX_ADDRESS),
554     "Large device address space size outside allowable range");
555 _Static_assert((ARM64_MAX_OFFSET_DEVICE_SMALL > ARM64_MIN_MAX_ADDRESS) && (ARM64_MAX_OFFSET_DEVICE_SMALL <= MACH_VM_MAX_ADDRESS),
556     "Small device address space size outside allowable range");
557 
558 #  ifdef XNU_TARGET_OS_OSX
559 SECURITY_READ_ONLY_LATE(vm_map_offset_t) arm64_pmap_max_offset_default = MACH_VM_MAX_ADDRESS;
560 #  else
561 SECURITY_READ_ONLY_LATE(vm_map_offset_t) arm64_pmap_max_offset_default = 0x0;
562 #  endif
563 
564 #if PMAP_PANIC_DEV_WIMG_ON_MANAGED && (DEVELOPMENT || DEBUG)
565 SECURITY_READ_ONLY_LATE(boolean_t)   pmap_panic_dev_wimg_on_managed = TRUE;
566 #else
567 SECURITY_READ_ONLY_LATE(boolean_t)   pmap_panic_dev_wimg_on_managed = FALSE;
568 #endif
569 
570 MARK_AS_PMAP_DATA SIMPLE_LOCK_DECLARE(asid_lock, 0);
571 SECURITY_READ_ONLY_LATE(uint32_t) pmap_max_asids = 0;
572 SECURITY_READ_ONLY_LATE(static bitmap_t*) asid_bitmap;
573 #if !HAS_16BIT_ASID
574 static bitmap_t asid_plru_bitmap[BITMAP_LEN(MAX_HW_ASIDS)] MARK_AS_PMAP_DATA;
575 static uint64_t asid_plru_generation[BITMAP_LEN(MAX_HW_ASIDS)] MARK_AS_PMAP_DATA = {0};
576 static uint64_t asid_plru_gencount MARK_AS_PMAP_DATA = 0;
577 SECURITY_READ_ONLY_LATE(int) pmap_asid_plru = 1;
578 #else
579 static uint16_t last_allocated_asid = 0;
580 #endif /* !HAS_16BIT_ASID */
581 
582 
583 SECURITY_READ_ONLY_LATE(static pmap_paddr_t) commpage_default_table;
584 //SECURITY_READ_ONLY_LATE(static pmap_paddr_t) commpage32_default_table;
585 #if __ARM_MIXED_PAGE_SIZE__
586 SECURITY_READ_ONLY_LATE(static pmap_paddr_t) commpage_4k_table;
587 //SECURITY_READ_ONLY_LATE(static pmap_paddr_t) commpage32_4k_table;
588 #endif
589 SECURITY_READ_ONLY_LATE(static pmap_paddr_t) commpage_data_pa = 0;
590 SECURITY_READ_ONLY_LATE(static pmap_paddr_t) commpage_text_pa = 0;
591 SECURITY_READ_ONLY_LATE(static vm_map_address_t) commpage_text_user_va = 0;
592 SECURITY_READ_ONLY_LATE(static pmap_paddr_t) commpage_ro_data_pa = 0;
593 
594 
595 #if (DEVELOPMENT || DEBUG)
596 /* Caches whether the SPTM sysreg API has been enabled by the SPTM */
597 SECURITY_READ_ONLY_LATE(static bool) sptm_sysreg_available = false;
598 #endif /* (DEVELOPMENT || DEBUG) */
599 
600 /* PTE Define Macros */
601 
602 #ifndef SPTM_PTE_IN_FLIGHT_MARKER
603 /* SPTM TODO: Get rid of this once we export SPTM_PTE_IN_FLIGHT_MARKER from the SPTM. */
604 #define SPTM_PTE_IN_FLIGHT_MARKER 0x80U
605 #endif /* SPTM_PTE_IN_FLIGHT_MARKER */
606 
607 /**
608  * Determine whether a PTE has been marked as compressed.  This function also panics if
609  * the PTE contains bits that shouldn't be present in a compressed PTE, which is most of them.
610  *
611  * @param pte the PTE contents to check
612  * @param ptep the address of the PTE contents, for diagnostic purposes only
613  *
614  * @return true if the PTE is compressed, false otherwise
615  */
616 static inline bool
pte_is_compressed(pt_entry_t pte,pt_entry_t * ptep)617 pte_is_compressed(pt_entry_t pte, pt_entry_t *ptep)
618 {
619 	const bool compressed = (((pte & ARM_PTE_TYPE_VALID) == ARM_PTE_TYPE_FAULT) && (pte & ARM_PTE_COMPRESSED));
620 	/**
621 	 * Check for bits that shouldn't be present in a compressed PTE.  This is everything except the
622 	 * compressed/compressed-alt bits, as well as the SPTM's in-flight marker which may be set while
623 	 * the SPTM is in the process of flushing the TLBs after marking a previously-valid PTE as
624 	 * compressed.
625 	 */
626 	if (__improbable(compressed && (pte & ~(ARM_PTE_COMPRESSED_MASK | SPTM_PTE_IN_FLIGHT_MARKER)))) {
627 		panic("compressed PTE %p 0x%llx has extra bits 0x%llx: corrupted?",
628 		    ptep, pte, pte & ~(ARM_PTE_COMPRESSED_MASK | SPTM_PTE_IN_FLIGHT_MARKER));
629 	}
630 	return compressed;
631 }
632 
633 #define pte_is_wired(pte)                                                               \
634 	(((pte) & ARM_PTE_WIRED_MASK) == ARM_PTE_WIRED)
635 
636 #define pte_was_writeable(pte) \
637 	(((pte) & ARM_PTE_WRITEABLE) == ARM_PTE_WRITEABLE)
638 
639 #define pte_set_was_writeable(pte, was_writeable) \
640 	do {                                         \
641 	        if ((was_writeable)) {               \
642 	                (pte) |= ARM_PTE_WRITEABLE;  \
643 	        } else {                             \
644 	                (pte) &= ~ARM_PTE_WRITEABLE; \
645 	        }                                    \
646 	} while(0)
647 
648 
649 /**
650  * Updated wired-mapping accountings in the PTD and ledger.
651  *
652  * @param pmap The pmap against which to update accounting
653  * @param pte_p The PTE whose wired state is being changed
654  * @param wired Indicates whether the PTE is being wired or unwired.
655  */
656 static inline void
pte_update_wiredcnt(pmap_t pmap,pt_entry_t * pte_p,boolean_t wired)657 pte_update_wiredcnt(pmap_t pmap, pt_entry_t *pte_p, boolean_t wired)
658 {
659 	const pt_attr_t * const pt_attr = pmap_get_pt_attr(pmap);
660 	unsigned short *ptd_wiredcnt_ptr = &(ptep_get_info(pte_p)->wiredcnt);
661 	if (wired) {
662 		if (__improbable(os_atomic_inc_orig(ptd_wiredcnt_ptr, relaxed) == UINT16_MAX)) {
663 			panic("pmap %p (pte %p): wired count overflow", pmap, pte_p);
664 		}
665 		pmap_ledger_credit(pmap, task_ledgers.wired_mem, pt_attr_page_size(pt_attr) * PAGE_RATIO);
666 	} else {
667 		if (__improbable(os_atomic_dec_orig(ptd_wiredcnt_ptr, relaxed) == 0)) {
668 			panic("pmap %p (pte %p): wired count underflow", pmap, pte_p);
669 		}
670 		pmap_ledger_debit(pmap, task_ledgers.wired_mem, pt_attr_page_size(pt_attr) * PAGE_RATIO);
671 	}
672 }
673 
674 /*
675  * Synchronize updates to PTEs that were previously invalid or had the AF bit cleared,
676  * therefore not requiring TLBI.  Use a store-load barrier to ensure subsequent loads
677  * will observe the updated PTE.
678  */
679 #define FLUSH_PTE()                                                                     \
680 	__builtin_arm_dmb(DMB_ISH);
681 
682 /*
683  * Synchronize updates to PTEs that were previously valid and thus may be cached in
684  * TLBs.  DSB is required to ensure the PTE stores have completed prior to the ensuing
685  * TLBI.  This should only require a store-store barrier, as subsequent accesses in
686  * program order will not issue until the DSB completes.  Prior loads may be reordered
687  * after the barrier, but their behavior should not be materially affected by the
688  * reordering.  For fault-driven PTE updates such as COW, PTE contents should not
689  * matter for loads until the access is re-driven well after the TLB update is
690  * synchronized.   For "involuntary" PTE access restriction due to paging lifecycle,
691  * we should be in a position to handle access faults.  For "voluntary" PTE access
692  * restriction due to unmapping or protection, the decision to restrict access should
693  * have a data dependency on prior loads in order to avoid a data race.
694  */
695 #define FLUSH_PTE_STRONG()                                                             \
696 	__builtin_arm_dsb(DSB_ISHST);
697 
698 /**
699  * Write enough page table entries to map a single VM page. On systems where the
700  * VM page size does not match the hardware page size, multiple page table
701  * entries will need to be written.
702  *
703  * @note This function does not emit a barrier to ensure these page table writes
704  *       have completed before continuing. This is commonly needed. In the case
705  *       where a DMB or DSB barrier is needed, then use the write_pte() and
706  *       write_pte_strong() functions respectively instead of this one.
707  *
708  * @param ptep Pointer to the first page table entry to update.
709  * @param pte The value to write into each page table entry. In the case that
710  *            multiple PTEs are updated to a non-empty value, then the address
711  *            in this value will automatically be incremented for each PTE
712  *            write.
713  */
714 static void
write_pte_fast(pt_entry_t * ptep,pt_entry_t pte)715 write_pte_fast(pt_entry_t *ptep, pt_entry_t pte)
716 {
717 	/**
718 	 * The PAGE_SHIFT (and in turn, the PAGE_RATIO) can be a variable on some
719 	 * systems, which is why it's checked at runtime instead of compile time.
720 	 * The "unreachable" warning needs to be suppressed because it still is a
721 	 * compile time constant on some systems.
722 	 */
723 	__unreachable_ok_push
724 	if (TEST_PAGE_RATIO_4) {
725 		if (((uintptr_t)ptep) & 0x1f) {
726 			panic("%s: PTE write is unaligned, ptep=%p, pte=%p",
727 			    __func__, ptep, (void*)pte);
728 		}
729 
730 		if ((pte & ~ARM_PTE_COMPRESSED_MASK) == ARM_PTE_EMPTY) {
731 			/**
732 			 * If we're writing an empty/compressed PTE value, then don't
733 			 * auto-increment the address for each PTE write.
734 			 */
735 			*ptep = pte;
736 			*(ptep + 1) = pte;
737 			*(ptep + 2) = pte;
738 			*(ptep + 3) = pte;
739 		} else {
740 			*ptep = pte;
741 			*(ptep + 1) = pte | 0x1000;
742 			*(ptep + 2) = pte | 0x2000;
743 			*(ptep + 3) = pte | 0x3000;
744 		}
745 	} else {
746 		*ptep = pte;
747 	}
748 	__unreachable_ok_pop
749 }
750 
751 /**
752  * Writes enough page table entries to map a single VM page and then ensures
753  * those writes complete by executing a Data Memory Barrier.
754  *
755  * @note The DMB issued by this function is not strong enough to protect against
756  *       TLB invalidates from being reordered above the PTE writes. If a TLBI
757  *       instruction is going to immediately be called after this write, it's
758  *       recommended to call write_pte_strong() instead of this function.
759  *
760  * See the function header for write_pte_fast() for more details on the
761  * parameters.
762  */
763 void
write_pte(pt_entry_t * ptep,pt_entry_t pte)764 write_pte(pt_entry_t *ptep, pt_entry_t pte)
765 {
766 	write_pte_fast(ptep, pte);
767 	FLUSH_PTE();
768 }
769 
770 /**
771  * Retrieve the pmap structure for the thread running on the current CPU.
772  */
773 pmap_t
current_pmap()774 current_pmap()
775 {
776 	const pmap_t current = vm_map_pmap(current_thread()->map);
777 	assert(current != NULL);
778 	return current;
779 }
780 
781 #if DEVELOPMENT || DEBUG
782 
783 /*
784  * Trace levels are controlled by a bitmask in which each
785  * level can be enabled/disabled by the (1<<level) position
786  * in the boot arg
787  * Level 0: PPL extension functionality
788  * Level 1: pmap lifecycle (create/destroy/switch)
789  * Level 2: mapping lifecycle (enter/remove/protect/nest/unnest)
790  * Level 3: internal state management (attributes/fast-fault)
791  * Level 4-7: TTE traces for paging levels 0-3.  TTBs are traced at level 4.
792  */
793 
794 SECURITY_READ_ONLY_LATE(unsigned int) pmap_trace_mask = 0;
795 
796 #define PMAP_TRACE(level, ...) \
797 	if (__improbable((1 << (level)) & pmap_trace_mask)) { \
798 	        KDBG_RELEASE(__VA_ARGS__); \
799 	}
800 #else /* DEVELOPMENT || DEBUG */
801 
802 #define PMAP_TRACE(level, ...)
803 
804 #endif /* DEVELOPMENT || DEBUG */
805 
806 
807 /*
808  * Internal function prototypes (forward declarations).
809  */
810 
811 static vm_map_size_t pmap_user_va_size(pmap_t pmap);
812 
813 static void pmap_set_reference(ppnum_t pn);
814 
815 pmap_paddr_t pmap_vtophys(pmap_t pmap, addr64_t va);
816 
817 static kern_return_t pmap_expand(
818 	pmap_t, vm_map_address_t, unsigned int options, unsigned int level);
819 
820 static void pmap_remove_range(pmap_t, vm_map_address_t, vm_map_address_t);
821 
822 static tt_entry_t *pmap_tt1_allocate(pmap_t, uint8_t);
823 
824 static void pmap_tt1_deallocate(pmap_t, tt_entry_t *);
825 
826 static kern_return_t pmap_tt_allocate(
827 	pmap_t, tt_entry_t **, unsigned int, unsigned int);
828 
829 const unsigned int arm_hardware_page_size = ARM_PGBYTES;
830 const unsigned int arm_pt_desc_size = sizeof(pt_desc_t);
831 const unsigned int arm_pt_root_size = PMAP_ROOT_ALLOC_SIZE;
832 
833 static void pmap_unmap_commpage(
834 	pmap_t pmap);
835 
836 static boolean_t
837 pmap_is_64bit(pmap_t);
838 
839 
840 static void pmap_flush_tlb_for_paddr_async(pmap_paddr_t);
841 
842 static void pmap_update_pp_attr_wimg_bits_locked(unsigned int, unsigned int);
843 
844 static boolean_t arm_clear_fast_fault(
845 	ppnum_t ppnum,
846 	vm_prot_t fault_type,
847 	uintptr_t pvh,
848 	pt_entry_t *pte_p,
849 	pp_attr_t attrs_to_clear);
850 
851 static void pmap_trim_self(pmap_t pmap);
852 static void pmap_trim_subord(pmap_t subord);
853 
854 
855 /*
856  * Temporary prototypes, while we wait for pmap_enter to move to taking an
857  * address instead of a page number.
858  */
859 kern_return_t
860 pmap_enter(
861 	pmap_t pmap,
862 	vm_map_address_t v,
863 	ppnum_t pn,
864 	vm_prot_t prot,
865 	vm_prot_t fault_type,
866 	unsigned int flags,
867 	boolean_t wired,
868 	pmap_mapping_type_t mapping_type);
869 
870 static kern_return_t
871 pmap_enter_addr(
872 	pmap_t pmap,
873 	vm_map_address_t v,
874 	pmap_paddr_t pa,
875 	vm_prot_t prot,
876 	vm_prot_t fault_type,
877 	unsigned int flags,
878 	boolean_t wired,
879 	pmap_mapping_type_t mapping_type);
880 
881 kern_return_t
882 pmap_enter_options_addr(
883 	pmap_t pmap,
884 	vm_map_address_t v,
885 	pmap_paddr_t pa,
886 	vm_prot_t prot,
887 	vm_prot_t fault_type,
888 	unsigned int flags,
889 	boolean_t wired,
890 	unsigned int options,
891 	__unused void   *arg,
892 	pmap_mapping_type_t mapping_type);
893 
894 #ifdef CONFIG_XNUPOST
895 kern_return_t pmap_test(void);
896 #endif /* CONFIG_XNUPOST */
897 
898 PMAP_SUPPORT_PROTOTYPES(
899 	kern_return_t,
900 	arm_fast_fault, (pmap_t pmap,
901 	vm_map_address_t va,
902 	vm_prot_t fault_type,
903 	bool was_af_fault,
904 	bool from_user), ARM_FAST_FAULT_INDEX);
905 
906 PMAP_SUPPORT_PROTOTYPES(
907 	boolean_t,
908 	arm_force_fast_fault, (ppnum_t ppnum,
909 	vm_prot_t allow_mode,
910 	int options), ARM_FORCE_FAST_FAULT_INDEX);
911 
912 MARK_AS_PMAP_TEXT static boolean_t
913 arm_force_fast_fault_with_flush_range(
914 	ppnum_t ppnum,
915 	vm_prot_t allow_mode,
916 	int options,
917 	locked_pvh_t *locked_pvh,
918 	pp_attr_t bits_to_clear,
919 	pmap_tlb_flush_range_t *flush_range);
920 
921 PMAP_SUPPORT_PROTOTYPES(
922 	void,
923 	pmap_batch_set_cache_attributes, (
924 		const unified_page_list_t * page_list,
925 		unsigned int cacheattr,
926 		bool update_attr_table), PMAP_BATCH_SET_CACHE_ATTRIBUTES_INDEX);
927 
928 PMAP_SUPPORT_PROTOTYPES(
929 	void,
930 	pmap_change_wiring, (pmap_t pmap,
931 	vm_map_address_t v,
932 	boolean_t wired), PMAP_CHANGE_WIRING_INDEX);
933 
934 PMAP_SUPPORT_PROTOTYPES(
935 	pmap_t,
936 	pmap_create_options, (ledger_t ledger,
937 	vm_map_size_t size,
938 	unsigned int flags,
939 	kern_return_t * kr), PMAP_CREATE_INDEX);
940 
941 PMAP_SUPPORT_PROTOTYPES(
942 	void,
943 	pmap_destroy, (pmap_t pmap), PMAP_DESTROY_INDEX);
944 
945 PMAP_SUPPORT_PROTOTYPES(
946 	kern_return_t,
947 	pmap_enter_options, (pmap_t pmap,
948 	vm_map_address_t v,
949 	pmap_paddr_t pa,
950 	vm_prot_t prot,
951 	vm_prot_t fault_type,
952 	unsigned int flags,
953 	boolean_t wired,
954 	unsigned int options,
955 	pmap_mapping_type_t mapping_type), PMAP_ENTER_OPTIONS_INDEX);
956 
957 PMAP_SUPPORT_PROTOTYPES(
958 	pmap_paddr_t,
959 	pmap_find_pa, (pmap_t pmap,
960 	addr64_t va), PMAP_FIND_PA_INDEX);
961 
962 PMAP_SUPPORT_PROTOTYPES(
963 	kern_return_t,
964 	pmap_insert_commpage, (pmap_t pmap), PMAP_INSERT_COMMPAGE_INDEX);
965 
966 
967 PMAP_SUPPORT_PROTOTYPES(
968 	boolean_t,
969 	pmap_is_empty, (pmap_t pmap,
970 	vm_map_offset_t va_start,
971 	vm_map_offset_t va_end), PMAP_IS_EMPTY_INDEX);
972 
973 
974 PMAP_SUPPORT_PROTOTYPES(
975 	unsigned int,
976 	pmap_map_cpu_windows_copy, (ppnum_t pn,
977 	vm_prot_t prot,
978 	unsigned int wimg_bits), PMAP_MAP_CPU_WINDOWS_COPY_INDEX);
979 
980 PMAP_SUPPORT_PROTOTYPES(
981 	void,
982 	pmap_ro_zone_memcpy, (zone_id_t zid,
983 	vm_offset_t va,
984 	vm_offset_t offset,
985 	const vm_offset_t new_data,
986 	vm_size_t new_data_size), PMAP_RO_ZONE_MEMCPY_INDEX);
987 
988 PMAP_SUPPORT_PROTOTYPES(
989 	uint64_t,
990 	pmap_ro_zone_atomic_op, (zone_id_t zid,
991 	vm_offset_t va,
992 	vm_offset_t offset,
993 	zro_atomic_op_t op,
994 	uint64_t value), PMAP_RO_ZONE_ATOMIC_OP_INDEX);
995 
996 PMAP_SUPPORT_PROTOTYPES(
997 	void,
998 	pmap_ro_zone_bzero, (zone_id_t zid,
999 	vm_offset_t va,
1000 	vm_offset_t offset,
1001 	vm_size_t size), PMAP_RO_ZONE_BZERO_INDEX);
1002 
1003 PMAP_SUPPORT_PROTOTYPES(
1004 	kern_return_t,
1005 	pmap_nest, (pmap_t grand,
1006 	pmap_t subord,
1007 	addr64_t vstart,
1008 	uint64_t size), PMAP_NEST_INDEX);
1009 
1010 PMAP_SUPPORT_PROTOTYPES(
1011 	void,
1012 	pmap_page_protect_options, (ppnum_t ppnum,
1013 	vm_prot_t prot,
1014 	unsigned int options,
1015 	void *arg), PMAP_PAGE_PROTECT_OPTIONS_INDEX);
1016 
1017 PMAP_SUPPORT_PROTOTYPES(
1018 	vm_map_address_t,
1019 	pmap_protect_options, (pmap_t pmap,
1020 	vm_map_address_t start,
1021 	vm_map_address_t end,
1022 	vm_prot_t prot,
1023 	unsigned int options,
1024 	void *args), PMAP_PROTECT_OPTIONS_INDEX);
1025 
1026 PMAP_SUPPORT_PROTOTYPES(
1027 	kern_return_t,
1028 	pmap_query_page_info, (pmap_t pmap,
1029 	vm_map_offset_t va,
1030 	int *disp_p), PMAP_QUERY_PAGE_INFO_INDEX);
1031 
1032 PMAP_SUPPORT_PROTOTYPES(
1033 	mach_vm_size_t,
1034 	pmap_query_resident, (pmap_t pmap,
1035 	vm_map_address_t start,
1036 	vm_map_address_t end,
1037 	mach_vm_size_t * compressed_bytes_p), PMAP_QUERY_RESIDENT_INDEX);
1038 
1039 PMAP_SUPPORT_PROTOTYPES(
1040 	void,
1041 	pmap_reference, (pmap_t pmap), PMAP_REFERENCE_INDEX);
1042 
1043 PMAP_SUPPORT_PROTOTYPES(
1044 	vm_map_address_t,
1045 	pmap_remove_options, (pmap_t pmap,
1046 	vm_map_address_t start,
1047 	vm_map_address_t end,
1048 	int options), PMAP_REMOVE_OPTIONS_INDEX);
1049 
1050 
1051 PMAP_SUPPORT_PROTOTYPES(
1052 	void,
1053 	pmap_set_cache_attributes, (ppnum_t pn,
1054 	unsigned int cacheattr,
1055 	bool update_attr_table), PMAP_SET_CACHE_ATTRIBUTES_INDEX);
1056 
1057 PMAP_SUPPORT_PROTOTYPES(
1058 	void,
1059 	pmap_update_compressor_page, (ppnum_t pn,
1060 	unsigned int prev_cacheattr, unsigned int new_cacheattr), PMAP_UPDATE_COMPRESSOR_PAGE_INDEX);
1061 
1062 PMAP_SUPPORT_PROTOTYPES(
1063 	void,
1064 	pmap_set_nested, (pmap_t pmap), PMAP_SET_NESTED_INDEX);
1065 
1066 #if MACH_ASSERT
1067 PMAP_SUPPORT_PROTOTYPES(
1068 	void,
1069 	pmap_set_process, (pmap_t pmap,
1070 	int pid,
1071 	char *procname), PMAP_SET_PROCESS_INDEX);
1072 #endif
1073 
1074 PMAP_SUPPORT_PROTOTYPES(
1075 	void,
1076 	pmap_unmap_cpu_windows_copy, (unsigned int index), PMAP_UNMAP_CPU_WINDOWS_COPY_INDEX);
1077 
1078 PMAP_SUPPORT_PROTOTYPES(
1079 	void,
1080 	pmap_unnest_options, (pmap_t grand,
1081 	addr64_t vaddr,
1082 	uint64_t size,
1083 	unsigned int option), PMAP_UNNEST_OPTIONS_INDEX);
1084 
1085 PMAP_SUPPORT_PROTOTYPES(
1086 	void,
1087 	phys_attribute_set, (ppnum_t pn,
1088 	unsigned int bits), PHYS_ATTRIBUTE_SET_INDEX);
1089 
1090 PMAP_SUPPORT_PROTOTYPES(
1091 	void,
1092 	phys_attribute_clear, (ppnum_t pn,
1093 	unsigned int bits,
1094 	int options,
1095 	void *arg), PHYS_ATTRIBUTE_CLEAR_INDEX);
1096 
1097 #if __ARM_RANGE_TLBI__
1098 PMAP_SUPPORT_PROTOTYPES(
1099 	vm_map_address_t,
1100 	phys_attribute_clear_range, (pmap_t pmap,
1101 	vm_map_address_t start,
1102 	vm_map_address_t end,
1103 	unsigned int bits,
1104 	unsigned int options), PHYS_ATTRIBUTE_CLEAR_RANGE_INDEX);
1105 #endif /* __ARM_RANGE_TLBI__ */
1106 
1107 
1108 PMAP_SUPPORT_PROTOTYPES(
1109 	void,
1110 	pmap_switch, (pmap_t pmap), PMAP_SWITCH_INDEX);
1111 
1112 PMAP_SUPPORT_PROTOTYPES(
1113 	void,
1114 	pmap_clear_user_ttb, (void), PMAP_CLEAR_USER_TTB_INDEX);
1115 
1116 PMAP_SUPPORT_PROTOTYPES(
1117 	void,
1118 	pmap_set_vm_map_cs_enforced, (pmap_t pmap, bool new_value), PMAP_SET_VM_MAP_CS_ENFORCED_INDEX);
1119 
1120 PMAP_SUPPORT_PROTOTYPES(
1121 	void,
1122 	pmap_set_tpro, (pmap_t pmap), PMAP_SET_TPRO_INDEX);
1123 
1124 PMAP_SUPPORT_PROTOTYPES(
1125 	void,
1126 	pmap_set_jit_entitled, (pmap_t pmap), PMAP_SET_JIT_ENTITLED_INDEX);
1127 
1128 #if __has_feature(ptrauth_calls) && (defined(XNU_TARGET_OS_OSX) || (DEVELOPMENT || DEBUG))
1129 PMAP_SUPPORT_PROTOTYPES(
1130 	void,
1131 	pmap_disable_user_jop, (pmap_t pmap), PMAP_DISABLE_USER_JOP_INDEX);
1132 #endif /* __has_feature(ptrauth_calls) && (defined(XNU_TARGET_OS_OSX) || (DEVELOPMENT || DEBUG)) */
1133 
1134 PMAP_SUPPORT_PROTOTYPES(
1135 	void,
1136 	pmap_trim, (pmap_t grand,
1137 	pmap_t subord,
1138 	addr64_t vstart,
1139 	uint64_t size), PMAP_TRIM_INDEX);
1140 
1141 #if HAS_APPLE_PAC
1142 PMAP_SUPPORT_PROTOTYPES(
1143 	void *,
1144 	pmap_sign_user_ptr, (void *value, ptrauth_key key, uint64_t discriminator, uint64_t jop_key), PMAP_SIGN_USER_PTR);
1145 PMAP_SUPPORT_PROTOTYPES(
1146 	void *,
1147 	pmap_auth_user_ptr, (void *value, ptrauth_key key, uint64_t discriminator, uint64_t jop_key), PMAP_AUTH_USER_PTR);
1148 #endif /* HAS_APPLE_PAC */
1149 
1150 
1151 void pmap_footprint_suspend(vm_map_t    map,
1152     boolean_t   suspend);
1153 PMAP_SUPPORT_PROTOTYPES(
1154 	void,
1155 	pmap_footprint_suspend, (vm_map_t map,
1156 	boolean_t suspend),
1157 	PMAP_FOOTPRINT_SUSPEND_INDEX);
1158 
1159 
1160 
1161 
1162 
1163 /*
1164  * The low global vector page is mapped at a fixed alias.
1165  * Since the page size is 16k for H8 and newer we map the globals to a 16k
1166  * aligned address. Readers of the globals (e.g. lldb, panic server) need
1167  * to check both addresses anyway for backward compatibility. So for now
1168  * we leave H6 and H7 where they were.
1169  */
1170 #if (ARM_PGSHIFT == 14)
1171 #define LOWGLOBAL_ALIAS         (LOW_GLOBAL_BASE_ADDRESS + 0x4000)
1172 #else
1173 #define LOWGLOBAL_ALIAS         (LOW_GLOBAL_BASE_ADDRESS + 0x2000)
1174 #endif
1175 
1176 static inline void
PMAP_ZINFO_PALLOC(pmap_t pmap,int bytes)1177 PMAP_ZINFO_PALLOC(
1178 	pmap_t pmap, int bytes)
1179 {
1180 	pmap_ledger_credit(pmap, task_ledgers.tkm_private, bytes);
1181 }
1182 
1183 static inline void
PMAP_ZINFO_PFREE(pmap_t pmap,int bytes)1184 PMAP_ZINFO_PFREE(
1185 	pmap_t pmap,
1186 	int bytes)
1187 {
1188 	pmap_ledger_debit(pmap, task_ledgers.tkm_private, bytes);
1189 }
1190 
1191 void
pmap_tt_ledger_credit(pmap_t pmap,vm_size_t size)1192 pmap_tt_ledger_credit(
1193 	pmap_t          pmap,
1194 	vm_size_t       size)
1195 {
1196 	if (pmap != kernel_pmap) {
1197 		pmap_ledger_credit(pmap, task_ledgers.phys_footprint, size);
1198 		pmap_ledger_credit(pmap, task_ledgers.page_table, size);
1199 	}
1200 }
1201 
1202 void
pmap_tt_ledger_debit(pmap_t pmap,vm_size_t size)1203 pmap_tt_ledger_debit(
1204 	pmap_t          pmap,
1205 	vm_size_t       size)
1206 {
1207 	if (pmap != kernel_pmap) {
1208 		pmap_ledger_debit(pmap, task_ledgers.phys_footprint, size);
1209 		pmap_ledger_debit(pmap, task_ledgers.page_table, size);
1210 	}
1211 }
1212 
1213 static inline void
pmap_update_plru(uint16_t asid_index __unused)1214 pmap_update_plru(uint16_t asid_index __unused)
1215 {
1216 #if !HAS_16BIT_ASID
1217 	if (__probable(pmap_asid_plru)) {
1218 		unsigned plru_index = asid_index >> 6;
1219 		if (__improbable(os_atomic_andnot(&asid_plru_bitmap[plru_index], (1ULL << (asid_index & 63)), relaxed) == 0)) {
1220 			asid_plru_generation[plru_index] = ++asid_plru_gencount;
1221 			asid_plru_bitmap[plru_index] = ((plru_index == 0) ? ~1ULL : UINT64_MAX);
1222 		}
1223 	}
1224 #endif /* !HAS_16BIT_ASID */
1225 }
1226 
1227 static bool
alloc_asid(pmap_t pmap)1228 alloc_asid(pmap_t pmap)
1229 {
1230 	int vasid = -1;
1231 
1232 	pmap_simple_lock(&asid_lock);
1233 
1234 #if !HAS_16BIT_ASID
1235 	if (__probable(pmap_asid_plru)) {
1236 		unsigned plru_index = 0;
1237 		uint64_t lowest_gen = asid_plru_generation[0];
1238 		uint64_t lowest_gen_bitmap = asid_plru_bitmap[0];
1239 		for (unsigned i = 1; i < (sizeof(asid_plru_generation) / sizeof(asid_plru_generation[0])); ++i) {
1240 			if (asid_plru_generation[i] < lowest_gen) {
1241 				plru_index = i;
1242 				lowest_gen = asid_plru_generation[i];
1243 				lowest_gen_bitmap = asid_plru_bitmap[i];
1244 			}
1245 		}
1246 
1247 		for (; plru_index < BITMAP_LEN(pmap_max_asids); plru_index += (MAX_HW_ASIDS >> 6)) {
1248 			uint64_t temp_plru = lowest_gen_bitmap & asid_bitmap[plru_index];
1249 			if (temp_plru) {
1250 				vasid = (plru_index << 6) + lsb_first(temp_plru);
1251 #if DEVELOPMENT || DEBUG
1252 				++pmap_asid_hits;
1253 #endif
1254 				break;
1255 			}
1256 		}
1257 	}
1258 #else
1259 	/**
1260 	 * For 16-bit ASID targets, we assume a 1:1 correspondence between ASIDs and active tasks and
1261 	 * therefore allocate directly from the ASID bitmap instead of using the pLRU allocator.
1262 	 * However, we first try to allocate starting from the position of the most-recently allocated
1263 	 * ASID.  This is done both as an allocator performance optimization (as it avoids crowding the
1264 	 * lower bit positions and then re-checking those same lower positions every time we allocate
1265 	 * an ASID) as well as a security mitigation to increase the temporal distance between ASID
1266 	 * reuse.  This increases the difficulty of leveraging ASID reuse to train branch predictor
1267 	 * logic, without requiring prohibitively expensive RCTX instructions.
1268 	 */
1269 	vasid = bitmap_lsb_next(&asid_bitmap[0], pmap_max_asids, last_allocated_asid);
1270 #endif /* !HAS_16BIT_ASID */
1271 	if (__improbable(vasid < 0)) {
1272 		// bitmap_first() returns highest-order bits first, but a 0-based scheme works
1273 		// slightly better with the collision detection scheme used by pmap_switch_internal().
1274 		vasid = bitmap_lsb_first(&asid_bitmap[0], pmap_max_asids);
1275 #if DEVELOPMENT || DEBUG
1276 		++pmap_asid_misses;
1277 #endif
1278 	}
1279 	if (__improbable(vasid < 0)) {
1280 		pmap_simple_unlock(&asid_lock);
1281 		return false;
1282 	}
1283 	assert((uint32_t)vasid < pmap_max_asids);
1284 	assert(bitmap_test(&asid_bitmap[0], (unsigned int)vasid));
1285 	bitmap_clear(&asid_bitmap[0], (unsigned int)vasid);
1286 	const uint16_t hw_asid = (uint16_t)(vasid & (MAX_HW_ASIDS - 1));
1287 #if HAS_16BIT_ASID
1288 	last_allocated_asid = hw_asid;
1289 #endif /* HAS_16BIT_ASID */
1290 	pmap_simple_unlock(&asid_lock);
1291 	assert(hw_asid != 0); // Should never alias kernel ASID
1292 	pmap->asid = (uint16_t)vasid;
1293 	pmap_update_plru(hw_asid);
1294 	return true;
1295 }
1296 
1297 static void
free_asid(pmap_t pmap)1298 free_asid(pmap_t pmap)
1299 {
1300 	const uint16_t vasid = os_atomic_xchg(&pmap->asid, 0, relaxed);
1301 	if (__improbable(vasid == 0)) {
1302 		return;
1303 	}
1304 
1305 #if !HAS_16BIT_ASID
1306 	if (pmap_asid_plru) {
1307 		const uint16_t hw_asid = vasid & (MAX_HW_ASIDS - 1);
1308 		os_atomic_or(&asid_plru_bitmap[hw_asid >> 6], (1ULL << (hw_asid & 63)), relaxed);
1309 	}
1310 #endif /* !HAS_16BIT_ASID */
1311 	pmap_simple_lock(&asid_lock);
1312 	assert(!bitmap_test(&asid_bitmap[0], vasid));
1313 	bitmap_set(&asid_bitmap[0], vasid);
1314 	pmap_simple_unlock(&asid_lock);
1315 }
1316 
1317 
1318 boolean_t
pmap_valid_address(pmap_paddr_t addr)1319 pmap_valid_address(
1320 	pmap_paddr_t addr)
1321 {
1322 	return pa_valid(addr);
1323 }
1324 
1325 
1326 
1327 
1328 
1329 
1330 /*
1331  *      Map memory at initialization.  The physical addresses being
1332  *      mapped are not managed and are never unmapped.
1333  *
1334  *      For now, VM is already on, we only need to map the
1335  *      specified memory.
1336  */
1337 vm_map_address_t
pmap_map(vm_map_address_t virt,vm_offset_t start,vm_offset_t end,vm_prot_t prot,unsigned int flags)1338 pmap_map(
1339 	vm_map_address_t virt,
1340 	vm_offset_t start,
1341 	vm_offset_t end,
1342 	vm_prot_t prot,
1343 	unsigned int flags)
1344 {
1345 	kern_return_t   kr;
1346 	vm_size_t       ps;
1347 
1348 	ps = PAGE_SIZE;
1349 	while (start < end) {
1350 		kr = pmap_enter(kernel_pmap, virt, (ppnum_t)atop(start),
1351 		    prot, VM_PROT_NONE, flags, FALSE, PMAP_MAPPING_TYPE_INFER);
1352 
1353 		if (kr != KERN_SUCCESS) {
1354 			panic("%s: failed pmap_enter, "
1355 			    "virt=%p, start_addr=%p, end_addr=%p, prot=%#x, flags=%#x",
1356 			    __FUNCTION__,
1357 			    (void *) virt, (void *) start, (void *) end, prot, flags);
1358 		}
1359 
1360 		virt += ps;
1361 		start += ps;
1362 	}
1363 	return virt;
1364 }
1365 
1366 /**
1367  * Force the permission of a PTE to be kernel RO if a page has XNU_PROTECTED_IO type.
1368  *
1369  * @param paddr The physical address of the page.
1370  * @param tmplate The PTE value to be evaluated.
1371  *
1372  * @return A new PTE value with permission bits modified.
1373  */
1374 static inline
1375 pt_entry_t
pmap_force_pte_kernel_ro_if_protected_io(pmap_paddr_t paddr,pt_entry_t tmplate)1376 pmap_force_pte_kernel_ro_if_protected_io(pmap_paddr_t paddr, pt_entry_t tmplate)
1377 {
1378 	/**
1379 	 * When requesting RW mappings to an XNU_PROTECTED_IO frame, downgrade
1380 	 * the mapping to RO. This is required because IOKit relies on this
1381 	 * behavior currently in the PPL.
1382 	 */
1383 	const sptm_frame_type_t frame_type = sptm_get_frame_type(paddr);
1384 	if (frame_type == XNU_PROTECTED_IO) {
1385 		/* SPTM to own the page by converting KERN_RW to PPL_RW. */
1386 		const uint64_t xprr_perm = pte_to_xprr_perm(tmplate);
1387 		switch (xprr_perm) {
1388 		case XPRR_KERN_RO_PERM:
1389 			break;
1390 		case XPRR_KERN_RW_PERM:
1391 			tmplate &= ~ARM_PTE_XPRR_MASK;
1392 			tmplate |= xprr_perm_to_pte(XPRR_KERN_RO_PERM);
1393 			break;
1394 		default:
1395 			panic("%s: Unsupported xPRR perm %llu for pte 0x%llx", __func__, xprr_perm, (uint64_t)tmplate);
1396 		}
1397 	}
1398 
1399 	return tmplate;
1400 }
1401 
1402 vm_map_address_t
pmap_map_bd_with_options(vm_map_address_t virt,vm_offset_t start,vm_offset_t end,vm_prot_t prot,int32_t options)1403 pmap_map_bd_with_options(
1404 	vm_map_address_t virt,
1405 	vm_offset_t start,
1406 	vm_offset_t end,
1407 	vm_prot_t prot,
1408 	int32_t options)
1409 {
1410 	pt_entry_t      tmplate;
1411 	vm_map_address_t vaddr;
1412 	vm_offset_t     paddr;
1413 	pt_entry_t      mem_attr;
1414 
1415 	switch (options & PMAP_MAP_BD_MASK) {
1416 	case PMAP_MAP_BD_WCOMB:
1417 		mem_attr = ARM_PTE_ATTRINDX(CACHE_ATTRINDX_WRITECOMB);
1418 		mem_attr |= ARM_PTE_SH(SH_OUTER_MEMORY);
1419 		break;
1420 	case PMAP_MAP_BD_POSTED:
1421 		mem_attr = ARM_PTE_ATTRINDX(CACHE_ATTRINDX_POSTED);
1422 		break;
1423 	case PMAP_MAP_BD_POSTED_REORDERED:
1424 		mem_attr = ARM_PTE_ATTRINDX(CACHE_ATTRINDX_POSTED_REORDERED);
1425 		break;
1426 	case PMAP_MAP_BD_POSTED_COMBINED_REORDERED:
1427 		mem_attr = ARM_PTE_ATTRINDX(CACHE_ATTRINDX_POSTED_COMBINED_REORDERED);
1428 		break;
1429 	default:
1430 		mem_attr = ARM_PTE_ATTRINDX(CACHE_ATTRINDX_DISABLE);
1431 		break;
1432 	}
1433 
1434 	tmplate = ARM_PTE_AP((prot & VM_PROT_WRITE) ? AP_RWNA : AP_RONA) |
1435 	    mem_attr | ARM_PTE_TYPE | ARM_PTE_NX | ARM_PTE_PNX | ARM_PTE_AF;
1436 
1437 #if __ARM_KERNEL_PROTECT__
1438 	tmplate |= ARM_PTE_NG;
1439 #endif /* __ARM_KERNEL_PROTECT__ */
1440 
1441 	vaddr = virt;
1442 	paddr = start;
1443 	while (paddr < end) {
1444 		__assert_only sptm_return_t ret = sptm_map_page(kernel_pmap->ttep, vaddr, pmap_force_pte_kernel_ro_if_protected_io(paddr, tmplate) | pa_to_pte(paddr));
1445 		assert((ret == SPTM_SUCCESS) || (ret == SPTM_MAP_VALID));
1446 
1447 		vaddr += PAGE_SIZE;
1448 		paddr += PAGE_SIZE;
1449 	}
1450 
1451 	return vaddr;
1452 }
1453 
1454 /*
1455  *      Back-door routine for mapping kernel VM at initialization.
1456  *      Useful for mapping memory outside the range
1457  *      [vm_first_phys, vm_last_phys] (i.e., devices).
1458  *      Otherwise like pmap_map.
1459  */
1460 vm_map_address_t
pmap_map_bd(vm_map_address_t virt,vm_offset_t start,vm_offset_t end,vm_prot_t prot)1461 pmap_map_bd(
1462 	vm_map_address_t virt,
1463 	vm_offset_t start,
1464 	vm_offset_t end,
1465 	vm_prot_t prot)
1466 {
1467 	return pmap_map_bd_with_options(virt, start, end, prot, 0);
1468 }
1469 
1470 /*
1471  *      Back-door routine for mapping kernel VM at initialization.
1472  *      Useful for mapping memory specific physical addresses in early
1473  *      boot (i.e., before kernel_map is initialized).
1474  *
1475  *      Maps are in the VM_HIGH_KERNEL_WINDOW area.
1476  */
1477 
1478 vm_map_address_t
pmap_map_high_window_bd(vm_offset_t pa_start,vm_size_t len,vm_prot_t prot)1479 pmap_map_high_window_bd(
1480 	vm_offset_t pa_start,
1481 	vm_size_t len,
1482 	vm_prot_t prot)
1483 {
1484 	pt_entry_t              *ptep, pte;
1485 	vm_map_address_t        va_start = VREGION1_START;
1486 	vm_map_address_t        va_max = VREGION1_START + VREGION1_SIZE;
1487 	vm_map_address_t        va_end;
1488 	vm_map_address_t        va;
1489 	vm_size_t               offset;
1490 
1491 	offset = pa_start & PAGE_MASK;
1492 	pa_start -= offset;
1493 	len += offset;
1494 
1495 	if (len > (va_max - va_start)) {
1496 		panic("%s: area too large, "
1497 		    "pa_start=%p, len=%p, prot=0x%x",
1498 		    __FUNCTION__,
1499 		    (void*)pa_start, (void*)len, prot);
1500 	}
1501 
1502 scan:
1503 	for (; va_start < va_max; va_start += PAGE_SIZE) {
1504 		ptep = pmap_pte(kernel_pmap, va_start);
1505 		assert(!pte_is_compressed(*ptep, ptep));
1506 		if (*ptep == ARM_PTE_TYPE_FAULT) {
1507 			break;
1508 		}
1509 	}
1510 	if (va_start > va_max) {
1511 		panic("%s: insufficient pages, "
1512 		    "pa_start=%p, len=%p, prot=0x%x",
1513 		    __FUNCTION__,
1514 		    (void*)pa_start, (void*)len, prot);
1515 	}
1516 
1517 	for (va_end = va_start + PAGE_SIZE; va_end < va_start + len; va_end += PAGE_SIZE) {
1518 		ptep = pmap_pte(kernel_pmap, va_end);
1519 		assert(!pte_is_compressed(*ptep, ptep));
1520 		if (*ptep != ARM_PTE_TYPE_FAULT) {
1521 			va_start = va_end + PAGE_SIZE;
1522 			goto scan;
1523 		}
1524 	}
1525 
1526 	for (va = va_start; va < va_end; va += PAGE_SIZE, pa_start += PAGE_SIZE) {
1527 		ptep = pmap_pte(kernel_pmap, va);
1528 		pte = pa_to_pte(pa_start)
1529 		    | ARM_PTE_TYPE | ARM_PTE_AF | ARM_PTE_NX | ARM_PTE_PNX
1530 		    | ARM_PTE_AP((prot & VM_PROT_WRITE) ? AP_RWNA : AP_RONA)
1531 		    | ARM_PTE_ATTRINDX(CACHE_ATTRINDX_DEFAULT)
1532 		    | ARM_PTE_SH(SH_OUTER_MEMORY);
1533 #if __ARM_KERNEL_PROTECT__
1534 		pte |= ARM_PTE_NG;
1535 #endif /* __ARM_KERNEL_PROTECT__ */
1536 		__assert_only sptm_return_t ret = sptm_map_page(kernel_pmap->ttep, va, pte);
1537 		assert((ret == SPTM_SUCCESS) || (ret == SPTM_MAP_VALID));
1538 	}
1539 #if KASAN
1540 	kasan_notify_address(va_start, len);
1541 #endif
1542 	return va_start;
1543 }
1544 
1545 /*
1546  * pmap_get_arm64_prot
1547  *
1548  * return effective armv8 VMSA block protections including
1549  * table AP/PXN/XN overrides of a pmap entry
1550  *
1551  */
1552 
1553 uint64_t
pmap_get_arm64_prot(pmap_t pmap,vm_offset_t addr)1554 pmap_get_arm64_prot(
1555 	pmap_t pmap,
1556 	vm_offset_t addr)
1557 {
1558 	tt_entry_t tte = 0;
1559 	unsigned int level = 0;
1560 	uint64_t tte_type = 0;
1561 	uint64_t effective_prot_bits = 0;
1562 	uint64_t aggregate_tte = 0;
1563 	uint64_t table_ap_bits = 0, table_xn = 0, table_pxn = 0;
1564 	const pt_attr_t * const pt_attr = pmap_get_pt_attr(pmap);
1565 
1566 	for (level = pt_attr->pta_root_level; level <= pt_attr->pta_max_level; level++) {
1567 		tte = *pmap_ttne(pmap, level, addr);
1568 
1569 		if (!(tte & ARM_TTE_VALID)) {
1570 			return 0;
1571 		}
1572 
1573 		tte_type = tte & ARM_TTE_TYPE_MASK;
1574 
1575 		if ((tte_type == ARM_TTE_TYPE_BLOCK) ||
1576 		    (level == pt_attr->pta_max_level)) {
1577 			/* Block or page mapping; both have the same protection bit layout. */
1578 			break;
1579 		} else if (tte_type == ARM_TTE_TYPE_TABLE) {
1580 			/* All of the table bits we care about are overrides, so just OR them together. */
1581 			aggregate_tte |= tte;
1582 		}
1583 	}
1584 
1585 	table_ap_bits = ((aggregate_tte >> ARM_TTE_TABLE_APSHIFT) & AP_MASK);
1586 	table_xn = (aggregate_tte & ARM_TTE_TABLE_XN);
1587 	table_pxn = (aggregate_tte & ARM_TTE_TABLE_PXN);
1588 
1589 	/* Start with the PTE bits. */
1590 	effective_prot_bits = tte & (ARM_PTE_APMASK | ARM_PTE_NX | ARM_PTE_PNX);
1591 
1592 	/* Table AP bits mask out block/page AP bits */
1593 	effective_prot_bits &= ~(ARM_PTE_AP(table_ap_bits));
1594 
1595 	/* XN/PXN bits can be OR'd in. */
1596 	effective_prot_bits |= (table_xn ? ARM_PTE_NX : 0);
1597 	effective_prot_bits |= (table_pxn ? ARM_PTE_PNX : 0);
1598 
1599 	return effective_prot_bits;
1600 }
1601 
1602 /*
1603  *	Bootstrap the system enough to run with virtual memory.
1604  *
1605  *	The early VM initialization code has already allocated
1606  *	the first CPU's translation table and made entries for
1607  *	all the one-to-one mappings to be found there.
1608  *
1609  *	We must set up the kernel pmap structures, the
1610  *	physical-to-virtual translation lookup tables for the
1611  *	physical memory to be managed (between avail_start and
1612  *	avail_end).
1613  *
1614  *	Map the kernel's code and data, and allocate the system page table.
1615  *	Page_size must already be set.
1616  *
1617  *	Parameters:
1618  *	first_avail	first available physical page -
1619  *			   after kernel page tables
1620  *	avail_start	PA of first managed physical page
1621  *	avail_end	PA of last managed physical page
1622  */
1623 
1624 void
pmap_bootstrap(vm_offset_t vstart)1625 pmap_bootstrap(
1626 	vm_offset_t vstart)
1627 {
1628 	vm_map_offset_t maxoffset;
1629 
1630 	lck_grp_init(&pmap_lck_grp, "pmap", LCK_GRP_ATTR_NULL);
1631 
1632 #if DEVELOPMENT || DEBUG
1633 	if (PE_parse_boot_argn("pmap_trace", &pmap_trace_mask, sizeof(pmap_trace_mask))) {
1634 		kprintf("Kernel traces for pmap operations enabled\n");
1635 	}
1636 #endif
1637 
1638 	/*
1639 	 *	Initialize the kernel pmap.
1640 	 */
1641 #if ARM_PARAMETERIZED_PMAP
1642 	kernel_pmap->pmap_pt_attr = native_pt_attr;
1643 #endif /* ARM_PARAMETERIZED_PMAP */
1644 #if HAS_APPLE_PAC
1645 	kernel_pmap->disable_jop = 0;
1646 #endif /* HAS_APPLE_PAC */
1647 	kernel_pmap->tte = cpu_tte;
1648 	kernel_pmap->ttep = cpu_ttep;
1649 	kernel_pmap->min = UINT64_MAX - (1ULL << (64 - T1SZ_BOOT)) + 1;
1650 	kernel_pmap->max = UINTPTR_MAX;
1651 	os_ref_init_count_raw(&kernel_pmap->ref_count, &pmap_refgrp, 1);
1652 	kernel_pmap->nx_enabled = TRUE;
1653 	kernel_pmap->is_64bit = TRUE;
1654 #if CONFIG_ROSETTA
1655 	kernel_pmap->is_rosetta = FALSE;
1656 #endif
1657 
1658 #if ARM_PARAMETERIZED_PMAP
1659 	kernel_pmap->pmap_pt_attr = native_pt_attr;
1660 #endif /* ARM_PARAMETERIZED_PMAP */
1661 
1662 	kernel_pmap->nested_region_addr = 0x0ULL;
1663 	kernel_pmap->nested_region_size = 0x0ULL;
1664 	kernel_pmap->nested_region_unnested_table_bitmap = NULL;
1665 	kernel_pmap->type = PMAP_TYPE_KERNEL;
1666 
1667 	kernel_pmap->asid = 0;
1668 
1669 	pmap_lock_init(kernel_pmap);
1670 
1671 	pmap_max_asids = SPTMArgs->num_asids;
1672 
1673 	const vm_size_t asid_table_size = sizeof(*asid_bitmap) * BITMAP_LEN(pmap_max_asids);
1674 
1675 	/**
1676 	 * Bootstrap the core pmap data structures (e.g., pv_head_table,
1677 	 * pp_attr_table, etc). This function will use `avail_start` to allocate
1678 	 * space for these data structures.
1679 	 * */
1680 	pmap_data_bootstrap();
1681 
1682 	/**
1683 	 * Bootstrap any necessary UAT data structures and values needed from the device tree.
1684 	 */
1685 	uat_bootstrap();
1686 
1687 	/**
1688 	 * Don't make any assumptions about the alignment of avail_start before this
1689 	 * point (i.e., pmap_data_bootstrap() performs allocations).
1690 	 */
1691 	avail_start = PMAP_ALIGN(avail_start, __alignof(bitmap_t));
1692 
1693 	const pmap_paddr_t pmap_struct_start = avail_start;
1694 
1695 	asid_bitmap = (bitmap_t*)phystokv(avail_start);
1696 	avail_start = round_page(avail_start + asid_table_size);
1697 
1698 	memset((char *)phystokv(pmap_struct_start), 0, avail_start - pmap_struct_start);
1699 
1700 	queue_init(&map_pmap_list);
1701 	queue_enter(&map_pmap_list, kernel_pmap, pmap_t, pmaps);
1702 
1703 	virtual_space_start = vstart;
1704 	virtual_space_end = VM_MAX_KERNEL_ADDRESS;
1705 
1706 	bitmap_full(&asid_bitmap[0], pmap_max_asids);
1707 	// Clear the ASIDs which will alias the reserved kernel ASID of 0
1708 	for (unsigned int i = 0; i < pmap_max_asids; i += MAX_HW_ASIDS) {
1709 		bitmap_clear(&asid_bitmap[0], i);
1710 	}
1711 
1712 #if !HAS_16BIT_ASID
1713 	/**
1714 	 * Align the range of available hardware ASIDs to a multiple of 64 to enable the
1715 	 * masking used by the PLRU scheme.  This means we must handle the case in which
1716 	 * the returned hardware ASID is 0, which we do by clearing all vASIDs that will
1717 	 * alias the kernel ASID.
1718 	 */
1719 	pmap_max_asids = pmap_max_asids & ~63ul;
1720 	if (__improbable(pmap_max_asids == 0)) {
1721 		panic("%s: insufficient number of ASIDs (%u) supplied by SPTM", __func__, (unsigned int)pmap_max_asids);
1722 	}
1723 	pmap_asid_plru = (pmap_max_asids > MAX_HW_ASIDS);
1724 	PE_parse_boot_argn("pmap_asid_plru", &pmap_asid_plru, sizeof(pmap_asid_plru));
1725 	_Static_assert(sizeof(asid_plru_bitmap[0] == sizeof(uint64_t)), "bitmap_t is not a 64-bit integer");
1726 	_Static_assert((MAX_HW_ASIDS % 64) == 0, "MAX_HW_ASIDS is not divisible by 64");
1727 	bitmap_full(&asid_plru_bitmap[0], MAX_HW_ASIDS);
1728 	bitmap_clear(&asid_plru_bitmap[0], 0);
1729 #endif /* !HAS_16BIT_ASID */
1730 
1731 
1732 	if (PE_parse_boot_argn("arm_maxoffset", &maxoffset, sizeof(maxoffset))) {
1733 		maxoffset = trunc_page(maxoffset);
1734 		if ((maxoffset >= pmap_max_offset(FALSE, ARM_PMAP_MAX_OFFSET_MIN))
1735 		    && (maxoffset <= pmap_max_offset(FALSE, ARM_PMAP_MAX_OFFSET_MAX))) {
1736 			arm_pmap_max_offset_default = maxoffset;
1737 		}
1738 	}
1739 	if (PE_parse_boot_argn("arm64_maxoffset", &maxoffset, sizeof(maxoffset))) {
1740 		maxoffset = trunc_page(maxoffset);
1741 		if ((maxoffset >= pmap_max_offset(TRUE, ARM_PMAP_MAX_OFFSET_MIN))
1742 		    && (maxoffset <= pmap_max_offset(TRUE, ARM_PMAP_MAX_OFFSET_MAX))) {
1743 			arm64_pmap_max_offset_default = maxoffset;
1744 		}
1745 	}
1746 
1747 	PE_parse_boot_argn("pmap_panic_dev_wimg_on_managed", &pmap_panic_dev_wimg_on_managed, sizeof(pmap_panic_dev_wimg_on_managed));
1748 
1749 
1750 #if DEVELOPMENT || DEBUG
1751 	PE_parse_boot_argn("vm_footprint_suspend_allowed",
1752 	    &vm_footprint_suspend_allowed,
1753 	    sizeof(vm_footprint_suspend_allowed));
1754 #endif /* DEVELOPMENT || DEBUG */
1755 
1756 #if KASAN
1757 	/* Shadow the CPU copy windows, as they fall outside of the physical aperture */
1758 	kasan_map_shadow(CPUWINDOWS_BASE, CPUWINDOWS_TOP - CPUWINDOWS_BASE, true);
1759 #endif /* KASAN */
1760 
1761 	/**
1762 	 * Ensure that avail_start is always left on a page boundary. The calling
1763 	 * code might not perform any alignment before allocating page tables so
1764 	 * this is important.
1765 	 */
1766 	avail_start = round_page(avail_start);
1767 
1768 
1769 #if (DEVELOPMENT || DEBUG)
1770 	sptm_features_available(SPTM_FEATURE_SYSREG, &sptm_sysreg_available);
1771 #endif /* (DEVELOPMENT || DEBUG) */
1772 
1773 	/* Signal that the pmap has been bootstrapped */
1774 	pmap_bootstrapped = true;
1775 }
1776 
1777 /**
1778  * Helper for creating a populated commpage table
1779  *
1780  * In order to avoid burning extra pages on mapping the commpage, we create a
1781  * dedicated table hierarchy for the commpage.  We forcibly nest the translation tables from
1782  * this pmap into other pmaps.  The level we will nest at depends on the MMU configuration (page
1783  * size, TTBR range, etc). Typically, this is at L1 for 4K tasks and L2 for 16K tasks.
1784  *
1785  * @note that this is NOT "the nested pmap" (which is used to nest the shared cache).
1786  *
1787  * @param rw_va Virtual address at which to insert a mapping to the kernel R/W commpage
1788  * @param ro_va Virtual address at which to insert a mapping to the kernel R/O commpage
1789  * @param rw_pa Physical address of kernel R/W commpage
1790  * @param ro_pa Physical address of kernel R/O commpage, may be 0 if not supported in this
1791  *              configuration
1792  * @param rx_pa Physical address of user executable (and kernel R/O) commpage, may be 0 if
1793  *              not supported in this configuration
1794  * @param pmap_create_flags Control flags for the temporary pmap created by this function
1795  *
1796  * @return the physical address of the created commpage table, typed as
1797  *         XNU_PAGE_TABLE_COMMPAGE and containing all relevant commpage mappings.
1798  */
1799 static pmap_paddr_t
pmap_create_commpage_table(vm_map_address_t rw_va,vm_map_address_t ro_va,pmap_paddr_t rw_pa,pmap_paddr_t ro_pa,pmap_paddr_t rx_pa,unsigned int pmap_create_flags)1800 pmap_create_commpage_table(vm_map_address_t rw_va, vm_map_address_t ro_va,
1801     pmap_paddr_t rw_pa, pmap_paddr_t ro_pa, pmap_paddr_t rx_pa, unsigned int pmap_create_flags)
1802 {
1803 	pmap_t temp_commpage_pmap = pmap_create_options(NULL, 0, pmap_create_flags);
1804 	assert(temp_commpage_pmap != NULL);
1805 	assert(rw_pa != 0);
1806 	const pt_attr_t *pt_attr = pmap_get_pt_attr(temp_commpage_pmap);
1807 
1808 	/*
1809 	 * We only use pmap_expand to expand the pmap up to the commpage nesting level.  At that level
1810 	 * and beyond, all the newly created tables will be nested directly into the userspace region
1811 	 * for each process, and as such they must be of the dedicated SPTM commpage table type so that
1812 	 * the SPTM can enforce the commpage security model which forbids random replacement of commpage
1813 	 * mappings.
1814 	 */
1815 	kern_return_t kr = pmap_expand(temp_commpage_pmap, rw_va, 0, pt_attr_commpage_level(pt_attr));
1816 	assert(kr == KERN_SUCCESS);
1817 
1818 	pmap_paddr_t commpage_table_pa = 0;
1819 	for (unsigned int i = pt_attr_commpage_level(pt_attr); i < pt_attr_leaf_level(pt_attr); i++) {
1820 		pmap_paddr_t new_table = 0;
1821 		kr = pmap_page_alloc(&new_table, 0);
1822 		assert((kr == KERN_SUCCESS) && (new_table != 0));
1823 		if (commpage_table_pa == 0) {
1824 			commpage_table_pa = new_table;
1825 		}
1826 
1827 		sptm_retype_params_t retype_params = {.raw = SPTM_RETYPE_PARAMS_NULL};
1828 		retype_params.level = (sptm_pt_level_t)pt_attr_leaf_level(pt_attr);
1829 		sptm_retype(new_table, XNU_DEFAULT, XNU_PAGE_TABLE_COMMPAGE, retype_params);
1830 
1831 		const sptm_tte_t table_tte = (new_table & ARM_TTE_TABLE_MASK) | ARM_TTE_TYPE_TABLE | ARM_TTE_VALID;
1832 
1833 		sptm_map_table(temp_commpage_pmap->ttep, pt_attr_align_va(pt_attr, i, rw_va),
1834 		    (sptm_pt_level_t)i, table_tte);
1835 	}
1836 
1837 	/*
1838 	 * Note the lack of ARM_PTE_NG here: commpage mappings are at fixed addresses and
1839 	 * frequently accessed, so we map them global to avoid unnecessary TLB pressure.
1840 	 */
1841 	static const sptm_pte_t commpage_pte_template = ARM_PTE_TYPE_VALID
1842 	    | ARM_PTE_ATTRINDX(CACHE_ATTRINDX_WRITEBACK)
1843 	    | ARM_PTE_SH(SH_INNER_MEMORY) | ARM_PTE_PNX
1844 	    | ARM_PTE_AP(AP_RORO) | ARM_PTE_AF;
1845 
1846 	sptm_return_t sptm_ret = sptm_map_page(temp_commpage_pmap->ttep, rw_va,
1847 	    commpage_pte_template | ARM_PTE_NX | pa_to_pte(rw_pa));
1848 	assert(sptm_ret == SPTM_SUCCESS);
1849 
1850 	if (ro_pa != 0) {
1851 		assert((ro_va & ~pt_attr_twig_offmask(pt_attr)) == (rw_va & ~pt_attr_twig_offmask(pt_attr)));
1852 		sptm_ret = sptm_map_page(temp_commpage_pmap->ttep, ro_va,
1853 		    commpage_pte_template | ARM_PTE_NX | pa_to_pte(ro_pa));
1854 		assert(sptm_ret == SPTM_SUCCESS);
1855 	}
1856 
1857 	if (rx_pa != 0) {
1858 		assert((commpage_text_user_va & ~pt_attr_twig_offmask(pt_attr)) == (rw_va & ~pt_attr_twig_offmask(pt_attr)));
1859 		assert((commpage_text_user_va != rw_va) && (commpage_text_user_va != ro_va));
1860 		sptm_ret = sptm_map_page(temp_commpage_pmap->ttep, commpage_text_user_va, commpage_pte_template | pa_to_pte(rx_pa));
1861 		assert(sptm_ret == SPTM_SUCCESS);
1862 	}
1863 
1864 	sptm_unmap_table(temp_commpage_pmap->ttep, pt_attr_align_va(pt_attr, pt_attr_commpage_level(pt_attr), rw_va),
1865 	    (sptm_pt_level_t)pt_attr_commpage_level(pt_attr));
1866 	pmap_destroy(temp_commpage_pmap);
1867 
1868 	return commpage_table_pa;
1869 }
1870 
1871 /**
1872  * Helper for creating all commpage tables applicable to the current configuration.
1873  *
1874  * @note This function is intended to be called during bootstrap.
1875  * @note This function assumes that pmap_create_commpages has already executed, and therefore
1876  *       the commpage_*_pa variables have been assigned to their final values.  commpage_data_pa
1877  *       is the kernel RW commpage and is assumed to be present on all configurations, so it
1878  *       therefore must be non-zero at this point.  The other variables are considered optional
1879  *       depending upon configuration and may be zero.
1880  */
1881 void pmap_prepare_commpages(void);
1882 void
pmap_prepare_commpages(void)1883 pmap_prepare_commpages(void)
1884 {
1885 	sptm_retype_params_t retype_params = {.raw = SPTM_RETYPE_PARAMS_NULL};
1886 	assert(commpage_data_pa != 0);
1887 	sptm_retype(commpage_data_pa, XNU_DEFAULT, XNU_COMMPAGE_RW, retype_params);
1888 	if (commpage_ro_data_pa != 0) {
1889 		sptm_retype(commpage_ro_data_pa, XNU_DEFAULT, XNU_COMMPAGE_RO, retype_params);
1890 	}
1891 	if (commpage_text_pa != 0) {
1892 		sptm_retype(commpage_text_pa, XNU_DEFAULT, XNU_COMMPAGE_RX, retype_params);
1893 	}
1894 
1895 	/*
1896 	 * User mapping of comm page text section for 64 bit mapping only
1897 	 *
1898 	 * We don't insert the text commpage into the 32 bit mapping because we don't want
1899 	 * 32-bit user processes to get this page mapped in, they should never call into
1900 	 * this page.
1901 	 */
1902 	commpage_default_table = pmap_create_commpage_table(_COMM_PAGE64_BASE_ADDRESS, _COMM_PAGE64_RO_ADDRESS,
1903 	    commpage_data_pa, commpage_ro_data_pa, commpage_text_pa, 0);
1904 
1905 	/*
1906 	 * SPTM TODO: Enable this, along with the appropriate 32-bit commpage address checks and flushes in the
1907 	 * SPTM, if we ever need to support arm64_32 processes in the SPTM.
1908 	 *
1909 	 * commpage32_default_table = pmap_create_commpage_table(_COMM_PAGE32_BASE_ADDRESS, _COMM_PAGE32_RO_ADDRESS,
1910 	 *    commpage_data_pa, commpage_ro_data_pa, 0, 0);
1911 	 */
1912 #if __ARM_MIXED_PAGE_SIZE__
1913 	commpage_4k_table = pmap_create_commpage_table(_COMM_PAGE64_BASE_ADDRESS, _COMM_PAGE64_RO_ADDRESS,
1914 	    commpage_data_pa, commpage_ro_data_pa, 0, PMAP_CREATE_FORCE_4K_PAGES);
1915 
1916 	/*
1917 	 * SPTM TODO: Enable this, along with the appropriate 32-bit commpage address checks and flushes in the
1918 	 * SPTM, if we ever need to support arm64_32 processes in the SPTM.
1919 	 * commpage32_4k_table = pmap_create_commpage_table(_COMM_PAGE32_BASE_ADDRESS, _COMM_PAGE32_RO_ADDRESS,
1920 	 *    commpage_data_pa, commpage_ro_data_pa, 0, PMAP_CREATE_FORCE_4K_PAGES);
1921 	 */
1922 #endif /* __ARM_MIXED_PAGE_SIZE__ */
1923 
1924 }
1925 
1926 void
pmap_virtual_space(vm_offset_t * startp,vm_offset_t * endp)1927 pmap_virtual_space(
1928 	vm_offset_t *startp,
1929 	vm_offset_t *endp
1930 	)
1931 {
1932 	*startp = virtual_space_start;
1933 	*endp = virtual_space_end;
1934 }
1935 
1936 
1937 boolean_t
pmap_virtual_region(unsigned int region_select,vm_map_offset_t * startp,vm_map_size_t * size)1938 pmap_virtual_region(
1939 	unsigned int region_select,
1940 	vm_map_offset_t *startp,
1941 	vm_map_size_t *size
1942 	)
1943 {
1944 	boolean_t       ret = FALSE;
1945 #if defined(KERNEL_INTEGRITY_KTRR) || defined(KERNEL_INTEGRITY_CTRR)
1946 	if (region_select == 0) {
1947 		/*
1948 		 * In this config, the bootstrap mappings should occupy their own L2
1949 		 * TTs, as they should be immutable after boot.  Having the associated
1950 		 * TTEs and PTEs in their own pages allows us to lock down those pages,
1951 		 * while allowing the rest of the kernel address range to be remapped.
1952 		 */
1953 		*startp = LOW_GLOBAL_BASE_ADDRESS & ~ARM_TT_L2_OFFMASK;
1954 #if defined(ARM_LARGE_MEMORY)
1955 		*size = ((KERNEL_PMAP_HEAP_RANGE_START - *startp) & ~PAGE_MASK);
1956 #else
1957 		*size = ((VM_MAX_KERNEL_ADDRESS - *startp) & ~PAGE_MASK);
1958 #endif
1959 		ret = TRUE;
1960 	}
1961 
1962 #if defined(ARM_LARGE_MEMORY)
1963 	if (region_select == 1) {
1964 		*startp = VREGION1_START;
1965 		*size = VREGION1_SIZE;
1966 		ret = TRUE;
1967 	}
1968 #endif
1969 #else /* !(defined(KERNEL_INTEGRITY_KTRR) || defined(KERNEL_INTEGRITY_CTRR)) */
1970 #if defined(ARM_LARGE_MEMORY)
1971 	/* For large memory systems with no KTRR/CTRR such as virtual machines */
1972 	if (region_select == 0) {
1973 		*startp = LOW_GLOBAL_BASE_ADDRESS & ~ARM_TT_L2_OFFMASK;
1974 		*size = ((KERNEL_PMAP_HEAP_RANGE_START - *startp) & ~PAGE_MASK);
1975 		ret = TRUE;
1976 	}
1977 
1978 	if (region_select == 1) {
1979 		*startp = VREGION1_START;
1980 		*size = VREGION1_SIZE;
1981 		ret = TRUE;
1982 	}
1983 #else /* !defined(ARM_LARGE_MEMORY) */
1984 	unsigned long low_global_vr_mask = 0;
1985 	vm_map_size_t low_global_vr_size = 0;
1986 
1987 	if (region_select == 0) {
1988 		/* Round to avoid overlapping with the V=P area; round to at least the L2 block size. */
1989 		if (!TEST_PAGE_SIZE_4K) {
1990 			*startp = gVirtBase & 0xFFFFFFFFFE000000;
1991 			*size = ((virtual_space_start - (gVirtBase & 0xFFFFFFFFFE000000)) + ~0xFFFFFFFFFE000000) & 0xFFFFFFFFFE000000;
1992 		} else {
1993 			*startp = gVirtBase & 0xFFFFFFFFFF800000;
1994 			*size = ((virtual_space_start - (gVirtBase & 0xFFFFFFFFFF800000)) + ~0xFFFFFFFFFF800000) & 0xFFFFFFFFFF800000;
1995 		}
1996 		ret = TRUE;
1997 	}
1998 	if (region_select == 1) {
1999 		*startp = VREGION1_START;
2000 		*size = VREGION1_SIZE;
2001 		ret = TRUE;
2002 	}
2003 	/* We need to reserve a range that is at least the size of an L2 block mapping for the low globals */
2004 	if (!TEST_PAGE_SIZE_4K) {
2005 		low_global_vr_mask = 0xFFFFFFFFFE000000;
2006 		low_global_vr_size = 0x2000000;
2007 	} else {
2008 		low_global_vr_mask = 0xFFFFFFFFFF800000;
2009 		low_global_vr_size = 0x800000;
2010 	}
2011 
2012 	if (((gVirtBase & low_global_vr_mask) != LOW_GLOBAL_BASE_ADDRESS) && (region_select == 2)) {
2013 		*startp = LOW_GLOBAL_BASE_ADDRESS;
2014 		*size = low_global_vr_size;
2015 		ret = TRUE;
2016 	}
2017 
2018 	if (region_select == 3) {
2019 		/* In this config, we allow the bootstrap mappings to occupy the same
2020 		 * page table pages as the heap.
2021 		 */
2022 		*startp = VM_MIN_KERNEL_ADDRESS;
2023 		*size = LOW_GLOBAL_BASE_ADDRESS - *startp;
2024 		ret = TRUE;
2025 	}
2026 #endif /* defined(ARM_LARGE_MEMORY) */
2027 #endif /* defined(KERNEL_INTEGRITY_KTRR) || defined(KERNEL_INTEGRITY_CTRR) */
2028 	return ret;
2029 }
2030 
2031 /*
2032  * Routines to track and allocate physical pages during early boot.
2033  * On most systems that memory runs from first_avail through to avail_end
2034  * with no gaps.
2035  *
2036  * If the system supports ECC and ecc_bad_pages_count > 0, we
2037  * need to skip those pages.
2038  */
2039 
2040 static unsigned int avail_page_count = 0;
2041 static bool need_ram_ranges_init = true;
2042 
2043 
2044 /**
2045  * Checks to see if a given page is in
2046  * the array of known bad pages
2047  *
2048  * @param ppn page number to check
2049  */
2050 bool
pmap_is_bad_ram(__unused ppnum_t ppn)2051 pmap_is_bad_ram(__unused ppnum_t ppn)
2052 {
2053 	return false;
2054 }
2055 
2056 /**
2057  * Prepare bad ram pages to be skipped.
2058  */
2059 
2060 
2061 /*
2062  * Initialize the count of available pages. No lock needed here,
2063  * as this code is called while kernel boot up is single threaded.
2064  */
2065 static void
initialize_ram_ranges(void)2066 initialize_ram_ranges(void)
2067 {
2068 	pmap_paddr_t first = first_avail;
2069 	pmap_paddr_t end = avail_end;
2070 
2071 	assert(first <= end);
2072 	assert(first == (first & ~PAGE_MASK));
2073 	assert(end == (end & ~PAGE_MASK));
2074 	avail_page_count = atop(end - first);
2075 
2076 	need_ram_ranges_init = false;
2077 
2078 }
2079 
2080 unsigned int
pmap_free_pages(void)2081 pmap_free_pages(
2082 	void)
2083 {
2084 	if (need_ram_ranges_init) {
2085 		initialize_ram_ranges();
2086 	}
2087 	return avail_page_count;
2088 }
2089 
2090 unsigned int
pmap_free_pages_span(void)2091 pmap_free_pages_span(
2092 	void)
2093 {
2094 	if (need_ram_ranges_init) {
2095 		initialize_ram_ranges();
2096 	}
2097 	return (unsigned int)atop(avail_end - first_avail);
2098 }
2099 
2100 
2101 boolean_t
pmap_next_page_hi(ppnum_t * pnum,__unused boolean_t might_free)2102 pmap_next_page_hi(
2103 	ppnum_t            * pnum,
2104 	__unused boolean_t might_free)
2105 {
2106 	return pmap_next_page(pnum);
2107 }
2108 
2109 
2110 boolean_t
pmap_next_page(ppnum_t * pnum)2111 pmap_next_page(
2112 	ppnum_t *pnum)
2113 {
2114 	if (need_ram_ranges_init) {
2115 		initialize_ram_ranges();
2116 	}
2117 
2118 
2119 	if (first_avail != avail_end) {
2120 		*pnum = (ppnum_t)atop(first_avail);
2121 		first_avail += PAGE_SIZE;
2122 		assert(avail_page_count > 0);
2123 		--avail_page_count;
2124 		return TRUE;
2125 	}
2126 	assert(avail_page_count == 0);
2127 	return FALSE;
2128 }
2129 
2130 
2131 
2132 
2133 /*
2134  *	Initialize the pmap module.
2135  *	Called by vm_init, to initialize any structures that the pmap
2136  *	system needs to map virtual memory.
2137  */
2138 void
pmap_init(void)2139 pmap_init(
2140 	void)
2141 {
2142 	/*
2143 	 *	Protect page zero in the kernel map.
2144 	 *	(can be overruled by permanent transltion
2145 	 *	table entries at page zero - see arm_vm_init).
2146 	 */
2147 	vm_protect(kernel_map, 0, PAGE_SIZE, TRUE, VM_PROT_NONE);
2148 
2149 	pmap_initialized = TRUE;
2150 
2151 	/*
2152 	 *	Create the zone of physical maps
2153 	 *	and the physical-to-virtual entries.
2154 	 */
2155 	pmap_zone = zone_create_ext("pmap", sizeof(struct pmap),
2156 	    ZC_ZFREE_CLEARMEM, ZONE_ID_PMAP, NULL);
2157 
2158 
2159 	/*
2160 	 *	Initialize the pmap object (for tracking the vm_page_t
2161 	 *	structures for pages we allocate to be page tables in
2162 	 *	pmap_expand().
2163 	 */
2164 	_vm_object_allocate(mem_size, pmap_object);
2165 	pmap_object->copy_strategy = MEMORY_OBJECT_COPY_NONE;
2166 
2167 	/*
2168 	 *	Initialize the TXM VM object in the same way as the
2169 	 *	PMAP VM object.
2170 	 */
2171 	_vm_object_allocate(mem_size, txm_vm_object);
2172 	txm_vm_object->copy_strategy = MEMORY_OBJECT_COPY_NONE;
2173 
2174 	/*
2175 	 * The values of [hard_]maxproc may have been scaled, make sure
2176 	 * they are still less than the value of pmap_max_asids.
2177 	 */
2178 	if ((uint32_t)maxproc > pmap_max_asids) {
2179 		maxproc = pmap_max_asids;
2180 	}
2181 	if ((uint32_t)hard_maxproc > pmap_max_asids) {
2182 		hard_maxproc = pmap_max_asids;
2183 	}
2184 }
2185 
2186 /**
2187  * Verify that a given physical page contains no mappings (outside of the
2188  * default physical aperture mapping).
2189  *
2190  * @param ppnum Physical page number to check there are no mappings to.
2191  *
2192  * @return True if there are no mappings, false otherwise or if the page is not
2193  *         kernel-managed.
2194  */
2195 bool
pmap_verify_free(ppnum_t ppnum)2196 pmap_verify_free(ppnum_t ppnum)
2197 {
2198 	const pmap_paddr_t pa = ptoa(ppnum);
2199 
2200 	assert(pa != vm_page_fictitious_addr);
2201 
2202 	/* Only mappings to kernel-managed physical memory are tracked. */
2203 	if (!pa_valid(pa)) {
2204 		return false;
2205 	}
2206 
2207 	const unsigned int pai = pa_index(pa);
2208 
2209 	return pvh_test_type(pai_to_pvh(pai), PVH_TYPE_NULL);
2210 }
2211 
2212 #if MACH_ASSERT
2213 /**
2214  * Verify that a given physical page contains no mappings (outside of the
2215  * default physical aperture mapping) and if it does, then panic.
2216  *
2217  * @note It's recommended to use pmap_verify_free() directly when operating in
2218  *       the PPL since the PVH lock isn't getting grabbed here (due to this code
2219  *       normally being called from outside of the PPL, and the pv_head_table
2220  *       can't be modified outside of the PPL).
2221  *
2222  * @param ppnum Physical page number to check there are no mappings to.
2223  */
2224 void
pmap_assert_free(ppnum_t ppnum)2225 pmap_assert_free(ppnum_t ppnum)
2226 {
2227 	const pmap_paddr_t pa = ptoa(ppnum);
2228 
2229 	/* Only mappings to kernel-managed physical memory are tracked. */
2230 	if (__probable(!pa_valid(pa) || pmap_verify_free(ppnum))) {
2231 		return;
2232 	}
2233 
2234 	const unsigned int pai = pa_index(pa);
2235 	const uintptr_t pvh = pai_to_pvh(pai);
2236 
2237 	/**
2238 	 * This function is always called from outside of the PPL. Because of this,
2239 	 * the PVH entry can't be locked. This function is generally only called
2240 	 * before the VM reclaims a physical page and shouldn't be creating new
2241 	 * mappings. Even if a new mapping is created while parsing the hierarchy,
2242 	 * the worst case is that the system will panic in another way, and we were
2243 	 * already about to panic anyway.
2244 	 */
2245 
2246 	/**
2247 	 * Since pmap_verify_free() returned false, that means there is at least one
2248 	 * mapping left. Let's get some extra info on the first mapping we find to
2249 	 * dump in the panic string (the common case is that there is one spare
2250 	 * mapping that was never unmapped).
2251 	 */
2252 	pt_entry_t *first_ptep = PT_ENTRY_NULL;
2253 
2254 	if (pvh_test_type(pvh, PVH_TYPE_PTEP)) {
2255 		first_ptep = pvh_ptep(pvh);
2256 	} else if (pvh_test_type(pvh, PVH_TYPE_PVEP)) {
2257 		pv_entry_t *pvep = pvh_pve_list(pvh);
2258 
2259 		/* Each PVE can contain multiple PTEs. Let's find the first one. */
2260 		for (int pve_ptep_idx = 0; pve_ptep_idx < PTE_PER_PVE; pve_ptep_idx++) {
2261 			first_ptep = pve_get_ptep(pvep, pve_ptep_idx);
2262 			if (first_ptep != PT_ENTRY_NULL) {
2263 				break;
2264 			}
2265 		}
2266 
2267 		/* The PVE should have at least one valid PTE. */
2268 		assert(first_ptep != PT_ENTRY_NULL);
2269 	} else if (pvh_test_type(pvh, PVH_TYPE_PTDP)) {
2270 		panic("%s: Physical page is being used as a page table at PVH %p (pai: %d)",
2271 		    __func__, (void*)pvh, pai);
2272 	} else {
2273 		/**
2274 		 * The mapping disappeared between here and the pmap_verify_free() call.
2275 		 * The only way that can happen is if the VM was racing this call with
2276 		 * a call that unmaps PTEs. Operations on this page should not be
2277 		 * occurring at the same time as this check, and unfortunately we can't
2278 		 * lock the PVH entry to prevent it, so just panic instead.
2279 		 */
2280 		panic("%s: Mapping was detected but is now gone. Is the VM racing this "
2281 		    "call with an operation that unmaps PTEs? PVH %p (pai: %d)",
2282 		    __func__, (void*)pvh, pai);
2283 	}
2284 
2285 	/* Panic with a unique string identifying the first bad mapping and owner. */
2286 	{
2287 		/* First PTE is mapped by the main CPUs. */
2288 		pmap_t pmap = ptep_get_pmap(first_ptep);
2289 		const char *type = (pmap == kernel_pmap) ? "Kernel" : "User";
2290 
2291 		panic("%s: Found at least one mapping to %#llx. First PTEP (%p) is a "
2292 		    "%s CPU mapping (pmap: %p)",
2293 		    __func__, (uint64_t)pa, first_ptep, type, pmap);
2294 	}
2295 }
2296 #endif
2297 
2298 
2299 static vm_size_t
pmap_root_alloc_size(pmap_t pmap)2300 pmap_root_alloc_size(pmap_t pmap)
2301 {
2302 #pragma unused(pmap)
2303 	const pt_attr_t * const pt_attr = pmap_get_pt_attr(pmap);
2304 	unsigned int root_level = pt_attr_root_level(pt_attr);
2305 	return ((pt_attr_ln_index_mask(pt_attr, root_level) >> pt_attr_ln_shift(pt_attr, root_level)) + 1) * sizeof(tt_entry_t);
2306 }
2307 
2308 /*
2309  *	Create and return a physical map.
2310  *
2311  *	If the size specified for the map
2312  *	is zero, the map is an actual physical
2313  *	map, and may be referenced by the
2314  *	hardware.
2315  *
2316  *	If the size specified is non-zero,
2317  *	the map will be used in software only, and
2318  *	is bounded by that size.
2319  */
2320 MARK_AS_PMAP_TEXT pmap_t
pmap_create_options_internal(ledger_t ledger,vm_map_size_t size,unsigned int flags,kern_return_t * kr)2321 pmap_create_options_internal(
2322 	ledger_t ledger,
2323 	vm_map_size_t size,
2324 	unsigned int flags,
2325 	kern_return_t *kr)
2326 {
2327 	pmap_t          p;
2328 	bool is_64bit = flags & PMAP_CREATE_64BIT;
2329 #if defined(HAS_APPLE_PAC)
2330 	bool disable_jop = flags & PMAP_CREATE_DISABLE_JOP;
2331 #endif /* defined(HAS_APPLE_PAC) */
2332 	kern_return_t   local_kr = KERN_SUCCESS;
2333 	__unused uint8_t sptm_root_flags = SPTM_ROOT_PT_FLAGS_DEFAULT;
2334 	TXMAddressSpaceFlags_t txm_flags = kTXMAddressSpaceFlagInit;
2335 	const bool is_stage2 = false;
2336 
2337 	if (size != 0) {
2338 		{
2339 			// Size parameter should only be set for stage 2.
2340 			return PMAP_NULL;
2341 		}
2342 	}
2343 
2344 	if (0 != (flags & ~PMAP_CREATE_KNOWN_FLAGS)) {
2345 		return PMAP_NULL;
2346 	}
2347 
2348 	/*
2349 	 *	Allocate a pmap struct from the pmap_zone.  Then allocate
2350 	 *	the translation table of the right size for the pmap.
2351 	 */
2352 	if ((p = (pmap_t) zalloc(pmap_zone)) == PMAP_NULL) {
2353 		local_kr = KERN_RESOURCE_SHORTAGE;
2354 		goto pmap_create_fail;
2355 	}
2356 
2357 	p->ledger = ledger;
2358 
2359 
2360 	p->pmap_vm_map_cs_enforced = false;
2361 	p->min = 0;
2362 
2363 
2364 #if CONFIG_ROSETTA
2365 	if (flags & PMAP_CREATE_ROSETTA) {
2366 		p->is_rosetta = TRUE;
2367 	} else {
2368 		p->is_rosetta = FALSE;
2369 	}
2370 #endif /* CONFIG_ROSETTA */
2371 #if defined(HAS_APPLE_PAC)
2372 	p->disable_jop = disable_jop;
2373 
2374 	if (p->disable_jop) {
2375 		sptm_root_flags &= ~SPTM_ROOT_PT_FLAG_JOP;
2376 	}
2377 #endif /* defined(HAS_APPLE_PAC) */
2378 
2379 	p->nested_region_true_start = 0;
2380 	p->nested_region_true_end = ~0;
2381 
2382 	p->nx_enabled = true;
2383 	p->is_64bit = is_64bit;
2384 	p->nested_pmap = PMAP_NULL;
2385 	p->type = PMAP_TYPE_USER;
2386 
2387 #if ARM_PARAMETERIZED_PMAP
2388 	/* Default to the native pt_attr */
2389 	p->pmap_pt_attr = native_pt_attr;
2390 #endif /* ARM_PARAMETERIZED_PMAP */
2391 #if __ARM_MIXED_PAGE_SIZE__
2392 	if (flags & PMAP_CREATE_FORCE_4K_PAGES) {
2393 		p->pmap_pt_attr = &pmap_pt_attr_4k;
2394 	}
2395 #endif /* __ARM_MIXED_PAGE_SIZE__ */
2396 	p->max = pmap_user_va_size(p);
2397 
2398 	if (!pmap_get_pt_ops(p)->alloc_id(p)) {
2399 		local_kr = KERN_NO_SPACE;
2400 		goto id_alloc_fail;
2401 	}
2402 
2403 	/**
2404 	 * We expect top level translation tables to always fit into a single
2405 	 * physical page. This would also catch a misconfiguration if 4K
2406 	 * concatenated page tables needed more than one physical tt1 page.
2407 	 */
2408 	vm_size_t pmap_root_size = pmap_root_alloc_size(p);
2409 	if (__improbable(pmap_root_size > PAGE_SIZE)) {
2410 		panic("%s: translation tables do not fit into a single physical page %u", __FUNCTION__, (unsigned)pmap_root_size);
2411 	}
2412 
2413 	pmap_lock_init(p);
2414 
2415 	p->tte = pmap_tt1_allocate(p, sptm_root_flags);
2416 	if (!(p->tte)) {
2417 		local_kr = KERN_RESOURCE_SHORTAGE;
2418 		goto tt1_alloc_fail;
2419 	}
2420 
2421 	p->ttep = kvtophys_nofail((vm_offset_t)p->tte);
2422 	PMAP_TRACE(4, PMAP_CODE(PMAP__TTE), VM_KERNEL_ADDRHIDE(p), VM_KERNEL_ADDRHIDE(p->min), VM_KERNEL_ADDRHIDE(p->max), p->ttep);
2423 
2424 	/*
2425 	 *  initialize the rest of the structure
2426 	 */
2427 	p->nested_region_addr = 0x0ULL;
2428 	p->nested_region_size = 0x0ULL;
2429 	p->nested_region_unnested_table_bitmap = NULL;
2430 
2431 	p->nested_has_no_bounds_ref = false;
2432 	p->nested_no_bounds_refcnt = 0;
2433 	p->nested_bounds_set = false;
2434 
2435 
2436 #if MACH_ASSERT
2437 	p->pmap_pid = 0;
2438 	strlcpy(p->pmap_procname, "<nil>", sizeof(p->pmap_procname));
2439 #endif /* MACH_ASSERT */
2440 #if DEVELOPMENT || DEBUG
2441 	p->footprint_was_suspended = FALSE;
2442 #endif /* DEVELOPMENT || DEBUG */
2443 
2444 	os_ref_init_count_raw(&p->ref_count, &pmap_refgrp, 1);
2445 	pmap_simple_lock(&pmaps_lock);
2446 	queue_enter(&map_pmap_list, p, pmap_t, pmaps);
2447 	pmap_simple_unlock(&pmaps_lock);
2448 
2449 	/**
2450 	 * The SPTM pmap's concurrency model can sometimes allow ledger balances to transiently
2451 	 * go negative.  Note that we still check overall ledger balance on pmap destruction.
2452 	 */
2453 	ledger_disable_panic_on_negative(p->ledger, task_ledgers.phys_footprint);
2454 	ledger_disable_panic_on_negative(p->ledger, task_ledgers.internal);
2455 	ledger_disable_panic_on_negative(p->ledger, task_ledgers.internal_compressed);
2456 	ledger_disable_panic_on_negative(p->ledger, task_ledgers.iokit_mapped);
2457 	ledger_disable_panic_on_negative(p->ledger, task_ledgers.alternate_accounting);
2458 	ledger_disable_panic_on_negative(p->ledger, task_ledgers.alternate_accounting_compressed);
2459 	ledger_disable_panic_on_negative(p->ledger, task_ledgers.external);
2460 	ledger_disable_panic_on_negative(p->ledger, task_ledgers.reusable);
2461 	ledger_disable_panic_on_negative(p->ledger, task_ledgers.wired_mem);
2462 
2463 	if (!is_stage2) {
2464 		/*
2465 		 * Complete initialization for the TXM address space. This needs to be done
2466 		 * after the SW ASID has been registered with the SPTM.
2467 		 * TXM enforcement does not apply to virtual machines.
2468 		 */
2469 		if (flags & PMAP_CREATE_TEST) {
2470 			txm_flags |= kTXMAddressSpaceFlagTest;
2471 		}
2472 
2473 		pmap_txmlock_init(p);
2474 		txm_register_address_space(p, p->asid, txm_flags);
2475 		p->txm_trust_level = kCSTrustUntrusted;
2476 	}
2477 
2478 	return p;
2479 
2480 tt1_alloc_fail:
2481 	pmap_get_pt_ops(p)->free_id(p);
2482 id_alloc_fail:
2483 	zfree(pmap_zone, p);
2484 pmap_create_fail:
2485 	*kr = local_kr;
2486 	return PMAP_NULL;
2487 }
2488 
2489 pmap_t
pmap_create_options(ledger_t ledger,vm_map_size_t size,unsigned int flags)2490 pmap_create_options(
2491 	ledger_t ledger,
2492 	vm_map_size_t size,
2493 	unsigned int flags)
2494 {
2495 	pmap_t pmap;
2496 	kern_return_t kr = KERN_SUCCESS;
2497 
2498 	PMAP_TRACE(1, PMAP_CODE(PMAP__CREATE) | DBG_FUNC_START, size, flags);
2499 
2500 	ledger_reference(ledger);
2501 
2502 	pmap = pmap_create_options_internal(ledger, size, flags, &kr);
2503 
2504 	if (pmap == PMAP_NULL) {
2505 		ledger_dereference(ledger);
2506 	}
2507 
2508 	PMAP_TRACE(1, PMAP_CODE(PMAP__CREATE) | DBG_FUNC_END, VM_KERNEL_ADDRHIDE(pmap), PMAP_VASID(pmap), PMAP_HWASID(pmap));
2509 
2510 	return pmap;
2511 }
2512 
2513 #if MACH_ASSERT
2514 MARK_AS_PMAP_TEXT void
pmap_set_process_internal(__unused pmap_t pmap,__unused int pid,__unused char * procname)2515 pmap_set_process_internal(
2516 	__unused pmap_t pmap,
2517 	__unused int pid,
2518 	__unused char *procname)
2519 {
2520 	if (pmap == NULL || pmap->pmap_pid == -1) {
2521 		return;
2522 	}
2523 
2524 	validate_pmap_mutable(pmap);
2525 
2526 	pmap->pmap_pid = pid;
2527 	strlcpy(pmap->pmap_procname, procname, sizeof(pmap->pmap_procname));
2528 }
2529 #endif /* MACH_ASSERT */
2530 
2531 #if MACH_ASSERT
2532 void
pmap_set_process(pmap_t pmap,int pid,char * procname)2533 pmap_set_process(
2534 	pmap_t pmap,
2535 	int pid,
2536 	char *procname)
2537 {
2538 	pmap_set_process_internal(pmap, pid, procname);
2539 }
2540 #endif /* MACH_ASSERT */
2541 
2542 /*
2543  * pmap_deallocate_all_leaf_tts:
2544  *
2545  * Recursive function for deallocating all leaf TTEs.  Walks the given TT,
2546  * removing and deallocating all TTEs.
2547  */
2548 MARK_AS_PMAP_TEXT static void
pmap_deallocate_all_leaf_tts(pmap_t pmap,tt_entry_t * first_ttep,vm_map_address_t start_va,unsigned level)2549 pmap_deallocate_all_leaf_tts(pmap_t pmap, tt_entry_t * first_ttep, vm_map_address_t start_va, unsigned level)
2550 {
2551 	tt_entry_t tte = ARM_TTE_EMPTY;
2552 	tt_entry_t * ttep = NULL;
2553 	tt_entry_t * last_ttep = NULL;
2554 
2555 	const pt_attr_t * const pt_attr = pmap_get_pt_attr(pmap);
2556 	const uint64_t size = pt_attr->pta_level_info[level].size;
2557 
2558 	assert(level < pt_attr_leaf_level(pt_attr));
2559 
2560 	last_ttep = &first_ttep[ttn_index(pt_attr, ~0, level)];
2561 
2562 	const uint64_t page_ratio = PAGE_SIZE / pt_attr_page_size(pt_attr);
2563 	vm_map_address_t va = start_va;
2564 	for (ttep = first_ttep; ttep <= last_ttep; ttep += page_ratio, va += (size * page_ratio)) {
2565 		if (!(*ttep & ARM_TTE_VALID)) {
2566 			continue;
2567 		}
2568 
2569 		for (unsigned i = 0; i < page_ratio; i++) {
2570 			tte = ttep[i];
2571 
2572 			if (!(tte & ARM_TTE_VALID)) {
2573 				panic("%s: found unexpectedly invalid tte, ttep=%p, tte=%p, "
2574 				    "pmap=%p, first_ttep=%p, level=%u",
2575 				    __FUNCTION__, ttep + i, (void *)tte,
2576 				    pmap, first_ttep, level);
2577 			}
2578 
2579 			if ((tte & ARM_TTE_TYPE_MASK) == ARM_TTE_TYPE_BLOCK) {
2580 				panic("%s: found block mapping, ttep=%p, tte=%p, "
2581 				    "pmap=%p, first_ttep=%p, level=%u",
2582 				    __FUNCTION__, ttep + i, (void *)tte,
2583 				    pmap, first_ttep, level);
2584 			}
2585 
2586 			/* Must be valid, type table */
2587 			if (level < pt_attr_twig_level(pt_attr)) {
2588 				/* If we haven't reached the twig level, recurse to the next level. */
2589 				pmap_deallocate_all_leaf_tts(pmap, (tt_entry_t *)phystokv((tte) & ARM_TTE_TABLE_MASK),
2590 				    va + (size * i), level + 1);
2591 			}
2592 		}
2593 
2594 		/* Remove the TTE. */
2595 		pmap_lock(pmap, PMAP_LOCK_EXCLUSIVE);
2596 		pmap_tte_deallocate(pmap, va, ttep, level);
2597 	}
2598 }
2599 
2600 /*
2601  * We maintain stats and ledgers so that a task's physical footprint is:
2602  * phys_footprint = ((internal - alternate_accounting)
2603  *                   + (internal_compressed - alternate_accounting_compressed)
2604  *                   + iokit_mapped
2605  *                   + purgeable_nonvolatile
2606  *                   + purgeable_nonvolatile_compressed
2607  *                   + page_table)
2608  * where "alternate_accounting" includes "iokit" and "purgeable" memory.
2609  */
2610 
2611 /*
2612  *	Retire the given physical map from service.
2613  *	Should only be called if the map contains
2614  *	no valid mappings.
2615  */
2616 MARK_AS_PMAP_TEXT void
pmap_destroy_internal(pmap_t pmap)2617 pmap_destroy_internal(
2618 	pmap_t pmap)
2619 {
2620 	if (pmap == PMAP_NULL) {
2621 		return;
2622 	}
2623 
2624 	validate_pmap(pmap);
2625 
2626 	__unused const pt_attr_t * const pt_attr = pmap_get_pt_attr(pmap);
2627 	const bool is_stage2_pmap = false;
2628 
2629 	if (os_ref_release_raw(&pmap->ref_count, &pmap_refgrp) > 0) {
2630 		return;
2631 	}
2632 
2633 	if (!is_stage2_pmap) {
2634 		/*
2635 		 * Complete all clean up required for TXM. This needs to happen before the
2636 		 * SW ASID has been unregistered with the SPTM.
2637 		 */
2638 		txm_unregister_address_space(pmap);
2639 		pmap_txmlock_destroy(pmap);
2640 	}
2641 
2642 	/**
2643 	 * Drain any concurrent retype-sensitive SPTM operations.  This is needed to
2644 	 * ensure that we don't unmap and retype the page tables while those operations
2645 	 * are still finishing on other CPUs, leading to an SPTM violation.  In particular,
2646 	 * the multipage batched cacheability/attribute update code may issue SPTM calls
2647 	 * without holding the relevant PVH or pmap locks, so we can't guarantee those
2648 	 * calls have actually completed despite observing refcnt == 0.
2649 	 *
2650 	 * At this point, we CAN guarantee that:
2651 	 * 1) All prior PTE removals required to empty the pmap have completed and
2652 	 *    been synchronized with DSB, *except* the commpage removal which doesn't
2653 	 *    involve pages that can ever be retyped.  Subsequent calls not already
2654 	 *    in the retype epoch will no longer observe these mappings.
2655 	 * 2) The pmap now has a zero refcount, so in a correctly functioning system
2656 	 *    no further mappings will be requested for it.
2657 	 */
2658 	pmap_retype_epoch_prepare_drain();
2659 
2660 	if (!is_stage2_pmap) {
2661 		pmap_unmap_commpage(pmap);
2662 	}
2663 
2664 	pmap_simple_lock(&pmaps_lock);
2665 	queue_remove(&map_pmap_list, pmap, pmap_t, pmaps);
2666 	pmap_simple_unlock(&pmaps_lock);
2667 
2668 	pmap_retype_epoch_drain();
2669 
2670 	pmap_trim_self(pmap);
2671 
2672 	/*
2673 	 *	Free the memory maps, then the
2674 	 *	pmap structure.
2675 	 */
2676 	pmap_deallocate_all_leaf_tts(pmap, pmap->tte, pmap->min, pt_attr_root_level(pt_attr));
2677 
2678 	if (pmap->tte) {
2679 		pmap_tt1_deallocate(pmap, pmap->tte);
2680 		pmap->tte = (tt_entry_t *) NULL;
2681 		pmap->ttep = 0;
2682 	}
2683 
2684 	if (pmap->type != PMAP_TYPE_NESTED) {
2685 		/* return its asid to the pool */
2686 		pmap_get_pt_ops(pmap)->free_id(pmap);
2687 		if (pmap->nested_pmap != NULL) {
2688 			/* release the reference we hold on the nested pmap */
2689 			pmap_destroy_internal(pmap->nested_pmap);
2690 		}
2691 	}
2692 
2693 	pmap_check_ledgers(pmap);
2694 
2695 	if (pmap->nested_region_unnested_table_bitmap) {
2696 		bitmap_free(pmap->nested_region_unnested_table_bitmap, pmap->nested_region_size >> pt_attr_twig_shift(pt_attr));
2697 	}
2698 
2699 	pmap_lock_destroy(pmap);
2700 	zfree(pmap_zone, pmap);
2701 }
2702 
2703 void
pmap_destroy(pmap_t pmap)2704 pmap_destroy(
2705 	pmap_t pmap)
2706 {
2707 	PMAP_TRACE(1, PMAP_CODE(PMAP__DESTROY) | DBG_FUNC_START, VM_KERNEL_ADDRHIDE(pmap), PMAP_VASID(pmap), PMAP_HWASID(pmap));
2708 
2709 	ledger_t ledger = pmap->ledger;
2710 
2711 	pmap_destroy_internal(pmap);
2712 
2713 	ledger_dereference(ledger);
2714 
2715 	PMAP_TRACE(1, PMAP_CODE(PMAP__DESTROY) | DBG_FUNC_END);
2716 }
2717 
2718 
2719 /*
2720  *	Add a reference to the specified pmap.
2721  */
2722 MARK_AS_PMAP_TEXT void
pmap_reference_internal(pmap_t pmap)2723 pmap_reference_internal(
2724 	pmap_t pmap)
2725 {
2726 	if (pmap != PMAP_NULL) {
2727 		validate_pmap_mutable(pmap);
2728 		os_ref_retain_raw(&pmap->ref_count, &pmap_refgrp);
2729 	}
2730 }
2731 
2732 void
pmap_reference(pmap_t pmap)2733 pmap_reference(
2734 	pmap_t pmap)
2735 {
2736 	pmap_reference_internal(pmap);
2737 }
2738 
2739 static sptm_frame_type_t
get_sptm_pt_type(pmap_t pmap)2740 get_sptm_pt_type(pmap_t pmap)
2741 {
2742 	const bool is_stage2_pmap = false;
2743 	if (is_stage2_pmap) {
2744 		assert(pmap->type != PMAP_TYPE_NESTED);
2745 		return XNU_STAGE2_PAGE_TABLE;
2746 	} else {
2747 		return pmap->type == PMAP_TYPE_NESTED ? XNU_PAGE_TABLE_SHARED : XNU_PAGE_TABLE;
2748 	}
2749 }
2750 
2751 static tt_entry_t *
pmap_tt1_allocate(pmap_t pmap,uint8_t sptm_root_flags)2752 pmap_tt1_allocate(pmap_t pmap, uint8_t sptm_root_flags)
2753 {
2754 	pmap_paddr_t pa = 0;
2755 	const pt_attr_t * const pt_attr = pmap_get_pt_attr(pmap);
2756 	const bool is_stage2_pmap = false;
2757 
2758 	const kern_return_t ret = pmap_page_alloc(&pa, PMAP_PAGE_NOZEROFILL);
2759 
2760 	if (ret != KERN_SUCCESS) {
2761 		return (tt_entry_t *)0;
2762 	}
2763 
2764 	/**
2765 	 * Drain the epochs to ensure any lingering batched operations that may have taken
2766 	 * an in-flight reference to this page are complete.
2767 	 */
2768 	pmap_retype_epoch_prepare_drain();
2769 
2770 	assert(pa);
2771 
2772 	/* Always report root allocations in units of PMAP_ROOT_ALLOC_SIZE, which can be obtained by sysctl arm_pt_root_size.
2773 	 * Depending on the device, this can vary between 512b and 16K. */
2774 	OSAddAtomic(1, (pmap == kernel_pmap ? &inuse_kernel_tteroot_count : &inuse_user_tteroot_count));
2775 	pmap_tt_ledger_credit(pmap, PAGE_SIZE);
2776 
2777 	sptm_retype_params_t retype_params = {.raw = SPTM_RETYPE_PARAMS_NULL};
2778 	retype_params.attr_idx = pt_attr->geometry_id;
2779 	retype_params.flags = sptm_root_flags;
2780 	if (is_stage2_pmap) {
2781 		retype_params.vmid = pmap->vmid;
2782 	} else {
2783 		retype_params.asid = pmap->asid;
2784 	}
2785 
2786 	pmap_retype_epoch_drain();
2787 
2788 	sptm_retype(pa, XNU_DEFAULT, is_stage2_pmap ? XNU_STAGE2_ROOT_TABLE : XNU_USER_ROOT_TABLE,
2789 	    retype_params);
2790 
2791 	return (tt_entry_t *) phystokv(pa);
2792 }
2793 
2794 static void
pmap_tt1_deallocate(pmap_t pmap,tt_entry_t * tt)2795 pmap_tt1_deallocate(
2796 	pmap_t pmap,
2797 	tt_entry_t *tt)
2798 {
2799 	pmap_paddr_t pa = kvtophys_nofail((vm_offset_t)tt);
2800 	const bool is_stage2_pmap = false;
2801 	const sptm_frame_type_t page_type = is_stage2_pmap ? XNU_STAGE2_ROOT_TABLE :
2802 	    pmap->type == PMAP_TYPE_NESTED ? XNU_SHARED_ROOT_TABLE : XNU_USER_ROOT_TABLE;
2803 
2804 	sptm_retype_params_t retype_params = {.raw = SPTM_RETYPE_PARAMS_NULL};
2805 	sptm_retype(pa, page_type, XNU_DEFAULT, retype_params);
2806 	pmap_page_free(pa);
2807 
2808 	OSAddAtomic(-1, (pmap == kernel_pmap ? &inuse_kernel_tteroot_count : &inuse_user_tteroot_count));
2809 	pmap_tt_ledger_debit(pmap, PAGE_SIZE);
2810 }
2811 
2812 MARK_AS_PMAP_TEXT static kern_return_t
pmap_tt_allocate(pmap_t pmap,tt_entry_t ** ttp,unsigned int level,unsigned int options)2813 pmap_tt_allocate(
2814 	pmap_t pmap,
2815 	tt_entry_t **ttp,
2816 	unsigned int level,
2817 	unsigned int options)
2818 {
2819 	pmap_paddr_t pa;
2820 	*ttp = NULL;
2821 
2822 	if (*ttp == NULL) {
2823 		const unsigned int alloc_flags =
2824 		    (options & PMAP_TT_ALLOCATE_NOWAIT) ? PMAP_PAGE_ALLOCATE_NOWAIT : 0;
2825 
2826 		/* Allocate a VM page to be used as the page table. */
2827 		if (pmap_page_alloc(&pa, alloc_flags) != KERN_SUCCESS) {
2828 			return KERN_RESOURCE_SHORTAGE;
2829 		}
2830 
2831 		pt_desc_t *ptdp = ptd_alloc(pmap, alloc_flags);
2832 		if (ptdp == NULL) {
2833 			pmap_page_free(pa);
2834 			return KERN_RESOURCE_SHORTAGE;
2835 		}
2836 
2837 		unsigned int pai = pa_index(pa);
2838 		locked_pvh_t locked_pvh = pvh_lock(pai);
2839 		assertf(pvh_test_type(locked_pvh.pvh, PVH_TYPE_NULL), "%s: non-empty PVH %p",
2840 		    __func__, (void*)locked_pvh.pvh);
2841 
2842 		/**
2843 		 * Drain the epochs to ensure any lingering batched operations that may have taken
2844 		 * an in-flight reference to this page are complete.
2845 		 */
2846 		pmap_retype_epoch_prepare_drain();
2847 
2848 		if (level < pt_attr_leaf_level(pmap_get_pt_attr(pmap))) {
2849 			OSAddAtomic(1, (pmap == kernel_pmap ? &inuse_kernel_ttepages_count : &inuse_user_ttepages_count));
2850 		} else {
2851 			OSAddAtomic(1, (pmap == kernel_pmap ? &inuse_kernel_ptepages_count : &inuse_user_ptepages_count));
2852 		}
2853 
2854 		pmap_tt_ledger_credit(pmap, PAGE_SIZE);
2855 
2856 		PMAP_ZINFO_PALLOC(pmap, PAGE_SIZE);
2857 
2858 		pvh_update_head(&locked_pvh, ptdp, PVH_TYPE_PTDP);
2859 		pvh_unlock(&locked_pvh);
2860 
2861 		sptm_retype_params_t retype_params = {.raw = SPTM_RETYPE_PARAMS_NULL};
2862 		retype_params.level = (sptm_pt_level_t)level;
2863 
2864 		/**
2865 		 * SPTM TODO: To reduce the cost of draining and retyping, consider caching freed page table pages
2866 		 * in a small per-CPU bucket and reusing them in preference to calling pmap_page_alloc() above.
2867 		 */
2868 		pmap_retype_epoch_drain();
2869 
2870 		sptm_retype(pa, XNU_DEFAULT, get_sptm_pt_type(pmap), retype_params);
2871 
2872 		*ttp = (tt_entry_t *)phystokv(pa);
2873 	}
2874 
2875 	assert(*ttp);
2876 
2877 	return KERN_SUCCESS;
2878 }
2879 
2880 static void
pmap_tt_deallocate(pmap_t pmap,tt_entry_t * ttp,unsigned int level)2881 pmap_tt_deallocate(
2882 	pmap_t pmap,
2883 	tt_entry_t *ttp,
2884 	unsigned int level)
2885 {
2886 	pt_desc_t *ptdp;
2887 	vm_offset_t     free_page = 0;
2888 	const pt_attr_t * const pt_attr = pmap_get_pt_attr(pmap);
2889 
2890 	ptdp = ptep_get_ptd(ttp);
2891 	ptdp->va = (vm_offset_t)-1;
2892 
2893 	const uint16_t refcnt = sptm_get_page_table_refcnt(kvtophys_nofail((vm_offset_t)ttp));
2894 
2895 	if (__improbable(refcnt != 0)) {
2896 		panic("pmap_tt_deallocate(): ptdp %p, count %d", ptdp, refcnt);
2897 	}
2898 
2899 	free_page = (vm_offset_t)ttp & ~PAGE_MASK;
2900 	if (free_page != 0) {
2901 		pmap_paddr_t pa = kvtophys_nofail(free_page);
2902 		sptm_retype_params_t retype_params = {.raw = SPTM_RETYPE_PARAMS_NULL};
2903 		sptm_retype(pa, get_sptm_pt_type(pmap), XNU_DEFAULT, retype_params);
2904 		ptd_deallocate(ptep_get_ptd((pt_entry_t*)free_page));
2905 
2906 		unsigned int pai = pa_index(pa);
2907 		locked_pvh_t locked_pvh = pvh_lock(pai);
2908 		assertf(pvh_test_type(locked_pvh.pvh, PVH_TYPE_PTDP), "%s: non-PTD PVH %p",
2909 		    __func__, (void*)locked_pvh.pvh);
2910 		pvh_update_head(&locked_pvh, NULL, PVH_TYPE_NULL);
2911 		pvh_unlock(&locked_pvh);
2912 		pmap_page_free(pa);
2913 		if (level < pt_attr_leaf_level(pt_attr)) {
2914 			OSAddAtomic(-1, (pmap == kernel_pmap ? &inuse_kernel_ttepages_count : &inuse_user_ttepages_count));
2915 		} else {
2916 			OSAddAtomic(-1, (pmap == kernel_pmap ? &inuse_kernel_ptepages_count : &inuse_user_ptepages_count));
2917 		}
2918 		PMAP_ZINFO_PFREE(pmap, PAGE_SIZE);
2919 		pmap_tt_ledger_debit(pmap, PAGE_SIZE);
2920 	}
2921 }
2922 
2923 /**
2924  * Check table refcounts after clearing a translation table entry pointing to that table
2925  *
2926  * @note If the cleared TTE points to a leaf table, then that leaf table
2927  *       must have a refcnt of zero before the TTE can be removed.
2928  *
2929  * @param pmap The pmap containing the page table whose TTE is being removed.
2930  * @param tte Value stored in the TTE prior to clearing it
2931  * @param level The level of the page table that contains the TTE being removed
2932  */
2933 static void
pmap_tte_check_refcounts(pmap_t pmap,tt_entry_t tte,unsigned int level)2934 pmap_tte_check_refcounts(
2935 	pmap_t pmap,
2936 	tt_entry_t tte,
2937 	unsigned int level)
2938 {
2939 	const pt_attr_t * const pt_attr = pmap_get_pt_attr(pmap);
2940 
2941 	/**
2942 	 * Remember, the passed in "level" parameter refers to the level above the
2943 	 * table that's getting removed (e.g., removing an L2 TTE will unmap an L3
2944 	 * page table).
2945 	 */
2946 	const bool remove_leaf_table = (level == pt_attr_twig_level(pt_attr));
2947 
2948 	unsigned short refcnt = 0;
2949 
2950 	/**
2951 	 * It's possible that a concurrent pmap_disconnect() operation may need to reference
2952 	 * a PTE on the pagetable page to be removed.  A full disconnect() may have cleared
2953 	 * one or more PTEs on this page but not yet dropped the refcount, which would cause
2954 	 * us to panic in this function on a non-zero refcount.  Moreover, it's possible for
2955 	 * a disconnect-to-compress operation to set the compressed marker on a PTE, and
2956 	 * for pmap_remove_range_options() to concurrently observe that marker, clear it, and
2957 	 * drop the pagetable refcount accordingly, without taking any PVH locks that could
2958 	 * synchronize it against the disconnect operation.  If that removal caused the
2959 	 * refcount to reach zero, the pagetable page could be freed before the disconnect
2960 	 * operation is finished using the relevant pagetable descriptor.
2961 	 * Address these cases by waiting until all CPUs have been observed to not be
2962 	 * executing pmap_disconnect().
2963 	 */
2964 	if (remove_leaf_table) {
2965 		bitmap_t active_disconnects[BITMAP_LEN(MAX_CPUS)];
2966 		const int max_cpu = ml_get_max_cpu_number();
2967 		bitmap_full(&active_disconnects[0], max_cpu + 1);
2968 		bool inflight_disconnect;
2969 
2970 		/*
2971 		 * Ensure the ensuing load of per-CPU inflight_disconnect is not speculated
2972 		 * ahead of any prior PTE load which may have observed the effect of a
2973 		 * concurrent disconnect operation.  An acquire fence is required for this;
2974 		 * a load-acquire operation is insufficient.
2975 		 */
2976 		os_atomic_thread_fence(acquire);
2977 		do {
2978 			inflight_disconnect = false;
2979 			for (int i = bitmap_first(&active_disconnects[0], max_cpu + 1);
2980 			    i >= 0;
2981 			    i = bitmap_next(&active_disconnects[0], i)) {
2982 				const pmap_cpu_data_t *cpu_data = pmap_get_remote_cpu_data(i);
2983 				if (cpu_data == NULL) {
2984 					continue;
2985 				}
2986 				if (os_atomic_load_exclusive(&cpu_data->inflight_disconnect, relaxed)) {
2987 					__builtin_arm_wfe();
2988 					inflight_disconnect = true;
2989 					continue;
2990 				}
2991 				os_atomic_clear_exclusive();
2992 				bitmap_clear(&active_disconnects[0], (unsigned int)i);
2993 			}
2994 		} while (inflight_disconnect);
2995 		/* Ensure the refcount is observed after any observation of inflight_disconnect */
2996 		os_atomic_thread_fence(acquire);
2997 		refcnt = sptm_get_page_table_refcnt(tte_to_pa(tte));
2998 	}
2999 
3000 #if MACH_ASSERT
3001 	/**
3002 	 * On internal devices, always do the page table consistency check
3003 	 * regardless of page table level or the actual refcnt value.
3004 	 */
3005 	{
3006 #else /* MACH_ASSERT */
3007 	/**
3008 	 * Only perform the page table consistency check when deleting leaf page
3009 	 * tables and it seems like there might be valid/compressed mappings
3010 	 * leftover.
3011 	 */
3012 	if (__improbable(remove_leaf_table && refcnt != 0)) {
3013 #endif /* MACH_ASSERT */
3014 
3015 		/**
3016 		 * There are multiple problems that can arise as a non-zero refcnt:
3017 		 * 1. A bug in the refcnt management logic.
3018 		 * 2. A memory stomper or hardware failure.
3019 		 * 3. The VM forgetting to unmap all of the valid mappings in an address
3020 		 *    space before destroying a pmap.
3021 		 *
3022 		 * By looping over the page table and determining how many valid or
3023 		 * compressed entries there actually are, we can narrow down which of
3024 		 * these three cases is causing this panic. If the expected refcnt
3025 		 * (valid + compressed) and the actual refcnt don't match then the
3026 		 * problem is probably either a memory corruption issue (if the
3027 		 * non-empty entries don't match valid+compressed, that could also be a
3028 		 * sign of corruption) or refcnt management bug. Otherwise, there
3029 		 * actually are leftover mappings and the higher layers of xnu are
3030 		 * probably at fault.
3031 		 *
3032 		 * Note that we use PAGE_SIZE to govern the range of the table check,
3033 		 * because even for 4K processes we still allocate a 16K page for each
3034 		 * page table; we simply map it using 4 adjacent TTEs for the 4K case.
3035 		 */
3036 		pt_entry_t *bpte = ((pt_entry_t *) (ttetokv(tte) & ~(PAGE_SIZE - 1)));
3037 
3038 		pt_entry_t *ptep = bpte;
3039 		unsigned short wiredcnt = ptep_get_info((pt_entry_t*)ttetokv(tte))->wiredcnt;
3040 		unsigned short non_empty = 0, valid = 0, comp = 0;
3041 		for (unsigned int i = 0; i < (PAGE_SIZE / sizeof(*ptep)); i++, ptep++) {
3042 			/* Keep track of all non-empty entries to detect memory corruption. */
3043 			if (__improbable(*ptep != ARM_PTE_EMPTY)) {
3044 				non_empty++;
3045 			}
3046 
3047 			if (__improbable(pte_is_compressed(*ptep, ptep))) {
3048 				comp++;
3049 			} else if (__improbable((*ptep & ARM_PTE_TYPE_VALID) == ARM_PTE_TYPE)) {
3050 				valid++;
3051 			}
3052 		}
3053 
3054 #if MACH_ASSERT
3055 		/**
3056 		 * On internal machines, panic whenever a page table getting deleted has
3057 		 * leftover mappings (valid or otherwise) or a leaf page table has a
3058 		 * non-zero refcnt.
3059 		 */
3060 		if (__improbable((non_empty != 0) || (remove_leaf_table && ((refcnt != 0) || (wiredcnt != 0))))) {
3061 #else /* MACH_ASSERT */
3062 		/* We already know the leaf page-table has a non-zero refcnt, so panic. */
3063 		{
3064 #endif /* MACH_ASSERT */
3065 			panic("%s: Found inconsistent state in soon to be deleted L%d table: %d valid, "
3066 			    "%d compressed, %d non-empty, refcnt=%d, wiredcnt=%d, L%d tte=%#llx, pmap=%p, bpte=%p", __func__,
3067 			    level + 1, valid, comp, non_empty, refcnt, wiredcnt, level, (uint64_t)tte, pmap, bpte);
3068 		}
3069 	}
3070 }
3071 
3072 /**
3073  * Remove translation table entry pointing to a nested shared region table
3074  *
3075  * @note The TTE to clear out is expected to point to a leaf table with a refcnt
3076  *       of zero.
3077  *
3078  * @param pmap The user pmap containing the nested page table whose TTE is being removed.
3079  * @param va_start Beginning of the VA range mapped by the table being removed, for TLB maintenance.
3080  * @param ttep Pointer to the TTE that should be cleared out.
3081  */
3082 static void
3083 pmap_tte_trim(
3084 	pmap_t pmap,
3085 	vm_offset_t va_start,
3086 	tt_entry_t *ttep)
3087 {
3088 	pmap_assert_locked(pmap, PMAP_LOCK_EXCLUSIVE);
3089 	assert(ttep != NULL);
3090 	const tt_entry_t tte = *ttep;
3091 	const pt_attr_t * const pt_attr = pmap_get_pt_attr(pmap);
3092 
3093 	if (__improbable(tte == ARM_TTE_EMPTY)) {
3094 		panic("%s: L%d TTE is already empty. Potential double unmap or memory "
3095 		    "stomper? pmap=%p ttep=%p", __func__, pt_attr_twig_level(pt_attr), pmap, ttep);
3096 	}
3097 
3098 	const uint64_t page_ratio = PAGE_SIZE / pt_attr_page_size(pt_attr);
3099 	sptm_unnest_region(pmap->ttep, pmap->nested_pmap->ttep, va_start, (pt_attr_twig_size(pt_attr) * page_ratio) >> pt_attr->pta_page_shift);
3100 
3101 	pmap_unlock(pmap, PMAP_LOCK_EXCLUSIVE);
3102 
3103 	pmap_tte_check_refcounts(pmap, tte, pt_attr_twig_level(pt_attr));
3104 }
3105 
3106 /**
3107  * Remove a translation table entry.
3108  *
3109  * @note If the TTE to clear out points to a leaf table, then that leaf table
3110  *       must have a mapping refcount of zero before the TTE can be removed.
3111  * @note This function expects to be called with pmap locked exclusive, and will
3112  *       return with pmap unlocked.
3113  *
3114  * @param pmap The pmap containing the page table whose TTE is being removed.
3115  * @param va_start Beginning of the VA range mapped by the table being removed, for TLB maintenance.
3116  * @param ttep Pointer to the TTE that should be cleared out.
3117  * @param level The level of the page table that contains the TTE to be removed.
3118  */
3119 static void
3120 pmap_tte_remove(
3121 	pmap_t pmap,
3122 	vm_offset_t va_start,
3123 	tt_entry_t *ttep,
3124 	unsigned int level)
3125 {
3126 	pmap_assert_locked(pmap, PMAP_LOCK_EXCLUSIVE);
3127 	assert(ttep != NULL);
3128 	const tt_entry_t tte = *ttep;
3129 
3130 	if (__improbable(tte == ARM_TTE_EMPTY)) {
3131 		panic("%s: L%d TTE is already empty. Potential double unmap or memory "
3132 		    "stomper? pmap=%p ttep=%p", __func__, level, pmap, ttep);
3133 	}
3134 
3135 	sptm_unmap_table(pmap->ttep, pt_attr_align_va(pmap_get_pt_attr(pmap), level, va_start), (sptm_pt_level_t)level);
3136 
3137 	pmap_unlock(pmap, PMAP_LOCK_EXCLUSIVE);
3138 
3139 	pmap_tte_check_refcounts(pmap, tte, level);
3140 }
3141 
3142 /**
3143  * Given a pointer to an entry within a `level` page table, delete the
3144  * page table at `level` + 1 that is represented by that entry. For instance,
3145  * to delete an unused L3 table, `ttep` would be a pointer to the L2 entry that
3146  * contains the PA of the L3 table, and `level` would be "2".
3147  *
3148  * @note If the table getting deallocated is a leaf table, then that leaf table
3149  *       must have a mapping refcount of zero before getting deallocated.
3150  * @note This function expects to be called with pmap locked exclusive and will
3151  *       return with pmap unlocked.
3152  *
3153  * @param pmap The pmap that owns the page table to be deallocated.
3154  * @param va_start Beginning of the VA range mapped by the table being removed, for TLB maintenance.
3155  * @param ttep Pointer to the `level` TTE to remove.
3156  * @param level The level of the table that contains an entry pointing to the
3157  *              table to be removed. The deallocated page table will be a
3158  *              `level` + 1 table (so if `level` is 2, then an L3 table will be
3159  *              deleted).
3160  */
3161 void
3162 pmap_tte_deallocate(
3163 	pmap_t pmap,
3164 	vm_offset_t va_start,
3165 	tt_entry_t *ttep,
3166 	unsigned int level)
3167 {
3168 	tt_entry_t tte;
3169 
3170 	pmap_assert_locked(pmap, PMAP_LOCK_EXCLUSIVE);
3171 
3172 	tte = *ttep;
3173 
3174 	if (tte_get_ptd(tte)->pmap != pmap) {
3175 		panic("%s: Passed in pmap doesn't own the page table to be deleted ptd=%p ptd->pmap=%p pmap=%p",
3176 		    __func__, tte_get_ptd(tte), tte_get_ptd(tte)->pmap, pmap);
3177 	}
3178 
3179 	assertf((tte & ARM_TTE_TYPE_MASK) == ARM_TTE_TYPE_TABLE, "%s: invalid TTE %p (0x%llx)",
3180 	    __func__, ttep, (unsigned long long)tte);
3181 
3182 	/* pmap_tte_remove() will drop the pmap lock */
3183 	pmap_tte_remove(pmap, va_start, ttep, level);
3184 
3185 	pmap_tt_deallocate(pmap, (tt_entry_t *) phystokv(tte_to_pa(tte)), level + 1);
3186 }
3187 
3188 /*
3189  *	Remove a range of hardware page-table entries.
3190  *	The range is given as the first (inclusive)
3191  *	and last (exclusive) virtual addresses mapped by
3192  *      the PTE region to be removed.
3193  *
3194  *	The pmap must be locked shared.
3195  *	If the pmap is not the kernel pmap, the range must lie
3196  *	entirely within one pte-page. Assumes that the pte-page exists.
3197  *
3198  *	Returns the number of PTE changed
3199  */
3200 MARK_AS_PMAP_TEXT static void
3201 pmap_remove_range(
3202 	pmap_t pmap,
3203 	vm_map_address_t va,
3204 	vm_map_address_t end)
3205 {
3206 	pmap_remove_range_options(pmap, va, end, PMAP_OPTIONS_REMOVE);
3207 }
3208 
3209 MARK_AS_PMAP_TEXT void
3210 pmap_remove_range_options(
3211 	pmap_t pmap,
3212 	vm_map_address_t start,
3213 	vm_map_address_t end,
3214 	int options)
3215 {
3216 	const unsigned int sptm_flags = ((options & PMAP_OPTIONS_REMOVE) ? SPTM_REMOVE_COMPRESSED : 0);
3217 	unsigned int num_removed = 0;
3218 	unsigned int num_external = 0, num_internal = 0, num_reusable = 0;
3219 	unsigned int num_alt_internal = 0;
3220 	unsigned int num_compressed = 0, num_alt_compressed = 0;
3221 	unsigned short num_unwired = 0;
3222 	bool need_strong_sync = false;
3223 
3224 	/*
3225 	 * The pmap lock should be held here.  It will only be held shared in most if not all cases.
3226 	 */
3227 	pmap_assert_locked(pmap, PMAP_LOCK_HELD);
3228 
3229 	const pt_attr_t * const pt_attr = pmap_get_pt_attr(pmap);
3230 	const uint64_t pmap_page_size = PAGE_RATIO * pt_attr_page_size(pt_attr);
3231 	const uint64_t pmap_page_shift = pt_attr_leaf_shift(pt_attr);
3232 	vm_map_address_t va = start;
3233 	pt_entry_t *cpte = pmap_pte(pmap, va);
3234 	assert(cpte != NULL);
3235 
3236 	while (va < end) {
3237 		/**
3238 		 * We may need to sleep when taking the PVH lock below, and our pmap_pv_remove()
3239 		 * call below may also place the lock in sleep mode if processing a large PV list.
3240 		 * We therefore can't leave preemption disabled across that code, which means we
3241 		 * can't directly use the per-CPU prev_ptes array in that code.  Since that code
3242 		 * only cares about the physical address stored in each prev_ptes entry, we'll
3243 		 * use a local array to stash off only the 4-byte physical address index in order
3244 		 * to reduce stack usage.
3245 		 */
3246 		unsigned int pai_list[SPTM_MAPPING_LIMIT];
3247 		_Static_assert(SPTM_MAPPING_LIMIT <= 64,
3248 		    "SPTM_MAPPING_LIMIT value causes excessive stack usage for pai_list");
3249 
3250 		unsigned int num_mappings = (end - va) >> pmap_page_shift;
3251 		if (num_mappings > SPTM_MAPPING_LIMIT) {
3252 			num_mappings = SPTM_MAPPING_LIMIT;
3253 		}
3254 
3255 		/**
3256 		 * Disable preemption to ensure that we can safely access per-CPU mapping data after
3257 		 * issuing the SPTM call.
3258 		 */
3259 		disable_preemption();
3260 		/**
3261 		 * Enter the retype epoch for the batched unmap operation.  This is necessary because we
3262 		 * cannot reasonably hold the PVH locks for all pages mapped by the region during this
3263 		 * call, so a concurrent pmap_page_protect() operation against one of those pages may
3264 		 * race this call.  That should be perfectly fine as far as the PTE updates are concerned,
3265 		 * but if pmap_page_protect() then needs to retype the page, an SPTM violation may result
3266 		 * if it does not first drain our epoch.
3267 		 */
3268 		pmap_retype_epoch_enter();
3269 		sptm_unmap_region(pmap->ttep, va, num_mappings, sptm_flags);
3270 		pmap_retype_epoch_exit();
3271 
3272 		sptm_pte_t *prev_ptes = PERCPU_GET(pmap_sptm_percpu)->sptm_prev_ptes;
3273 		for (unsigned int i = 0; i < num_mappings; ++i, ++cpte) {
3274 			const pt_entry_t prev_pte = prev_ptes[i];
3275 
3276 			if (pte_is_compressed(prev_pte, cpte)) {
3277 				if (options & PMAP_OPTIONS_REMOVE) {
3278 					++num_compressed;
3279 					if (prev_pte & ARM_PTE_COMPRESSED_ALT) {
3280 						++num_alt_compressed;
3281 					}
3282 				}
3283 				pai_list[i] = INVALID_PAI;
3284 				continue;
3285 			} else if ((prev_pte & ARM_PTE_TYPE_MASK) == ARM_PTE_TYPE_FAULT) {
3286 				pai_list[i] = INVALID_PAI;
3287 				continue;
3288 			}
3289 
3290 			if (pte_is_wired(prev_pte)) {
3291 				num_unwired++;
3292 			}
3293 
3294 			const pmap_paddr_t pa = pte_to_pa(prev_pte);
3295 
3296 			if (__improbable(!pa_valid(pa))) {
3297 				pai_list[i] = INVALID_PAI;
3298 				continue;
3299 			}
3300 			pai_list[i] = pa_index(pa);
3301 		}
3302 
3303 		enable_preemption();
3304 		cpte -= num_mappings;
3305 
3306 		for (unsigned int i = 0; i < num_mappings; ++i, ++cpte) {
3307 			if (pai_list[i] == INVALID_PAI) {
3308 				continue;
3309 			}
3310 			locked_pvh_t locked_pvh;
3311 			if (__improbable(options & PMAP_OPTIONS_NOPREEMPT)) {
3312 				locked_pvh = pvh_lock_nopreempt(pai_list[i]);
3313 			} else {
3314 				locked_pvh = pvh_lock(pai_list[i]);
3315 			}
3316 
3317 			bool is_internal, is_altacct;
3318 			pv_remove_return_t remove_status = pmap_remove_pv(pmap, cpte, &locked_pvh, &is_internal, &is_altacct);
3319 
3320 			switch (remove_status) {
3321 			case PV_REMOVE_SUCCESS:
3322 				++num_removed;
3323 				if (is_altacct) {
3324 					assert(is_internal);
3325 					num_internal++;
3326 					num_alt_internal++;
3327 				} else if (is_internal) {
3328 					if (ppattr_test_reusable(pai_list[i])) {
3329 						num_reusable++;
3330 					} else {
3331 						num_internal++;
3332 					}
3333 				} else {
3334 					num_external++;
3335 				}
3336 				break;
3337 			default:
3338 				/*
3339 				 * PVE already removed; this can happen due to a concurrent pmap_disconnect()
3340 				 * executing before we grabbed the PVH lock.
3341 				 */
3342 				break;
3343 			}
3344 
3345 			pvh_unlock(&locked_pvh);
3346 		}
3347 
3348 		va += (num_mappings << pmap_page_shift);
3349 	}
3350 
3351 	if (__improbable(need_strong_sync)) {
3352 		arm64_sync_tlb(true);
3353 	}
3354 
3355 	/*
3356 	 *	Update the counts
3357 	 */
3358 	pmap_ledger_debit(pmap, task_ledgers.phys_mem, num_removed * pmap_page_size);
3359 
3360 	if (pmap != kernel_pmap) {
3361 		if (num_unwired != 0) {
3362 			ptd_info_t * const ptd_info = ptep_get_info(cpte - 1);
3363 			if (__improbable(os_atomic_sub_orig(&ptd_info->wiredcnt, num_unwired, relaxed) < num_unwired)) {
3364 				panic("%s: pmap %p VA [0x%llx, 0x%llx) (ptd info %p) wired count underflow", __func__, pmap,
3365 				    (unsigned long long)start, (unsigned long long)end, ptd_info);
3366 			}
3367 		}
3368 
3369 		/* update ledgers */
3370 		pmap_ledger_debit(pmap, task_ledgers.external, (num_external) * pmap_page_size);
3371 		pmap_ledger_debit(pmap, task_ledgers.reusable, (num_reusable) * pmap_page_size);
3372 		pmap_ledger_debit(pmap, task_ledgers.wired_mem, (num_unwired) * pmap_page_size);
3373 		pmap_ledger_debit(pmap, task_ledgers.internal, (num_internal) * pmap_page_size);
3374 		pmap_ledger_debit(pmap, task_ledgers.alternate_accounting, (num_alt_internal) * pmap_page_size);
3375 		pmap_ledger_debit(pmap, task_ledgers.alternate_accounting_compressed, (num_alt_compressed) * pmap_page_size);
3376 		pmap_ledger_debit(pmap, task_ledgers.internal_compressed, (num_compressed) * pmap_page_size);
3377 		/* make needed adjustments to phys_footprint */
3378 		pmap_ledger_debit(pmap, task_ledgers.phys_footprint,
3379 		    ((num_internal -
3380 		    num_alt_internal) +
3381 		    (num_compressed -
3382 		    num_alt_compressed)) * pmap_page_size);
3383 	}
3384 }
3385 
3386 
3387 /*
3388  *	Remove the given range of addresses
3389  *	from the specified map.
3390  *
3391  *	It is assumed that the start and end are properly
3392  *	rounded to the hardware page size.
3393  */
3394 void
3395 pmap_remove(
3396 	pmap_t pmap,
3397 	vm_map_address_t start,
3398 	vm_map_address_t end)
3399 {
3400 	pmap_remove_options(pmap, start, end, PMAP_OPTIONS_REMOVE);
3401 }
3402 
3403 MARK_AS_PMAP_TEXT vm_map_address_t
3404 pmap_remove_options_internal(
3405 	pmap_t pmap,
3406 	vm_map_address_t start,
3407 	vm_map_address_t end,
3408 	int options)
3409 {
3410 	vm_map_address_t eva = end;
3411 	tt_entry_t     *tte_p;
3412 	bool            unlock = true;
3413 
3414 	if (__improbable(end < start)) {
3415 		panic("%s: invalid address range %p, %p", __func__, (void*)start, (void*)end);
3416 	}
3417 	if (__improbable(pmap->type == PMAP_TYPE_COMMPAGE)) {
3418 		panic("%s: attempt to remove mappings from commpage pmap %p", __func__, pmap);
3419 	}
3420 
3421 	validate_pmap_mutable(pmap);
3422 
3423 	__unused const pt_attr_t * const pt_attr = pmap_get_pt_attr(pmap);
3424 
3425 	pmap_lock_mode_t lock_mode = PMAP_LOCK_SHARED;
3426 	pmap_lock(pmap, lock_mode);
3427 
3428 	tte_p = pmap_tte(pmap, start);
3429 
3430 	if ((tte_p == NULL) || ((*tte_p & ARM_TTE_TYPE_MASK) == ARM_TTE_TYPE_FAULT)) {
3431 		goto done;
3432 	}
3433 
3434 	assertf((*tte_p & ARM_TTE_TYPE_MASK) == ARM_TTE_TYPE_TABLE, "%s: invalid TTE %p (0x%llx) for pmap %p va 0x%llx",
3435 	    __func__, tte_p, (unsigned long long)*tte_p, pmap, (unsigned long long)start);
3436 
3437 	pmap_remove_range_options(pmap, start, end, options);
3438 
3439 	if (pmap->type != PMAP_TYPE_USER) {
3440 		goto done;
3441 	}
3442 
3443 	uint16_t refcnt = sptm_get_page_table_refcnt(tte_to_pa(*tte_p));
3444 	if (__improbable(refcnt == 0)) {
3445 		ptd_info_t *ptd_info = ptep_get_info((pt_entry_t*)ttetokv(*tte_p));
3446 		os_atomic_inc(&ptd_info->wiredcnt, relaxed); // Prevent someone else from freeing the table if we need to drop the lock
3447 		if (!pmap_lock_shared_to_exclusive(pmap)) {
3448 			pmap_lock(pmap, PMAP_LOCK_EXCLUSIVE);
3449 		}
3450 		lock_mode = PMAP_LOCK_EXCLUSIVE;
3451 		refcnt = sptm_get_page_table_refcnt(tte_to_pa(*tte_p));
3452 		if ((os_atomic_dec(&ptd_info->wiredcnt, relaxed) == 0) && (refcnt == 0)) {
3453 			/**
3454 			 * Drain any concurrent retype-sensitive SPTM operations.  This is needed to
3455 			 * ensure that we don't unmap the page table and retype it while those operations
3456 			 * are still finishing on other CPUs, leading to an SPTM violation.  In particular,
3457 			 * the multipage batched cacheability/attribute update code may issue SPTM calls
3458 			 * without holding the relevant PVH or pmap locks, so we can't guarantee those
3459 			 * calls have actually completed despite observing refcnt == 0.
3460 			 *
3461 			 * At this point, we CAN guarantee that:
3462 			 * 1) All prior PTE removals required to produce refcnt == 0 have
3463 			 *    completed and been synchronized for all observers by DSB, and the
3464 			 *    relevant PV list entries removed.  Subsequent calls not already in the
3465 			 *    retype epoch will no longer observe these mappings.
3466 			 * 2) We now hold the pmap lock exclusive, so there will be no further attempt
3467 			 *    to enter mappings in this page table before it is unmapped.
3468 			 */
3469 			pmap_retype_epoch_prepare_drain();
3470 			pmap_retype_epoch_drain();
3471 			pmap_tte_deallocate(pmap, start, tte_p, pt_attr_twig_level(pt_attr));
3472 			unlock = false; // pmap_tte_deallocate() has dropped the lock
3473 		}
3474 	}
3475 done:
3476 	if (unlock) {
3477 		pmap_unlock(pmap, lock_mode);
3478 	}
3479 
3480 	return eva;
3481 }
3482 
3483 void
3484 pmap_remove_options(
3485 	pmap_t pmap,
3486 	vm_map_address_t start,
3487 	vm_map_address_t end,
3488 	int options)
3489 {
3490 	vm_map_address_t va;
3491 
3492 	if (pmap == PMAP_NULL) {
3493 		return;
3494 	}
3495 
3496 	__unused const pt_attr_t * const pt_attr = pmap_get_pt_attr(pmap);
3497 
3498 	PMAP_TRACE(2, PMAP_CODE(PMAP__REMOVE) | DBG_FUNC_START,
3499 	    VM_KERNEL_ADDRHIDE(pmap), VM_KERNEL_ADDRHIDE(start),
3500 	    VM_KERNEL_ADDRHIDE(end));
3501 
3502 #if MACH_ASSERT
3503 	if ((start | end) & pt_attr_leaf_offmask(pt_attr)) {
3504 		panic("pmap_remove_options() pmap %p start 0x%llx end 0x%llx",
3505 		    pmap, (uint64_t)start, (uint64_t)end);
3506 	}
3507 	if ((end < start) || (start < pmap->min) || (end > pmap->max)) {
3508 		panic("pmap_remove_options(): invalid address range, pmap=%p, start=0x%llx, end=0x%llx",
3509 		    pmap, (uint64_t)start, (uint64_t)end);
3510 	}
3511 #endif
3512 
3513 	/*
3514 	 * We allow single-page requests to execute non-preemptibly,
3515 	 * as it doesn't make sense to sample AST_URGENT for a single-page
3516 	 * operation, and there are a couple of special use cases that
3517 	 * require a non-preemptible single-page operation.
3518 	 */
3519 	if ((end - start) > (pt_attr_page_size(pt_attr) * PAGE_RATIO)) {
3520 		pmap_verify_preemptible();
3521 	}
3522 
3523 	/*
3524 	 *      Invalidate the translation buffer first
3525 	 */
3526 	va = start;
3527 	while (va < end) {
3528 		vm_map_address_t l;
3529 
3530 		l = ((va + pt_attr_twig_size(pt_attr)) & ~pt_attr_twig_offmask(pt_attr));
3531 		if (l > end) {
3532 			l = end;
3533 		}
3534 
3535 		va = pmap_remove_options_internal(pmap, va, l, options);
3536 	}
3537 
3538 	PMAP_TRACE(2, PMAP_CODE(PMAP__REMOVE) | DBG_FUNC_END);
3539 }
3540 
3541 
3542 /*
3543  *	Remove phys addr if mapped in specified map
3544  */
3545 void
3546 pmap_remove_some_phys(
3547 	__unused pmap_t map,
3548 	__unused ppnum_t pn)
3549 {
3550 	/* Implement to support working set code */
3551 }
3552 
3553 /*
3554  * Implementation of PMAP_SWITCH_USER that Mach VM uses to
3555  * switch a thread onto a new vm_map.
3556  */
3557 void
3558 pmap_switch_user(thread_t thread, vm_map_t new_map)
3559 {
3560 	pmap_t new_pmap = new_map->pmap;
3561 
3562 
3563 	thread->map = new_map;
3564 	pmap_set_pmap(new_pmap, thread);
3565 
3566 }
3567 void
3568 pmap_set_pmap(
3569 	pmap_t pmap,
3570 	__unused thread_t thread)
3571 {
3572 	pmap_switch(pmap);
3573 }
3574 
3575 MARK_AS_PMAP_TEXT void
3576 pmap_switch_internal(
3577 	pmap_t pmap)
3578 {
3579 	validate_pmap_mutable(pmap);
3580 	__unused const pt_attr_t * const pt_attr = pmap_get_pt_attr(pmap);
3581 	const uint16_t asid_index = PMAP_HWASID(pmap);
3582 	if (__improbable((asid_index == 0) && (pmap != kernel_pmap))) {
3583 		panic("%s: attempt to activate pmap with invalid ASID %p", __func__, pmap);
3584 	}
3585 
3586 #if __ARM_KERNEL_PROTECT__
3587 	asid_index >>= 1;
3588 #endif
3589 
3590 	if (asid_index > 0) {
3591 		pmap_update_plru(asid_index);
3592 	}
3593 
3594 	__unused const sptm_return_t sptm_return = sptm_switch_root(pmap->ttep);
3595 
3596 #if DEVELOPMENT || DEBUG
3597 	if (__improbable(sptm_return & SPTM_SWITCH_ASID_TLBI_FLUSH)) {
3598 		os_atomic_inc(&pmap_asid_flushes, relaxed);
3599 	}
3600 
3601 	if (__improbable(sptm_return & SPTM_SWITCH_RCTX_FLUSH)) {
3602 		os_atomic_inc(&pmap_speculation_restrictions, relaxed);
3603 	}
3604 #endif /* DEVELOPMENT || DEBUG */
3605 }
3606 
3607 void
3608 pmap_switch(
3609 	pmap_t pmap)
3610 {
3611 	PMAP_TRACE(1, PMAP_CODE(PMAP__SWITCH) | DBG_FUNC_START, VM_KERNEL_ADDRHIDE(pmap), PMAP_VASID(pmap), PMAP_HWASID(pmap));
3612 	pmap_switch_internal(pmap);
3613 	PMAP_TRACE(1, PMAP_CODE(PMAP__SWITCH) | DBG_FUNC_END);
3614 }
3615 
3616 void
3617 pmap_page_protect(
3618 	ppnum_t ppnum,
3619 	vm_prot_t prot)
3620 {
3621 	pmap_page_protect_options(ppnum, prot, 0, NULL);
3622 }
3623 
3624 /**
3625  *  Helper function for performing per-mapping accounting following an SPTM disjoint unmap request.
3626  *
3627  * @note [pmap] cannot be the kernel pmap. This is because we do not maintain a ledger in the
3628  *       kernel pmap.
3629  *
3630  * @param pmap The pmap that contained the mapping
3631  * @param pai The physical page index mapped by the mapping
3632  * @param is_compressed Indicates whether the operation was an unmap-to-compress vs. a full unmap
3633  * @param is_internal Indicates whether the mapping was for an internal (aka anonymous) VM page
3634  * @param is_altacct Indicates whether the mapping was subject to alternate accounting.
3635  */
3636 static void
3637 pmap_disjoint_unmap_accounting(pmap_t pmap, unsigned int pai, bool is_compressed, bool is_internal, bool is_altacct)
3638 {
3639 	const pt_attr_t *const pt_attr = pmap_get_pt_attr(pmap);
3640 	pvh_assert_locked(pai);
3641 
3642 	assert(pmap != kernel_pmap);
3643 
3644 	if (is_internal &&
3645 	    !is_altacct &&
3646 	    ppattr_test_reusable(pai)) {
3647 		pmap_ledger_debit(pmap, task_ledgers.reusable, pt_attr_page_size(pt_attr) * PAGE_RATIO);
3648 	} else if (!is_internal) {
3649 		pmap_ledger_debit(pmap, task_ledgers.external, pt_attr_page_size(pt_attr) * PAGE_RATIO);
3650 	}
3651 
3652 	if (is_altacct) {
3653 		assert(is_internal);
3654 		pmap_ledger_debit(pmap, task_ledgers.internal, pt_attr_page_size(pt_attr) * PAGE_RATIO);
3655 		pmap_ledger_debit(pmap, task_ledgers.alternate_accounting, pt_attr_page_size(pt_attr) * PAGE_RATIO);
3656 		if (is_compressed) {
3657 			pmap_ledger_credit(pmap, task_ledgers.internal_compressed, pt_attr_page_size(pt_attr) * PAGE_RATIO);
3658 			pmap_ledger_credit(pmap, task_ledgers.alternate_accounting_compressed, pt_attr_page_size(pt_attr) * PAGE_RATIO);
3659 		}
3660 	} else if (ppattr_test_reusable(pai)) {
3661 		assert(is_internal);
3662 		if (is_compressed) {
3663 			pmap_ledger_credit(pmap, task_ledgers.internal_compressed, pt_attr_page_size(pt_attr) * PAGE_RATIO);
3664 			/* was not in footprint, but is now */
3665 			pmap_ledger_credit(pmap, task_ledgers.phys_footprint, pt_attr_page_size(pt_attr) * PAGE_RATIO);
3666 		}
3667 	} else if (is_internal) {
3668 		pmap_ledger_debit(pmap, task_ledgers.internal, pt_attr_page_size(pt_attr) * PAGE_RATIO);
3669 
3670 		/*
3671 		 * Update all stats related to physical footprint, which only
3672 		 * deals with internal pages.
3673 		 */
3674 		if (is_compressed) {
3675 			/*
3676 			 * This removal is only being done so we can send this page to
3677 			 * the compressor; therefore it mustn't affect total task footprint.
3678 			 */
3679 			pmap_ledger_credit(pmap, task_ledgers.internal_compressed, pt_attr_page_size(pt_attr) * PAGE_RATIO);
3680 		} else {
3681 			/*
3682 			 * This internal page isn't going to the compressor, so adjust stats to keep
3683 			 * phys_footprint up to date.
3684 			 */
3685 			pmap_ledger_debit(pmap, task_ledgers.phys_footprint, pt_attr_page_size(pt_attr) * PAGE_RATIO);
3686 		}
3687 	} else {
3688 		/* external page: no impact on ledgers */
3689 	}
3690 }
3691 
3692 /**
3693  * Helper function for issuing a disjoint unmap request to the SPTM and performing
3694  * related accounting.  This function uses the 'prev_ptes' list generated by
3695  * the sptm_unmap_disjoint() call to determine whether said call altered the
3696  * relevant PTEs in a manner that would require accounting updates.
3697  *
3698  * @param pa The physical address against which the disjoint unmap will be issued.
3699  * @param num_mappings The number of disjoint mappings for the SPTM to update.
3700  *                     The per-CPU sptm_ops array should contain the same number
3701  *                     of individual disjoint requests.
3702  */
3703 static void
3704 pmap_disjoint_unmap(pmap_paddr_t pa, unsigned int num_mappings)
3705 {
3706 	const unsigned int pai = pa_index(pa);
3707 
3708 	pvh_assert_locked(pai);
3709 
3710 	assert(num_mappings <= SPTM_MAPPING_LIMIT);
3711 
3712 	assert(get_preemption_level() > 0);
3713 	pmap_sptm_percpu_data_t *sptm_pcpu = PERCPU_GET(pmap_sptm_percpu);
3714 
3715 	sptm_unmap_disjoint(pa, sptm_pcpu->sptm_ops_pa, num_mappings);
3716 
3717 	for (unsigned int cur_mapping = 0; cur_mapping < num_mappings; ++cur_mapping) {
3718 		pt_entry_t prev_pte = sptm_pcpu->sptm_prev_ptes[cur_mapping];
3719 
3720 		pt_desc_t * const ptdp = sptm_pcpu->sptm_ptds[cur_mapping];
3721 		const pmap_t pmap = ptdp->pmap;
3722 
3723 		assertf(((prev_pte & ARM_PTE_TYPE_MASK) == ARM_PTE_TYPE_FAULT) ||
3724 		    ((pte_to_pa(prev_pte) & ~PAGE_MASK) == pa), "%s: prev_pte 0x%llx does not map pa 0x%llx",
3725 		    __func__, (unsigned long long)prev_pte, (unsigned long long)pa);
3726 
3727 		const pt_attr_t *const pt_attr = pmap_get_pt_attr(pmap);
3728 		pmap_ledger_debit(pmap, task_ledgers.phys_mem, pt_attr_page_size(pt_attr) * PAGE_RATIO);
3729 
3730 		if (pmap != kernel_pmap) {
3731 			/*
3732 			 * If the prior PTE is invalid (which may happen due to a concurrent remove operation),
3733 			 * the compressed marker won't be written so we shouldn't account the mapping as compressed.
3734 			 */
3735 			const bool is_compressed = (((prev_pte & ARM_PTE_TYPE_MASK) != ARM_PTE_TYPE_FAULT) &&
3736 			    ((sptm_pcpu->sptm_ops[cur_mapping].pte_template & ARM_PTE_COMPRESSED_MASK) != 0));
3737 			const bool is_internal = (sptm_pcpu->sptm_acct_flags[cur_mapping] & PMAP_SPTM_FLAG_INTERNAL) != 0;
3738 			const bool is_altacct = (sptm_pcpu->sptm_acct_flags[cur_mapping] & PMAP_SPTM_FLAG_ALTACCT) != 0;
3739 
3740 			/*
3741 			 * The rule is that accounting related to PTE contents (wired, PTD refcount)
3742 			 * must be updated by whoever clears the PTE, while accounting related to physical page
3743 			 * attributes must be updated by whoever clears the PVE.  We therefore always call
3744 			 * pmap_disjoint_unmap_accounting() here since we're removing the PVE, but only update
3745 			 * wired/PTD accounting if the prior PTE was valid.
3746 			 */
3747 			pmap_disjoint_unmap_accounting(pmap, pai, is_compressed, is_internal, is_altacct);
3748 
3749 			if ((prev_pte & ARM_PTE_TYPE_MASK) == ARM_PTE_TYPE_FAULT) {
3750 				continue;
3751 			}
3752 
3753 			if (pte_is_wired(prev_pte)) {
3754 				pmap_ledger_debit(pmap, task_ledgers.wired_mem, pt_attr_page_size(pt_attr) * PAGE_RATIO);
3755 				if (__improbable(os_atomic_dec_orig(&sptm_pcpu->sptm_ptd_info[cur_mapping]->wiredcnt, relaxed) == 0)) {
3756 					panic("%s: over-unwire of ptdp %p, ptd info %p", __func__,
3757 					    ptdp, sptm_pcpu->sptm_ptd_info[cur_mapping]);
3758 				}
3759 			}
3760 		}
3761 	}
3762 }
3763 
3764 /**
3765  * The following two functions, pmap_multipage_op_submit_disjoint() and
3766  * pmap_multipage_op_add_page(), are intended to allow callers to manage batched SPTM
3767  * operations that may span multiple physical pages.  They are intended to operate in
3768  * a way that allows callers such as pmap_page_protect_options_with_flush_range() to
3769  * insert mappings into the per-CPU SPTM disjoint ops array in the same manner that
3770  * they would for an ordinary single-page operation.
3771  * Functions such as pmap_page_protect_options_with_flush_range() operate on a single
3772  * physical page but may be passed a non-NULL flush_range object to indicate that the
3773  * call is part of a larger batched operation which may span multiple physical pages.
3774  * In that scenario, these functions are intended to be used as follows:
3775  * 1) Call pmap_multipage_op_add_page() to insert a "header" for the page into the per-
3776  *    CPU SPTM ops array.  Use the return value from this call as the starting index
3777  *    at which to add ordinary mapping entries into the same array.
3778  * 2) Insert sptm_disjoint_op_t entries into the ops array in the normal manner until
3779  *    the array is full, the SPTM options required for the upcoming sequence of pages
3780  *    need to change, or the current mapping matches flush_range->current_ptep.
3781  *    In the latter case, pmap_insert_flush_range_template() may instead be used
3782  *    to insert the mapping into the per-CPU SPTM region templates array.  See the
3783  *    documentation for pmap_insert_flush_range_template() below.
3784  * 3) If the array is full, call pmap_multipage_op_submit_disjoint() and return to step 1).
3785  * 4) If the SPTM options need to change, call pmap_multipage_op_add_page() to insert
3786  *    a new header with the updated options and, using the return value as the new
3787  *    insertion point for the ops array, resume step 2).
3788  * 5) Upon completion, if there are any pending not-yet-submitted mappings, do not
3789  *    submit those mappings to the SPTM as would ordinarily be done for a single-page
3790  *    call.  These trailing mappings will be submitted as part of the next batch,
3791  *    or by the next-higher caller if the range operation is complete.
3792  *
3793  * Note that, as a performance optimization, the caller may track the insertion
3794  * point in the disjoint ops array locally (i.e. without incrementing
3795  * flush_range->pending_disjoint_entries on every iteration, as long as it takes care to do the
3796  * following:
3797  * 1) Initialize and update that insertion point as described in steps 1) and 4) above.
3798  * 2) Pass the updated insertion point as the 'pending_disjoint_entries' parameter into the calls
3799  *    in steps 3) and 4) above.
3800  * 3) Update flush_range->pending_disjoint_entries with the locally-maintained value along with
3801  *    step 5) above.
3802  */
3803 
3804 /**
3805  * Submit any pending disjoint multi-page mapping updates to the SPTM.
3806  *
3807  * @note This function must be called with preemption disabled, and will drop
3808  *       the preemption-disable count upon submitting to the SPTM.
3809  * @note [pending_disjoint_entries] must include *all* pending entries in the SPTM ops array,
3810  *       including physical address "header" entries.
3811  * @note This function automatically updates the per_paddr_header.num_mappings field
3812  *       for the most recent physical address header in the SPTM ops array to its final
3813  *       value.
3814  *
3815  * @param pending_disjoint_entries The number of not-yet-submitted mappings according to the caller.
3816  *                        This value may be greater than [flush_range]->pending_disjoint_entries if
3817  *                        the caller has inserted mappings into the ops array without
3818  *                        updating [flush_range]->pending_disjoint_entries, in which case this
3819  *                        function will update [flush_range]->pending_disjoint_entries with the
3820  *                        caller's value.
3821  * @param flush_range The object tracking the current state of the multipage disjoint
3822  *                    operation.
3823  */
3824 static inline void
3825 pmap_multipage_op_submit_disjoint(unsigned int pending_disjoint_entries, pmap_tlb_flush_range_t *flush_range)
3826 {
3827 	/**
3828 	 * Reconcile the number of pending entries as tracked by the caller with the
3829 	 * number of pending entries tracked by flush_range.  If the caller's value is
3830 	 * greater, we assume the caller has inserted locally-tracked mappings into the
3831 	 * array without directly updating flush_range->pending_disjoint_entries.  Otherwise, we
3832 	 * assume the caller has no locally-tracked mappings and is simply trying to
3833 	 * purge any pending mappings from a prior call sequence.
3834 	 */
3835 	if (pending_disjoint_entries > flush_range->pending_disjoint_entries) {
3836 		flush_range->pending_disjoint_entries = pending_disjoint_entries;
3837 	} else {
3838 		assert(pending_disjoint_entries == 0);
3839 	}
3840 	if (flush_range->pending_disjoint_entries != 0) {
3841 		assert(get_preemption_level() > 0);
3842 		/**
3843 		 * Compute the correct number of mappings for the most recent paddr
3844 		 * header based on the current position in the SPTM ops array.
3845 		 */
3846 		flush_range->current_header->per_paddr_header.num_mappings =
3847 		    flush_range->pending_disjoint_entries - flush_range->current_header_first_mapping_index;
3848 		const sptm_return_t sptm_return = sptm_update_disjoint_multipage(
3849 			PERCPU_GET(pmap_sptm_percpu)->sptm_ops_pa, flush_range->pending_disjoint_entries);
3850 
3851 		/**
3852 		 * We may be submitting the batch and exiting the epoch partway through
3853 		 * processing the PV list for a page.  That's fine, because in that case we'll
3854 		 * hold the PV lock for that page, which will prevent mappings of that page from
3855 		 * being disconnected and will prevent the completion of pmap_remove() against
3856 		 * any of those mappings, thus also guaranteeing the relevant page table pages
3857 		 * can't be freed.  The epoch still protects mappings for any prior page in
3858 		 * the batch, whose PV locks are no longer held.
3859 		 */
3860 		pmap_retype_epoch_exit();
3861 		enable_preemption();
3862 		if (flush_range->pending_region_entries != 0) {
3863 			flush_range->processed_entries += flush_range->pending_disjoint_entries;
3864 		} else {
3865 			flush_range->processed_entries = 0;
3866 		}
3867 		flush_range->pending_disjoint_entries = 0;
3868 		if (sptm_return == SPTM_UPDATE_DELAYED_TLBI) {
3869 			flush_range->ptfr_flush_needed = true;
3870 		}
3871 	}
3872 }
3873 
3874 /**
3875  * Insert a new physical address "header" entry into the per-CPU SPTM ops array for a
3876  * multi-page SPTM operation.  It is expected that the caller will subsequently add
3877  * mapping entries for this physical address into the array.
3878  *
3879  * @note This function will disable preemption upon creation of the first paddr header
3880  *       (index 0 in the per-CPU SPTM ops array) and it is expected that
3881  *       pmap_multipage_op_submit() will subsequently be called on the same CPU.
3882  * @note Before inserting the new header, this function automatically updates the
3883  *       per_paddr_header.num_mappings field for the previous physical address header
3884  *       (if present) in the SPTM ops array to its final value.
3885  *
3886  * @param phys The physical address for which to insert a header entry.
3887  * @param inout_pending_disjoint_entries
3888  *              [input] The number of not-yet-submitted mappings according to the caller.
3889  *                      This value may be greater than [flush_range]->pending_disjoint_entries if
3890  *                      the caller has inserted mappings into the ops array without
3891  *                      updating [flush_range]->pending_disjoint_entries, in which case this
3892  *                      function will update [flush_range]->pending_disjoint_entries with the
3893  *                      caller's value.
3894  *              [output] Returns the starting index at which the caller should insert mapping
3895  *                       entries into the per-CPU SPTM ops array.
3896  * @param sptm_update_options SPTM_UPDATE_* flags to pass to the SPTM call.
3897  *                            SPTM_UPDATE_SKIP_PAPT is automatically inserted by this
3898  *                            function.
3899  * @param flush_range The object tracking the current state of the multipage operation.
3900  *
3901  * @return True if the region operation was submitted to the SPTM due to the ops array already
3902  *         being full, false otherwise.  In the former case, the new header will not be added
3903  *         to the array; the caller will need to re-invoke this function after taking any
3904  *         necessary post-submission action (such as enabling preemption).
3905  */
3906 static inline bool
3907 pmap_multipage_op_add_page(
3908 	pmap_paddr_t phys,
3909 	unsigned int *inout_pending_disjoint_entries,
3910 	uint32_t sptm_update_options,
3911 	pmap_tlb_flush_range_t *flush_range)
3912 {
3913 	unsigned int pending_disjoint_entries = *inout_pending_disjoint_entries;
3914 
3915 	/**
3916 	 * Reconcile the number of pending entries as tracked by the caller with the
3917 	 * number of pending entries tracked by flush_range.  If the caller's value is
3918 	 * greater, we assume the caller has inserted locally-tracked mappings into the
3919 	 * array without directly updating flush_range->pending_disjoint_entries.  Otherwise, we
3920 	 * assume the caller has no locally-tracked mappings and is adding its paddr
3921 	 * header for the first time.
3922 	 */
3923 	if (pending_disjoint_entries > flush_range->pending_disjoint_entries) {
3924 		flush_range->pending_disjoint_entries = pending_disjoint_entries;
3925 	} else {
3926 		assert(pending_disjoint_entries == 0);
3927 	}
3928 	if (flush_range->pending_disjoint_entries >= (SPTM_MAPPING_LIMIT - 1)) {
3929 		/**
3930 		 * If the SPTM ops array is either full or only has space for the paddr
3931 		 * header, there won't be room for mapping entries, so submit the pending
3932 		 * mappings to the SPTM now, and return to allow the caller to take
3933 		 * any necessary post-submission action.
3934 		 */
3935 		pmap_multipage_op_submit_disjoint(pending_disjoint_entries, flush_range);
3936 		*inout_pending_disjoint_entries = 0;
3937 		return true;
3938 	}
3939 	pending_disjoint_entries = flush_range->pending_disjoint_entries;
3940 
3941 	sptm_update_options |= SPTM_UPDATE_SKIP_PAPT;
3942 	if (pending_disjoint_entries == 0) {
3943 		disable_preemption();
3944 		/**
3945 		 * Enter the retype epoch while we gather the disjoint update arguments
3946 		 * and issue the SPTM call.  Since this operation may cover multiple physical
3947 		 * pages, we may construct the argument array and invoke the SPTM without holding
3948 		 * all relevant PVH locks or pmap locks.  We therefore need to record that we are
3949 		 * collecting and modifying mapping state so that e.g. pmap_page_protect() does
3950 		 * not attempt to retype the underlying pages and pmap_remove() does not attempt
3951 		 * to free the page tables used for these mappings without first draining our epoch.
3952 		 */
3953 		pmap_retype_epoch_enter();
3954 		flush_range->pending_disjoint_entries = 1;
3955 	} else {
3956 		/**
3957 		 * Before inserting the new header, update the prior header's number
3958 		 * of paddr-specific mappings to its final value.
3959 		 */
3960 		assert(flush_range->current_header != NULL);
3961 		flush_range->current_header->per_paddr_header.num_mappings =
3962 		    pending_disjoint_entries - flush_range->current_header_first_mapping_index;
3963 	}
3964 	sptm_disjoint_op_t *sptm_ops = PERCPU_GET(pmap_sptm_percpu)->sptm_ops;
3965 	flush_range->current_header = (sptm_update_disjoint_multipage_op_t*)&sptm_ops[pending_disjoint_entries];
3966 	flush_range->current_header_first_mapping_index = ++pending_disjoint_entries;
3967 	flush_range->current_header->per_paddr_header.paddr = phys;
3968 	flush_range->current_header->per_paddr_header.num_mappings = 0;
3969 	flush_range->current_header->per_paddr_header.options = sptm_update_options;
3970 
3971 	*inout_pending_disjoint_entries = pending_disjoint_entries;
3972 	return false;
3973 }
3974 
3975 /**
3976  * The following two functions, pmap_multipage_op_submit_region() and
3977  * pmap_insert_flush_range_template(), are meant to be used in a similar fashion
3978  * to pmap_multipage_op_submit_disjoint() and pmap_multipage_op_add_page(),
3979  * but for the specific case in which a given mapping within a PV list happens
3980  * to map the current VA within a VA region being operated on by
3981  * phys_attribute_clear_range().  This allows the pmap to further optimize
3982  * the SPTM calls by using sptm_update_region() to modify all mappings within
3983  * the VA region, which requires far fewer table walks than a disjoint operation.
3984  * Since the starting VA of the region, the owning pmap, and the insertion point
3985  * within the per-CPU region templates array are already known, these functions
3986  * don't require the special "header" entry or the complex array position tracking
3987  * of their disjoint equivalents above.
3988  * Note that these functions may be used together with the disjoint functions above;
3989  * these functions can be used for the "primary" mappings corresponding to the VA
3990  * region being manipulated by the VM layer, while the disjoint functions can be
3991  * used for any alias mappings of the underlying pages which fall outside that
3992  * VA region.
3993  */
3994 
3995 /**
3996  * Submit any pending region-based templates for the specified flush_range.
3997  *
3998  * @note This function must be called with preemption disabled, and will drop
3999  *       the preemption-disable count upon submitting to the SPTM.
4000  *
4001  * @param flush_range The object tracking the current state of the region operation.
4002  */
4003 static inline void
4004 pmap_multipage_op_submit_region(pmap_tlb_flush_range_t *flush_range)
4005 {
4006 	if (flush_range->pending_region_entries != 0) {
4007 		assert(get_preemption_level() > 0);
4008 		pmap_assert_locked(flush_range->ptfr_pmap, PMAP_LOCK_SHARED);
4009 		/**
4010 		 * If there are any pending disjoint entries, we're already in a retype epoch.
4011 		 * For disjoint entries, we need to hold the epoch during the entire time we
4012 		 * construct the disjoint ops array because those ops may point to some arbitrary
4013 		 * pmap and we need to ensure the relevant page tables and even the pmap itself
4014 		 * aren't concurrently reclaimed while our ops array points to them.
4015 		 * But for a region op like this, we know we already hold the relevant pmap lock
4016 		 * so none of the above can happen concurrently.  We therefore only need to hold
4017 		 * the epoch across the SPTM call itself to prevent a concurrent unmap operation
4018 		 * from attempting to retype the mapped pages while our SPTM call has them in-
4019 		 * flight.
4020 		 */
4021 		if (flush_range->pending_disjoint_entries == 0) {
4022 			pmap_retype_epoch_enter();
4023 		}
4024 		const sptm_return_t sptm_return = sptm_update_region(flush_range->ptfr_pmap->ttep,
4025 		    flush_range->pending_region_start, flush_range->pending_region_entries,
4026 		    PERCPU_GET(pmap_sptm_percpu)->sptm_templates_pa,
4027 		    SPTM_UPDATE_PERMS_AND_WAS_WRITABLE | SPTM_UPDATE_AF | SPTM_UPDATE_DEFER_TLBI);
4028 		if (flush_range->pending_disjoint_entries == 0) {
4029 			pmap_retype_epoch_exit();
4030 		}
4031 		enable_preemption();
4032 		if (flush_range->pending_disjoint_entries != 0) {
4033 			flush_range->processed_entries += flush_range->pending_region_entries;
4034 		} else {
4035 			flush_range->processed_entries = 0;
4036 		}
4037 		flush_range->pending_region_start += (flush_range->pending_region_entries <<
4038 		        pmap_get_pt_attr(flush_range->ptfr_pmap)->pta_page_shift);
4039 		flush_range->pending_region_entries = 0;
4040 		if (sptm_return == SPTM_UPDATE_DELAYED_TLBI) {
4041 			flush_range->ptfr_flush_needed = true;
4042 		}
4043 	}
4044 }
4045 
4046 /**
4047  * Insert a PTE template into the per-CPU SPTM region ops array.
4048  * This is meant to be used as a performance optimization for the case in which a given
4049  * mapping being processed by a function such as pmap_page_protect_options_with_flush_range()
4050  * happens to map the current iteration position within [flush_range]'s VA region.
4051  * In this case the mapping can be inserted as a region-based template rather than a disjoint
4052  * operation as would be done in the general case.  The idea is that region-based SPTM
4053  * operations are significantly less expensive than disjoint operations, because each region
4054  * operation only requires a single page table walk at the beginning vs. a table walk for
4055  * each mapping in the disjoint case.  Since the majority of mappings processed by a flush
4056  * range operation belong to the main flush range VA region (i.e. alias mappings outside
4057  * the region are less common), the performance improvement can be significant.
4058  *
4059  * @note This function will disable preemption upon inserting the first entry into the
4060  *       per-CPU templates array, and will re-enable preemption upon submitting the region
4061  *       operation to the SPTM.
4062  *
4063  * @param template The PTE template to insert into the per-CPU templates array.
4064  * @param flush_range The object tracking the current state of the region operation.
4065  *
4066  * @return True if the region operation was submitted to the SPTM, false otherwise.
4067  */
4068 static inline bool
4069 pmap_insert_flush_range_template(pt_entry_t template, pmap_tlb_flush_range_t *flush_range)
4070 {
4071 	if (flush_range->pending_region_entries == 0) {
4072 		disable_preemption();
4073 	}
4074 	flush_range->region_entry_added = true;
4075 	PERCPU_GET(pmap_sptm_percpu)->sptm_templates[flush_range->pending_region_entries++] = template;
4076 	if (flush_range->pending_region_entries == SPTM_MAPPING_LIMIT) {
4077 		pmap_multipage_op_submit_region(flush_range);
4078 		return true;
4079 	}
4080 	return false;
4081 }
4082 
4083 /**
4084  * Wrapper function for submitting any pending operations, region-based or disjoint,
4085  * tracked by a flush range object.  This is meant to be used by the top-level caller that
4086  * iterates over the flush range's VA region and calls functions such as
4087  * pmap_page_protect_options_with_flush_range() or arm_force_fast_fault_with_flush_range()
4088  * to construct the relevant SPTM operations arrays.
4089  *
4090  * @param flush_range The object tracking the current state of region and/or disjoint operations.
4091  */
4092 static inline void
4093 pmap_multipage_op_submit(pmap_tlb_flush_range_t *flush_range)
4094 {
4095 	pmap_multipage_op_submit_disjoint(0, flush_range);
4096 	pmap_multipage_op_submit_region(flush_range);
4097 }
4098 
4099 /**
4100  * This is an internal-only flag that indicates the caller of pmap_page_protect_options_with_flush_range()
4101  * is removing/updating all mappings in preparation for a retype operation.  In this case
4102  * pmap_page_protect_options() will assume (and assert) that the PVH lock for the physical page is held
4103  * by the calller, and will perform the necessary retype epoch drain prior to returning.
4104  */
4105 #define PMAP_OPTIONS_PPO_PENDING_RETYPE 0x80000000
4106 _Static_assert(PMAP_OPTIONS_PPO_PENDING_RETYPE & PMAP_OPTIONS_RESERVED_MASK,
4107     "PMAP_OPTIONS_PPO_PENDING_RETYPE outside reserved encoding space");
4108 
4109 /**
4110  * Lower the permission for all mappings to a given page. If VM_PROT_NONE is specified,
4111  * the mappings will be removed.
4112  *
4113  * @param ppnum Page number to lower the permission of.
4114  * @param prot The permission to lower to.
4115  * @param options PMAP_OPTIONS_NOFLUSH indicates TLBI flush is not needed.
4116  *                PMAP_OPTIONS_PPO_PENDING_RETYPE indicates the PVH lock for ppnum is
4117  *                already locked and a retype epoch drain shold be performed.
4118  *                PMAP_OPTIONS_COMPRESSOR indicates the function is called by the
4119  *                VM compressor.
4120  * @param locked_pvh If non-NULL, this indicates the PVH lock for [ppnum] is already locked
4121  *                   by the caller.  This is an input/output parameter which may be updated
4122  *                   to reflect a new PV head value to be passed to a later call to pvh_unlock().
4123  * @param flush_range When present, this function will skip the TLB flush for the
4124  *                    mappings that are covered by the range, leaving that to be
4125  *                    done later by the caller.  It may also avoid submitting mapping
4126  *                    updates directly to the SPTM, instead accumulating them in a
4127  *                    per-CPU array to be submitted later by the caller.
4128  *
4129  * @note PMAP_OPTIONS_NOFLUSH and flush_range cannot both be specified.
4130  */
4131 MARK_AS_PMAP_TEXT static void
4132 pmap_page_protect_options_with_flush_range(
4133 	ppnum_t ppnum,
4134 	vm_prot_t prot,
4135 	unsigned int options,
4136 	locked_pvh_t *locked_pvh,
4137 	pmap_tlb_flush_range_t *flush_range)
4138 {
4139 	pmap_paddr_t phys = ptoa(ppnum);
4140 	locked_pvh_t local_locked_pvh = {.pvh = 0};
4141 	pv_entry_t *pve_p = NULL;
4142 	pv_entry_t *pveh_p = NULL;
4143 	pv_entry_t *pvet_p = NULL;
4144 	pt_entry_t *pte_p = NULL;
4145 	pv_entry_t *new_pve_p = NULL;
4146 	pt_entry_t *new_pte_p = NULL;
4147 
4148 	bool remove = false;
4149 	unsigned int pvh_cnt = 0;
4150 	unsigned int num_mappings = 0, num_skipped_mappings = 0;
4151 
4152 	assert(ppnum != vm_page_fictitious_addr);
4153 
4154 	/**
4155 	 * Assert that PMAP_OPTIONS_NOFLUSH and flush_range cannot both be specified.
4156 	 *
4157 	 * PMAP_OPTIONS_NOFLUSH indicates there is no need of flushing the TLB in the entire operation, and
4158 	 * flush_range indicates the caller requests deferral of the TLB flushing. Fundemantally, the two
4159 	 * semantics conflict with each other, so assert they are not both true.
4160 	 */
4161 	assert(!(flush_range && (options & PMAP_OPTIONS_NOFLUSH)));
4162 
4163 	/* Only work with managed pages. */
4164 	if (!pa_valid(phys)) {
4165 		return;
4166 	}
4167 
4168 	/*
4169 	 * Determine the new protection.
4170 	 */
4171 	switch (prot) {
4172 	case VM_PROT_ALL:
4173 		return;         /* nothing to do */
4174 	case VM_PROT_READ:
4175 	case VM_PROT_READ | VM_PROT_EXECUTE:
4176 		break;
4177 	default:
4178 		/* PPL security model requires that we flush TLBs before we exit if the page may be recycled. */
4179 		options = options & ~PMAP_OPTIONS_NOFLUSH;
4180 		remove = true;
4181 		break;
4182 	}
4183 
4184 	/**
4185 	 * We don't support cross-page batching (indicated by flush_range being non-NULL) for removals,
4186 	 * as removals must use the SPTM prev_ptes array for accounting, which isn't supported for cross-
4187 	 * page batches.
4188 	 */
4189 	assert((flush_range == NULL) || !remove);
4190 
4191 	unsigned int pai = pa_index(phys);
4192 	if (__probable(locked_pvh == NULL)) {
4193 		if (flush_range != NULL) {
4194 			/**
4195 			 * If we're partway through processing a multi-page batched call,
4196 			 * preemption will already be disabled so we can't simply call
4197 			 * pvh_lock() which may block.  Instead, we first try to acquire
4198 			 * the lock without waiting, which in most cases should succeed.
4199 			 * If it fails, we submit the pending batched operations to re-
4200 			 * enable preemption and then acquire the lock normally.
4201 			 */
4202 			local_locked_pvh = pvh_try_lock(pai);
4203 			if (__improbable(!pvh_try_lock_success(&local_locked_pvh))) {
4204 				pmap_multipage_op_submit(flush_range);
4205 				local_locked_pvh = pvh_lock(pai);
4206 			}
4207 		} else {
4208 			local_locked_pvh = pvh_lock(pai);
4209 		}
4210 	} else {
4211 		local_locked_pvh = *locked_pvh;
4212 		assert(pai == local_locked_pvh.pai);
4213 	}
4214 	assert(local_locked_pvh.pvh != 0);
4215 	pvh_assert_locked(pai);
4216 
4217 	bool pvh_lock_sleep_mode_needed = false;
4218 
4219 	/*
4220 	 * PVH should be locked before accessing per-CPU data, as we're relying on the lock
4221 	 * to disable preemption.
4222 	 */
4223 	pmap_cpu_data_t *pmap_cpu_data = NULL;
4224 	pmap_sptm_percpu_data_t *sptm_pcpu = NULL;
4225 	sptm_disjoint_op_t *sptm_ops = NULL;
4226 	pt_desc_t **sptm_ptds = NULL;
4227 	ptd_info_t **sptm_ptd_info = NULL;
4228 
4229 	/* BEGIN IGNORE CODESTYLE */
4230 
4231 	/**
4232 	 * This would also work as a block, with the above variables declared using the
4233 	 * __block qualifier, but the extra runtime overhead of block syntax (e.g.
4234 	 * dereferencing __block variables through stack forwarding pointers) isn't needed
4235 	 * here, as we never need to use this code sequence as a closure.
4236 	 */
4237 	#define PPO_PERCPU_INIT() do { \
4238 	        disable_preemption(); \
4239 	        pmap_cpu_data = pmap_get_cpu_data(); \
4240 	        sptm_pcpu = PERCPU_GET(pmap_sptm_percpu); \
4241 	        sptm_ops = sptm_pcpu->sptm_ops; \
4242 	        sptm_ptds = sptm_pcpu->sptm_ptds; \
4243 	        sptm_ptd_info = sptm_pcpu->sptm_ptd_info; \
4244 	        if (remove) { \
4245 	                os_atomic_store(&pmap_cpu_data->inflight_disconnect, true, relaxed); \
4246 			/* \
4247 			 * Ensure the store to inflight_disconnect will be observed before any of the
4248 			 * ensuing PTE/refcount stores in this function.  This flag is used to avoid
4249 			 * a race in which the VM may clear a pmap's mappings and destroy the pmap on
4250 			 * another CPU, in between this function's clearing a PTE and dropping the
4251 			 * corresponding pagetable refcount.  That can lead to a panic if the
4252 			 * destroying thread observes a non-zero refcount.  For this we need a store-
4253 			 * store barrier; a store-release operation would not be sufficient.
4254 			 */ \
4255 	                os_atomic_thread_fence(release); \
4256 	        } \
4257 	} while (0)
4258 
4259 	/* END IGNORE CODESTYLE */
4260 
4261 
4262 	PPO_PERCPU_INIT();
4263 
4264 	pv_entry_t **pve_pp = NULL;
4265 
4266 	if (pvh_test_type(local_locked_pvh.pvh, PVH_TYPE_PTEP)) {
4267 		pte_p = pvh_ptep(local_locked_pvh.pvh);
4268 	} else if (pvh_test_type(local_locked_pvh.pvh, PVH_TYPE_PVEP)) {
4269 		pve_p = pvh_pve_list(local_locked_pvh.pvh);
4270 		pveh_p = pve_p;
4271 	} else if (__improbable(!pvh_test_type(local_locked_pvh.pvh, PVH_TYPE_NULL))) {
4272 		panic("%s: invalid PV head 0x%llx for PA 0x%llx", __func__, (uint64_t)local_locked_pvh.pvh, (uint64_t)phys);
4273 	}
4274 
4275 	int pve_ptep_idx = 0;
4276 	const bool compress = (options & PMAP_OPTIONS_COMPRESSOR);
4277 
4278 	/*
4279 	 * We need to keep track of whether a particular PVE list contains IOMMU
4280 	 * mappings when removing entries, because we should only remove CPU
4281 	 * mappings. If a PVE list contains at least one IOMMU mapping, we keep
4282 	 * it around.
4283 	 */
4284 	bool iommu_mapping_in_pve = false;
4285 
4286 	/**
4287 	 * With regard to TLBI, there are three cases:
4288 	 *
4289 	 * 1. PMAP_OPTIONS_NOFLUSH is specified. In such case, SPTM doesn't need to flush TLB and neither does pmap.
4290 	 * 2. PMAP_OPTIONS_NOFLUSH is not specified, but flush_range is, indicating the caller intends to flush TLB
4291 	 *    itself (with range TLBI). In such case, we check the flush_range limits and only issue the TLBI if a
4292 	 *    mapping is out of the range.
4293 	 * 3. Neither PMAP_OPTIONS_NOFLUSH nor a valid flush_range pointer is specified. In such case, we should just
4294 	 *    let SPTM handle TLBI flushing.
4295 	 */
4296 	const bool defer_tlbi = (options & PMAP_OPTIONS_NOFLUSH) || flush_range;
4297 	const uint32_t sptm_update_options = SPTM_UPDATE_PERMS_AND_WAS_WRITABLE | (defer_tlbi ? SPTM_UPDATE_DEFER_TLBI : 0);
4298 
4299 	while ((pve_p != PV_ENTRY_NULL) || (pte_p != PT_ENTRY_NULL)) {
4300 		if (__improbable(pvh_lock_sleep_mode_needed)) {
4301 			assert((num_mappings == 0) && (num_skipped_mappings == 0));
4302 			if (remove) {
4303 				/**
4304 				 * Clear the in-flight disconnect indicator for the current CPU, as we've
4305 				 * already submitted any prior pending SPTM operations, and we're about to
4306 				 * briefly re-enable preemption which may cause this thread to be migrated.
4307 				 */
4308 				os_atomic_store(&pmap_cpu_data->inflight_disconnect, false, release);
4309 			}
4310 			/**
4311 			 * Undo the explicit preemption disable done in the last call to PPO_PER_CPU_INIT().
4312 			 * If the PVH lock is placed in sleep mode, we can't rely on it to disable preemption,
4313 			 * so we need these explicit preemption twiddles to ensure we don't get migrated off-
4314 			 * core while processing SPTM per-CPU data.  At the same time, we also want preemption
4315 			 * to briefly be re-enabled every SPTM_MAPPING_LIMIT mappings so that any pending
4316 			 * urgent ASTs can be handled.
4317 			 */
4318 			enable_preemption();
4319 			pvh_lock_enter_sleep_mode(&local_locked_pvh);
4320 			pvh_lock_sleep_mode_needed = false;
4321 			PPO_PERCPU_INIT();
4322 		}
4323 
4324 		if (pve_p != PV_ENTRY_NULL) {
4325 			pte_p = pve_get_ptep(pve_p, pve_ptep_idx);
4326 			if (pte_p == PT_ENTRY_NULL) {
4327 				goto protect_skip_pve;
4328 			}
4329 		}
4330 
4331 #ifdef PVH_FLAG_IOMMU
4332 		if (pvh_ptep_is_iommu(pte_p)) {
4333 			iommu_mapping_in_pve = true;
4334 			if (__improbable(remove && (options & PMAP_OPTIONS_COMPRESSOR))) {
4335 				const iommu_instance_t iommu = ptep_get_iommu(pte_p);
4336 				panic("%s: attempt to compress ppnum 0x%x owned by iommu driver "
4337 				    "%u (token: %#x), pve_p=%p", __func__, ppnum, GET_IOMMU_ID(iommu),
4338 				    GET_IOMMU_TOKEN(iommu), pve_p);
4339 			}
4340 			if (remove && (pve_p == PV_ENTRY_NULL)) {
4341 				/*
4342 				 * We've found an IOMMU entry and it's the only entry in the PV list.
4343 				 * We don't discard IOMMU entries, so simply set up the new PV list to
4344 				 * contain the single IOMMU PTE and exit the loop.
4345 				 */
4346 				new_pte_p = pte_p;
4347 				break;
4348 			}
4349 			++num_skipped_mappings;
4350 			goto protect_skip_pve;
4351 		}
4352 #endif
4353 
4354 		const pt_entry_t spte = os_atomic_load(pte_p, relaxed);
4355 
4356 		if (__improbable(!remove && ((spte & ARM_PTE_TYPE_MASK) == ARM_PTE_TYPE_FAULT))) {
4357 			++num_skipped_mappings;
4358 			goto protect_skip_pve;
4359 		}
4360 
4361 		pt_desc_t *ptdp = NULL;
4362 		pmap_t pmap = NULL;
4363 		vm_map_address_t va = 0;
4364 
4365 		if ((flush_range != NULL) && (pte_p == flush_range->current_ptep)) {
4366 			/**
4367 			 * If the current mapping matches the flush range's current iteration position,
4368 			 * there's no need to do the work of getting the PTD.  We already know the pmap,
4369 			 * and the VA is implied by flush_range->pending_region_start.
4370 			 */
4371 			pmap = flush_range->ptfr_pmap;
4372 		} else {
4373 			ptdp = ptep_get_ptd(pte_p);
4374 			pmap = ptdp->pmap;
4375 			va = ptd_get_va(ptdp, pte_p);
4376 		}
4377 
4378 		/**
4379 		 * If the PTD is NULL, we're adding the current mapping to the pending region templates instead of the
4380 		 * pending disjoint ops, so we don't need to do flush range disjoint op management.
4381 		 */
4382 		if ((flush_range != NULL) && (ptdp != NULL)) {
4383 			/**
4384 			 * Insert a "header" entry for this physical page into the SPTM disjoint ops array.
4385 			 * We do this in three cases:
4386 			 * 1) We're at the beginning of the SPTM ops array (num_mappings == 0, flush_range->pending_disjoint_entries == 0).
4387 			 * 2) We may not be at the beginning of the SPTM ops array, but we are about to add the first operation
4388 			 *    for this physical page (num_mappings == 0, flush_range->pending_disjoint_entries == ?).
4389 			 * 3) We need to change the options passed to the SPTM for a run of one or more mappings.  Specifically,
4390 			 *    if we encounter a run of mappings that reside outside the VA region of our flush_range, or that
4391 			 *    belong to a pmap other than the one targeted by our flush_range, we should ask the SPTM to flush
4392 			 *    the TLB for us (i.e., clear SPTM_UPDATE_DEFER_TLBI), but only for those specific mappings.
4393 			 */
4394 			uint32_t per_mapping_sptm_update_options = sptm_update_options;
4395 			if ((flush_range->ptfr_pmap != pmap) || (va >= flush_range->ptfr_end) || (va < flush_range->ptfr_start)) {
4396 				per_mapping_sptm_update_options &= ~SPTM_UPDATE_DEFER_TLBI;
4397 			}
4398 			if ((num_mappings == 0) ||
4399 			    (flush_range->current_header->per_paddr_header.options != per_mapping_sptm_update_options)) {
4400 				if (pmap_multipage_op_add_page(phys, &num_mappings, per_mapping_sptm_update_options, flush_range)) {
4401 					/**
4402 					 * If we needed to submit the pending disjoint ops to make room for the new page,
4403 					 * flush any pending region ops to reenable preemption and restart the loop with
4404 					 * the lock in sleep mode.  This prevents preemption from being held disabled
4405 					 * for an arbitrary amount of time in the pathological case in which we have
4406 					 * both pending region ops and an excessively long PV list that repeatedly
4407 					 * requires new page headers with SPTM_MAPPING_LIMIT - 1 entries already pending.
4408 					 */
4409 					pmap_multipage_op_submit_region(flush_range);
4410 					assert(num_mappings == 0);
4411 					num_skipped_mappings = 0;
4412 					pvh_lock_sleep_mode_needed = true;
4413 					continue;
4414 				}
4415 			}
4416 		}
4417 
4418 		if (__improbable((pmap == NULL) ||
4419 		    (((spte & ARM_PTE_TYPE_MASK) != ARM_PTE_TYPE_FAULT) && (atop(pte_to_pa(spte)) != ppnum)))) {
4420 #if MACH_ASSERT
4421 			if ((pmap != NULL) && (pve_p != PV_ENTRY_NULL) && (kern_feature_override(KF_PMAPV_OVRD) == FALSE)) {
4422 				/* Temporarily set PTEP to NULL so that the logic below doesn't pick it up as duplicate. */
4423 				pt_entry_t *temp_ptep = pve_get_ptep(pve_p, pve_ptep_idx);
4424 				pve_set_ptep(pve_p, pve_ptep_idx, PT_ENTRY_NULL);
4425 
4426 				pv_entry_t *check_pvep = pve_p;
4427 
4428 				do {
4429 					if (pve_find_ptep_index(check_pvep, pte_p) != -1) {
4430 						panic_plain("%s: duplicate pve entry ptep=%p pmap=%p, pvh=%p, "
4431 						    "pvep=%p, pai=0x%x", __func__, pte_p, pmap, (void*)local_locked_pvh.pvh, pve_p, pai);
4432 					}
4433 				} while ((check_pvep = pve_next(check_pvep)) != PV_ENTRY_NULL);
4434 
4435 				/* Restore previous PTEP value. */
4436 				pve_set_ptep(pve_p, pve_ptep_idx, temp_ptep);
4437 			}
4438 #endif
4439 			panic("%s: bad PVE pte_p=%p pmap=%p prot=%d options=%u, pvh=%p, pveh_p=%p, pve_p=%p, pte=0x%llx, va=0x%llx ppnum: 0x%x",
4440 			    __func__, pte_p, pmap, prot, options, (void*)local_locked_pvh.pvh, pveh_p, pve_p, (uint64_t)*pte_p, (uint64_t)va, ppnum);
4441 		}
4442 
4443 		pt_entry_t pte_template = ARM_PTE_EMPTY;
4444 
4445 		if (ptdp != NULL) {
4446 			sptm_ops[num_mappings].root_pt_paddr = pmap->ttep;
4447 			sptm_ops[num_mappings].vaddr = va;
4448 		}
4449 
4450 		/* Remove the mapping if new protection is NONE */
4451 		if (remove) {
4452 			sptm_ptds[num_mappings] = ptdp;
4453 			sptm_ptd_info[num_mappings] = ptd_get_info(ptdp);
4454 			sptm_pcpu->sptm_acct_flags[num_mappings] = 0;
4455 			if (pmap != kernel_pmap) {
4456 				const bool is_internal = ppattr_pve_is_internal(pai, pve_p, pve_ptep_idx);
4457 				const bool is_altacct = ppattr_pve_is_altacct(pai, pve_p, pve_ptep_idx);
4458 
4459 				if (is_internal) {
4460 					sptm_pcpu->sptm_acct_flags[num_mappings] |= PMAP_SPTM_FLAG_INTERNAL;
4461 					ppattr_pve_clr_internal(pai, pve_p, pve_ptep_idx);
4462 				}
4463 				if (is_altacct) {
4464 					sptm_pcpu->sptm_acct_flags[num_mappings] |= PMAP_SPTM_FLAG_ALTACCT;
4465 					ppattr_pve_clr_altacct(pai, pve_p, pve_ptep_idx);
4466 				}
4467 				if (compress && is_internal) {
4468 					pte_template = ARM_PTE_COMPRESSED;
4469 					if (is_altacct) {
4470 						pte_template |= ARM_PTE_COMPRESSED_ALT;
4471 					}
4472 				}
4473 			}
4474 			/* Remove this CPU mapping from PVE list. */
4475 			if (pve_p != PV_ENTRY_NULL) {
4476 				pve_set_ptep(pve_p, pve_ptep_idx, PT_ENTRY_NULL);
4477 			}
4478 		} else {
4479 			const pt_attr_t *const pt_attr = pmap_get_pt_attr(pmap);
4480 
4481 			if (pmap == kernel_pmap) {
4482 				pte_template = ((spte & ~ARM_PTE_APMASK) | ARM_PTE_AP(AP_RONA));
4483 			} else {
4484 				pte_template = ((spte & ~ARM_PTE_APMASK) | pt_attr_leaf_ro(pt_attr));
4485 			}
4486 
4487 			/*
4488 			 * We must at least clear the 'was writeable' flag, as we're at least revoking write access,
4489 			 * meaning that the VM is effectively requesting that subsequent write accesses to these mappings
4490 			 * go through vm_fault() instead of being handled by arm_fast_fault().
4491 			 */
4492 			pte_set_was_writeable(pte_template, false);
4493 
4494 			/*
4495 			 * While the naive implementation of this would serve to add execute
4496 			 * permission, this is not how the VM uses this interface, or how
4497 			 * x86_64 implements it.  So ignore requests to add execute permissions.
4498 			 */
4499 #if DEVELOPMENT || DEBUG
4500 			if ((!(prot & VM_PROT_EXECUTE) && nx_enabled && pmap->nx_enabled) ||
4501 			    (pte_to_xprr_perm(spte) == XPRR_USER_TPRO_PERM))
4502 #else
4503 			if (!(prot & VM_PROT_EXECUTE) ||
4504 			    (pte_to_xprr_perm(spte) == XPRR_USER_TPRO_PERM))
4505 #endif
4506 			{
4507 				pte_template |= pt_attr_leaf_xn(pt_attr);
4508 			}
4509 		}
4510 
4511 		if (ptdp != NULL) {
4512 			sptm_ops[num_mappings].pte_template = pte_template;
4513 			++num_mappings;
4514 		} else if (pmap_insert_flush_range_template(pte_template, flush_range)) {
4515 			/**
4516 			 * We submit both the pending disjoint and pending region ops whenever
4517 			 * either category reaches the mapping limit.  Having pending operations
4518 			 * in either category will keep preemption disabled, and we want to ensure
4519 			 * that we can at least temporarily re-enable preemption roughly every
4520 			 * SPTM_MAPPING_LIMIT mappings.
4521 			 */
4522 			pmap_multipage_op_submit_disjoint(num_mappings, flush_range);
4523 			pvh_lock_sleep_mode_needed = true;
4524 			num_mappings = num_skipped_mappings = 0;
4525 		}
4526 
4527 protect_skip_pve:
4528 		if ((num_mappings + num_skipped_mappings) >= SPTM_MAPPING_LIMIT) {
4529 			if (flush_range != NULL) {
4530 				/* See comment above for why we submit both disjoint and region ops when we hit the limit. */
4531 				pmap_multipage_op_submit_disjoint(num_mappings, flush_range);
4532 				pmap_multipage_op_submit_region(flush_range);
4533 			} else if (num_mappings > 0) {
4534 				if (remove) {
4535 					pmap_disjoint_unmap(phys, num_mappings);
4536 				} else {
4537 					sptm_update_disjoint(phys, sptm_pcpu->sptm_ops_pa, num_mappings, sptm_update_options);
4538 				}
4539 			}
4540 			pvh_lock_sleep_mode_needed = true;
4541 			num_mappings = num_skipped_mappings = 0;
4542 		}
4543 		pte_p = PT_ENTRY_NULL;
4544 		if ((pve_p != PV_ENTRY_NULL) && (++pve_ptep_idx == PTE_PER_PVE)) {
4545 			pve_ptep_idx = 0;
4546 
4547 			if (remove) {
4548 				/**
4549 				 * If there are any IOMMU mappings in the PVE list, preserve
4550 				 * those mappings in a new PVE list (new_pve_p) which will later
4551 				 * become the new PVH entry. Keep track of the CPU mappings in
4552 				 * pveh_p/pvet_p so they can be deallocated later.
4553 				 */
4554 				if (iommu_mapping_in_pve) {
4555 					iommu_mapping_in_pve = false;
4556 					pv_entry_t *temp_pve_p = pve_next(pve_p);
4557 					pve_remove(&local_locked_pvh, pve_pp, pve_p);
4558 					if (pvh_test_type(local_locked_pvh.pvh, PVH_TYPE_PVEP)) {
4559 						pveh_p = pvh_pve_list(local_locked_pvh.pvh);
4560 					} else {
4561 						assert(pvh_test_type(local_locked_pvh.pvh, PVH_TYPE_NULL));
4562 						pveh_p = PV_ENTRY_NULL;
4563 					}
4564 					pve_p->pve_next = new_pve_p;
4565 					new_pve_p = pve_p;
4566 					pve_p = temp_pve_p;
4567 					continue;
4568 				} else {
4569 					pvet_p = pve_p;
4570 					pvh_cnt++;
4571 				}
4572 			}
4573 
4574 			pve_pp = pve_next_ptr(pve_p);
4575 			pve_p = pve_next(pve_p);
4576 			iommu_mapping_in_pve = false;
4577 		}
4578 	}
4579 
4580 	if (num_mappings != 0) {
4581 		if (remove) {
4582 			pmap_disjoint_unmap(phys, num_mappings);
4583 		} else if (flush_range == NULL) {
4584 			sptm_update_disjoint(phys, sptm_pcpu->sptm_ops_pa, num_mappings, sptm_update_options);
4585 		} else {
4586 			/* Resync the pending mapping state in flush_range with our local state. */
4587 			assert(num_mappings >= flush_range->pending_disjoint_entries);
4588 			flush_range->pending_disjoint_entries = num_mappings;
4589 		}
4590 	}
4591 
4592 	if (remove) {
4593 		os_atomic_store(&pmap_cpu_data->inflight_disconnect, false, release);
4594 	}
4595 
4596 	/**
4597 	 * Undo the explicit disable_preemption() done in PPO_PERCPU_INIT().
4598 	 * Note that enable_preemption() decrements a per-thread counter, so if
4599 	 * we happen to still hold the PVH lock in spin mode then preemption won't
4600 	 * actually be re-enabled until we drop the lock (which also decrements
4601 	 * the per-thread counter.
4602 	 */
4603 	enable_preemption();
4604 
4605 	/* if we removed a bunch of entries, take care of them now */
4606 	if (remove) {
4607 		/**
4608 		 * If we (or our caller as indicated by PMAP_OPTIONS_PPO_PENDING_RETYPE) will
4609 		 * be retyping the page, we need to drain the epochs to ensure that concurrent
4610 		 * calls to batched operations such as pmap_remove() and the various multipage
4611 		 * attribute update functions have finished consuming mappings of this page.
4612 		 */
4613 		const bool needs_retyping = pmap_prepare_unmapped_page_for_retype(phys);
4614 		if ((options & PMAP_OPTIONS_PPO_PENDING_RETYPE) && !needs_retyping) {
4615 			/**
4616 			 * pmap_prepare_unmapped_page_for_retype() will only return true if
4617 			 * the page belongs to a certain set of types that need to be auto-
4618 			 * retyped back to XNU_DEFAULT when they are unmapped.  But if the
4619 			 * caller indicated that it's going to retype the page, we need
4620 			 * to drain the epochs regardless of the current page type.
4621 			 */
4622 			pmap_retype_epoch_prepare_drain();
4623 		}
4624 		if (new_pve_p != PV_ENTRY_NULL) {
4625 			pvh_update_head(&local_locked_pvh, new_pve_p, PVH_TYPE_PVEP);
4626 		} else if (new_pte_p != PT_ENTRY_NULL) {
4627 			pvh_update_head(&local_locked_pvh, new_pte_p, PVH_TYPE_PTEP);
4628 		} else {
4629 			pvh_set_flags(&local_locked_pvh, 0);
4630 			pvh_update_head(&local_locked_pvh, PV_ENTRY_NULL, PVH_TYPE_NULL);
4631 		}
4632 
4633 		/* If removing the last mapping to a specially-protected page, retype the page back to XNU_DEFAULT. */
4634 		const bool retype_needed = pmap_retype_unmapped_page(phys);
4635 		if ((options & PMAP_OPTIONS_PPO_PENDING_RETYPE) && !retype_needed) {
4636 			pmap_retype_epoch_drain();
4637 		}
4638 	}
4639 
4640 	if (__probable(locked_pvh == NULL)) {
4641 		pvh_unlock(&local_locked_pvh);
4642 	} else {
4643 		*locked_pvh = local_locked_pvh;
4644 	}
4645 
4646 	if (remove && (pvet_p != PV_ENTRY_NULL)) {
4647 		assert(pveh_p != PV_ENTRY_NULL);
4648 		pv_list_free(pveh_p, pvet_p, pvh_cnt);
4649 	}
4650 
4651 	if ((flush_range != NULL) && !preemption_enabled()) {
4652 		flush_range->processed_entries += num_skipped_mappings;
4653 	}
4654 }
4655 
4656 MARK_AS_PMAP_TEXT void
4657 pmap_page_protect_options_internal(
4658 	ppnum_t ppnum,
4659 	vm_prot_t prot,
4660 	unsigned int options,
4661 	void *arg)
4662 {
4663 	if (arg != NULL) {
4664 		/*
4665 		 * This is a legacy argument from pre-ARM era that the VM layer passes in to hint that it will call
4666 		 * pmap_flush() later to flush the TLB. On ARM platforms, however, pmap_flush() is not implemented,
4667 		 * as it's typically more efficient to perform the TLB flushing inline with the page table updates
4668 		 * themselves. Therefore, if the argument is non-NULL, pmap will take care of TLB flushing itself
4669 		 * by clearing PMAP_OPTIONS_NOFLUSH.
4670 		 */
4671 		options &= ~PMAP_OPTIONS_NOFLUSH;
4672 	}
4673 	pmap_page_protect_options_with_flush_range(ppnum, prot, options, NULL, NULL);
4674 }
4675 
4676 void
4677 pmap_page_protect_options(
4678 	ppnum_t ppnum,
4679 	vm_prot_t prot,
4680 	unsigned int options,
4681 	void *arg)
4682 {
4683 	pmap_paddr_t    phys = ptoa(ppnum);
4684 
4685 	assert(ppnum != vm_page_fictitious_addr);
4686 
4687 	/* Only work with managed pages. */
4688 	if (!pa_valid(phys)) {
4689 		return;
4690 	}
4691 
4692 	/*
4693 	 * Determine the new protection.
4694 	 */
4695 	if (prot == VM_PROT_ALL) {
4696 		return;         /* nothing to do */
4697 	}
4698 
4699 	PMAP_TRACE(2, PMAP_CODE(PMAP__PAGE_PROTECT) | DBG_FUNC_START, ppnum, prot);
4700 
4701 	pmap_page_protect_options_internal(ppnum, prot, options, arg);
4702 
4703 	PMAP_TRACE(2, PMAP_CODE(PMAP__PAGE_PROTECT) | DBG_FUNC_END);
4704 }
4705 
4706 
4707 #if __has_feature(ptrauth_calls) && (defined(XNU_TARGET_OS_OSX) || (DEVELOPMENT || DEBUG))
4708 MARK_AS_PMAP_TEXT void
4709 pmap_disable_user_jop_internal(pmap_t pmap)
4710 {
4711 	if (pmap == kernel_pmap) {
4712 		panic("%s: called with kernel_pmap", __func__);
4713 	}
4714 	validate_pmap_mutable(pmap);
4715 	sptm_configure_root(pmap->ttep, 0, SPTM_ROOT_PT_FLAG_JOP);
4716 	pmap->disable_jop = true;
4717 }
4718 
4719 void
4720 pmap_disable_user_jop(pmap_t pmap)
4721 {
4722 	pmap_disable_user_jop_internal(pmap);
4723 }
4724 #endif /* __has_feature(ptrauth_calls) && (defined(XNU_TARGET_OS_OSX) || (DEVELOPMENT || DEBUG)) */
4725 
4726 /*
4727  * Indicates if the pmap layer enforces some additional restrictions on the
4728  * given set of protections.
4729  */
4730 bool
4731 pmap_has_prot_policy(__unused pmap_t pmap, __unused bool translated_allow_execute, __unused vm_prot_t prot)
4732 {
4733 	return false;
4734 }
4735 
4736 /*
4737  *	Set the physical protection on the
4738  *	specified range of this map as requested.
4739  *	VERY IMPORTANT: Will not increase permissions.
4740  *	VERY IMPORTANT: Only pmap_enter() is allowed to grant permissions.
4741  */
4742 void
4743 pmap_protect(
4744 	pmap_t pmap,
4745 	vm_map_address_t b,
4746 	vm_map_address_t e,
4747 	vm_prot_t prot)
4748 {
4749 	pmap_protect_options(pmap, b, e, prot, 0, NULL);
4750 }
4751 
4752 static bool
4753 pmap_protect_strong_sync(unsigned int num_mappings __unused)
4754 {
4755 	return false;
4756 }
4757 
4758 MARK_AS_PMAP_TEXT vm_map_address_t
4759 pmap_protect_options_internal(
4760 	pmap_t pmap,
4761 	vm_map_address_t start,
4762 	vm_map_address_t end,
4763 	vm_prot_t prot,
4764 	unsigned int options,
4765 	__unused void *args)
4766 {
4767 	pt_entry_t       *pte_p;
4768 	bool             set_NX = true;
4769 	bool             set_XO = false;
4770 	bool             should_have_removed = false;
4771 	bool             need_strong_sync = false;
4772 
4773 	/* Validate the pmap input before accessing its data. */
4774 	validate_pmap_mutable(pmap);
4775 
4776 	const pt_attr_t *const pt_attr = pmap_get_pt_attr(pmap);
4777 
4778 	if (__improbable((end < start) || (end > ((start + pt_attr_twig_size(pt_attr)) & ~pt_attr_twig_offmask(pt_attr))))) {
4779 		panic("%s: invalid address range %p, %p", __func__, (void*)start, (void*)end);
4780 	}
4781 
4782 #if DEVELOPMENT || DEBUG
4783 	if (options & PMAP_OPTIONS_PROTECT_IMMEDIATE) {
4784 		if ((prot & VM_PROT_ALL) == VM_PROT_NONE) {
4785 			should_have_removed = true;
4786 		}
4787 	} else
4788 #endif
4789 	{
4790 		/* Determine the new protection. */
4791 		switch (prot) {
4792 		case VM_PROT_EXECUTE:
4793 			set_XO = true;
4794 			OS_FALLTHROUGH;
4795 		case VM_PROT_READ:
4796 		case VM_PROT_READ | VM_PROT_EXECUTE:
4797 			break;
4798 		case VM_PROT_READ | VM_PROT_WRITE:
4799 		case VM_PROT_ALL:
4800 			return end;         /* nothing to do */
4801 		default:
4802 			should_have_removed = true;
4803 		}
4804 	}
4805 
4806 	if (__improbable(should_have_removed)) {
4807 		panic("%s: should have been a remove operation, "
4808 		    "pmap=%p, start=%p, end=%p, prot=%#x, options=%#x, args=%p",
4809 		    __FUNCTION__,
4810 		    pmap, (void *)start, (void *)end, prot, options, args);
4811 	}
4812 
4813 #if DEVELOPMENT || DEBUG
4814 	bool force_write = false;
4815 	if ((options & PMAP_OPTIONS_PROTECT_IMMEDIATE) && (prot & VM_PROT_WRITE)) {
4816 		force_write = true;
4817 	}
4818 	if ((prot & VM_PROT_EXECUTE) || !nx_enabled || !pmap->nx_enabled)
4819 #else
4820 	if ((prot & VM_PROT_EXECUTE))
4821 #endif
4822 	{
4823 		set_NX = false;
4824 	} else {
4825 		set_NX = true;
4826 	}
4827 
4828 	const uint64_t pmap_page_size = PAGE_RATIO * pt_attr_page_size(pt_attr);
4829 	vm_map_address_t va = start;
4830 	vm_map_address_t sptm_start_va = start;
4831 	unsigned int num_mappings = 0;
4832 
4833 	pmap_lock(pmap, PMAP_LOCK_SHARED);
4834 
4835 	pte_p = pmap_pte(pmap, start);
4836 
4837 	if (pte_p == NULL) {
4838 		pmap_unlock(pmap, PMAP_LOCK_SHARED);
4839 		return end;
4840 	}
4841 
4842 	pmap_sptm_percpu_data_t *sptm_pcpu = NULL;
4843 #if DEVELOPMENT || DEBUG
4844 	if (!force_write)
4845 #endif
4846 	{
4847 		disable_preemption();
4848 		sptm_pcpu = PERCPU_GET(pmap_sptm_percpu);
4849 	}
4850 
4851 	pt_entry_t tmplate = ARM_PTE_EMPTY;
4852 
4853 	if (pmap == kernel_pmap) {
4854 #if DEVELOPMENT || DEBUG
4855 		if (force_write) {
4856 			tmplate = ARM_PTE_AP(AP_RWNA);
4857 		} else
4858 #endif
4859 		{
4860 			tmplate = ARM_PTE_AP(AP_RONA);
4861 		}
4862 	} else {
4863 #if DEVELOPMENT || DEBUG
4864 		if (force_write) {
4865 			assert(pmap->type != PMAP_TYPE_NESTED);
4866 			tmplate = pt_attr_leaf_rw(pt_attr);
4867 		} else
4868 #endif
4869 		if (set_XO) {
4870 			tmplate = pt_attr_leaf_rona(pt_attr);
4871 		} else {
4872 			tmplate = pt_attr_leaf_ro(pt_attr);
4873 		}
4874 	}
4875 
4876 	if (set_NX) {
4877 		tmplate |= pt_attr_leaf_xn(pt_attr);
4878 	}
4879 
4880 	while (va < end) {
4881 		pt_entry_t spte = ARM_PTE_EMPTY;
4882 
4883 		/**
4884 		 * Removing "NX" would grant "execute" access immediately, bypassing any
4885 		 * checks VM might want to do in its soft fault path.
4886 		 * pmap_protect() and co. are not allowed to increase access permissions,
4887 		 * except in the PMAP_PROTECT_OPTIONS_IMMEDIATE internal-only case.
4888 		 * Therefore, if we are not explicitly clearing execute permissions, inherit
4889 		 * the existing permissions.
4890 		 */
4891 		if (!set_NX) {
4892 			spte = os_atomic_load(pte_p, relaxed);
4893 			if (__improbable((spte & ARM_PTE_TYPE_MASK) == ARM_PTE_TYPE_FAULT)) {
4894 				tmplate |= pt_attr_leaf_xn(pt_attr);
4895 			} else {
4896 				tmplate |= (spte & ARM_PTE_XMASK);
4897 			}
4898 		}
4899 
4900 #if DEVELOPMENT || DEBUG
4901 		/*
4902 		 * PMAP_OPTIONS_PROTECT_IMMEDIATE is an internal-only option that's intended to
4903 		 * provide a "backdoor" to allow normally write-protected compressor pages to be
4904 		 * be temporarily written without triggering expensive write faults.
4905 		 * SPTM TODO: Given the intended use of this flag, we may be able to relax some
4906 		 * of our assumptions below when it comes to ref/mod accounting, and we may be
4907 		 * able to avoid holding the PVH lock across the SPTM mapping operation and the
4908 		 * ref/mod updates.  This will be important if we move to a batched SPTM mapping
4909 		 * API.
4910 		 */
4911 		if (force_write) {
4912 			if (spte == ARM_PTE_EMPTY) {
4913 				spte = os_atomic_load(pte_p, relaxed);
4914 			}
4915 
4916 			/* A concurrent remove or disconnect may have cleared the PTE. */
4917 			if (__improbable((spte & ARM_PTE_TYPE_MASK) == ARM_PTE_TYPE_FAULT)) {
4918 				goto pmap_protect_insert_mapping;
4919 			}
4920 
4921 			/* Inherit permissions and "was_writeable" from the template. */
4922 			spte = (spte & ~(ARM_PTE_APMASK | ARM_PTE_XMASK | ARM_PTE_WRITEABLE)) |
4923 			    (tmplate & (ARM_PTE_APMASK | ARM_PTE_XMASK | ARM_PTE_WRITEABLE));
4924 
4925 			/* Access flag should be set for any immediate change in protections */
4926 			spte |= ARM_PTE_AF;
4927 			const pmap_paddr_t pa = pte_to_pa(spte);
4928 			const unsigned int pai = pa_index(pa);
4929 			locked_pvh_t locked_pvh;
4930 			if (pa_valid(pa)) {
4931 				locked_pvh = pvh_lock(pai);
4932 				ppattr_modify_bits(pai, PP_ATTR_REFFAULT | PP_ATTR_MODFAULT,
4933 				    PP_ATTR_REFERENCED | PP_ATTR_MODIFIED);
4934 			}
4935 
4936 			__assert_only const sptm_return_t sptm_status = sptm_map_page(pmap->ttep, va, spte);
4937 
4938 			/*
4939 			 * We don't expect the VM to be concurrently removing these compressor mappings.
4940 			 * If it does for some reason, we can check for SPTM_MAP_FLUSH_PENDING and continue
4941 			 * the main loop.
4942 			 */
4943 			assert((sptm_status == SPTM_SUCCESS) || (sptm_status == SPTM_MAP_VALID));
4944 
4945 			if (pa_valid(pa)) {
4946 				pvh_unlock(&locked_pvh);
4947 			}
4948 		}
4949 
4950 pmap_protect_insert_mapping:
4951 #endif /* DEVELOPMENT || DEBUG */
4952 
4953 		va += pmap_page_size;
4954 		++pte_p;
4955 
4956 #if DEVELOPMENT || DEBUG
4957 		if (!force_write)
4958 #endif
4959 		{
4960 			sptm_pcpu->sptm_templates[num_mappings] = tmplate;
4961 			++num_mappings;
4962 			if (num_mappings == SPTM_MAPPING_LIMIT) {
4963 				/**
4964 				 * Enter the retype epoch for the batched update operation.  This is necessary because we
4965 				 * cannot reasonably hold the PVH locks for all pages mapped by the region during this
4966 				 * call, so a concurrent pmap_page_protect() operation against one of those pages may
4967 				 * race this call.  That should be perfectly fine as far as the PTE updates are concerned,
4968 				 * but if pmap_page_protect() then needs to retype the page, an SPTM violation may result
4969 				 * if it does not first drain our epoch.
4970 				 */
4971 				pmap_retype_epoch_enter();
4972 				sptm_update_region(pmap->ttep, sptm_start_va, num_mappings, sptm_pcpu->sptm_templates_pa,
4973 				    SPTM_UPDATE_PERMS_AND_WAS_WRITABLE);
4974 				pmap_retype_epoch_exit();
4975 				need_strong_sync = need_strong_sync || pmap_protect_strong_sync(num_mappings);
4976 
4977 				/* Temporarily re-enable preemption to allow any urgent ASTs to be processed. */
4978 				enable_preemption();
4979 				num_mappings = 0;
4980 				sptm_start_va = va;
4981 				disable_preemption();
4982 				sptm_pcpu = PERCPU_GET(pmap_sptm_percpu);
4983 			}
4984 		}
4985 	}
4986 
4987 	/* This won't happen in the force_write case as we should never increment num_mappings. */
4988 	if (num_mappings != 0) {
4989 		pmap_retype_epoch_enter();
4990 		sptm_update_region(pmap->ttep, sptm_start_va, num_mappings, sptm_pcpu->sptm_templates_pa,
4991 		    SPTM_UPDATE_PERMS_AND_WAS_WRITABLE);
4992 		pmap_retype_epoch_exit();
4993 		need_strong_sync = need_strong_sync || pmap_protect_strong_sync(num_mappings);
4994 	}
4995 
4996 #if DEVELOPMENT || DEBUG
4997 	if (!force_write)
4998 #endif
4999 	{
5000 		enable_preemption();
5001 	}
5002 	pmap_unlock(pmap, PMAP_LOCK_SHARED);
5003 	if (__improbable(need_strong_sync)) {
5004 		arm64_sync_tlb(true);
5005 	}
5006 	return va;
5007 }
5008 
5009 void
5010 pmap_protect_options(
5011 	pmap_t pmap,
5012 	vm_map_address_t b,
5013 	vm_map_address_t e,
5014 	vm_prot_t prot,
5015 	unsigned int options,
5016 	__unused void *args)
5017 {
5018 	vm_map_address_t l, beg;
5019 
5020 	__unused const pt_attr_t * const pt_attr = pmap_get_pt_attr(pmap);
5021 
5022 	if ((b | e) & pt_attr_leaf_offmask(pt_attr)) {
5023 		panic("pmap_protect_options() pmap %p start 0x%llx end 0x%llx",
5024 		    pmap, (uint64_t)b, (uint64_t)e);
5025 	}
5026 
5027 	/*
5028 	 * We allow single-page requests to execute non-preemptibly,
5029 	 * as it doesn't make sense to sample AST_URGENT for a single-page
5030 	 * operation, and there are a couple of special use cases that
5031 	 * require a non-preemptible single-page operation.
5032 	 */
5033 	if ((e - b) > (pt_attr_page_size(pt_attr) * PAGE_RATIO)) {
5034 		pmap_verify_preemptible();
5035 	}
5036 
5037 #if DEVELOPMENT || DEBUG
5038 	if (options & PMAP_OPTIONS_PROTECT_IMMEDIATE) {
5039 		if ((prot & VM_PROT_ALL) == VM_PROT_NONE) {
5040 			pmap_remove_options(pmap, b, e, options);
5041 			return;
5042 		}
5043 	} else
5044 #endif
5045 	{
5046 		/* Determine the new protection. */
5047 		switch (prot) {
5048 		case VM_PROT_EXECUTE:
5049 		case VM_PROT_READ:
5050 		case VM_PROT_READ | VM_PROT_EXECUTE:
5051 			break;
5052 		case VM_PROT_READ | VM_PROT_WRITE:
5053 		case VM_PROT_ALL:
5054 			return;         /* nothing to do */
5055 		default:
5056 			pmap_remove_options(pmap, b, e, options);
5057 			return;
5058 		}
5059 	}
5060 
5061 	PMAP_TRACE(2, PMAP_CODE(PMAP__PROTECT) | DBG_FUNC_START,
5062 	    VM_KERNEL_ADDRHIDE(pmap), VM_KERNEL_ADDRHIDE(b),
5063 	    VM_KERNEL_ADDRHIDE(e));
5064 
5065 	beg = b;
5066 
5067 	while (beg < e) {
5068 		l = ((beg + pt_attr_twig_size(pt_attr)) & ~pt_attr_twig_offmask(pt_attr));
5069 
5070 		if (l > e) {
5071 			l = e;
5072 		}
5073 
5074 		beg = pmap_protect_options_internal(pmap, beg, l, prot, options, args);
5075 	}
5076 
5077 	PMAP_TRACE(2, PMAP_CODE(PMAP__PROTECT) | DBG_FUNC_END);
5078 }
5079 
5080 /**
5081  * Inserts an arbitrary number of physical pages ("block") in a pmap.
5082  *
5083  * @param pmap pmap to insert the pages into.
5084  * @param va virtual address to map the pages into.
5085  * @param pa page number of the first physical page to map.
5086  * @param size block size, in number of pages.
5087  * @param prot mapping protection attributes.
5088  * @param attr flags to pass to pmap_enter().
5089  *
5090  * @return KERN_SUCCESS.
5091  */
5092 kern_return_t
5093 pmap_map_block(
5094 	pmap_t pmap,
5095 	addr64_t va,
5096 	ppnum_t pa,
5097 	uint32_t size,
5098 	vm_prot_t prot,
5099 	int attr,
5100 	unsigned int flags)
5101 {
5102 	return pmap_map_block_addr(pmap, va, ((pmap_paddr_t)pa) << PAGE_SHIFT, size, prot, attr, flags);
5103 }
5104 
5105 /**
5106  * Inserts an arbitrary number of physical pages ("block") in a pmap.
5107  * As opposed to pmap_map_block(), this function takes
5108  * a physical address as an input and operates using the
5109  * page size associated with the input pmap.
5110  *
5111  * @param pmap pmap to insert the pages into.
5112  * @param va virtual address to map the pages into.
5113  * @param pa physical address of the first physical page to map.
5114  * @param size block size, in number of pages.
5115  * @param prot mapping protection attributes.
5116  * @param attr flags to pass to pmap_enter().
5117  *
5118  * @return KERN_SUCCESS.
5119  */
5120 kern_return_t
5121 pmap_map_block_addr(
5122 	pmap_t pmap,
5123 	addr64_t va,
5124 	pmap_paddr_t pa,
5125 	uint32_t size,
5126 	vm_prot_t prot,
5127 	int attr,
5128 	unsigned int flags)
5129 {
5130 #if __ARM_MIXED_PAGE_SIZE__
5131 	const pt_attr_t * const pt_attr = pmap_get_pt_attr(pmap);
5132 	const uint64_t pmap_page_size = pt_attr_page_size(pt_attr);
5133 #else
5134 	const uint64_t pmap_page_size = PAGE_SIZE;
5135 #endif
5136 
5137 	for (ppnum_t page = 0; page < size; page++) {
5138 		if (pmap_enter_addr(pmap, va, pa, prot, VM_PROT_NONE, attr, TRUE, PMAP_MAPPING_TYPE_INFER) != KERN_SUCCESS) {
5139 			panic("%s: failed pmap_enter_addr, "
5140 			    "pmap=%p, va=%#llx, pa=%llu, size=%u, prot=%#x, flags=%#x",
5141 			    __FUNCTION__,
5142 			    pmap, va, (uint64_t)pa, size, prot, flags);
5143 		}
5144 
5145 		va += pmap_page_size;
5146 		pa += pmap_page_size;
5147 	}
5148 
5149 	return KERN_SUCCESS;
5150 }
5151 
5152 kern_return_t
5153 pmap_enter_addr(
5154 	pmap_t pmap,
5155 	vm_map_address_t v,
5156 	pmap_paddr_t pa,
5157 	vm_prot_t prot,
5158 	vm_prot_t fault_type,
5159 	unsigned int flags,
5160 	boolean_t wired,
5161 	pmap_mapping_type_t mapping_type)
5162 {
5163 	return pmap_enter_options_addr(pmap, v, pa, prot, fault_type, flags, wired, 0, NULL, mapping_type);
5164 }
5165 
5166 /*
5167  *	Insert the given physical page (p) at
5168  *	the specified virtual address (v) in the
5169  *	target physical map with the protection requested.
5170  *
5171  *	If specified, the page will be wired down, meaning
5172  *	that the related pte can not be reclaimed.
5173  *
5174  *	NB:  This is the only routine which MAY NOT lazy-evaluate
5175  *	or lose information.  That is, this routine must actually
5176  *	insert this page into the given map eventually (must make
5177  *	forward progress eventually.
5178  */
5179 kern_return_t
5180 pmap_enter(
5181 	pmap_t pmap,
5182 	vm_map_address_t v,
5183 	ppnum_t pn,
5184 	vm_prot_t prot,
5185 	vm_prot_t fault_type,
5186 	unsigned int flags,
5187 	boolean_t wired,
5188 	pmap_mapping_type_t mapping_type)
5189 {
5190 	return pmap_enter_addr(pmap, v, ((pmap_paddr_t)pn) << PAGE_SHIFT, prot, fault_type, flags, wired, mapping_type);
5191 }
5192 
5193 /*
5194  * Attempt to update a PTE constructed by pmap_enter_options().
5195  *
5196  * @note performs no page table or accounting modifications, nor any lasting SPTM page type modification, on failure.
5197  * @note expects to be called with preemption disabled to guarantee safe access to SPTM per-CPU data.
5198  *
5199  * @param pmap The pmap representing the address space in which to store the new PTE
5200  * @param pte_p The physical aperture KVA of the PTE to store
5201  * @param new_pte The new value to store in *pte_p
5202  * @param v The virtual address mapped by pte_p
5203  * @param locked_pvh Input/Output parameter pointing to a wrapped pv_head_table entry returned by
5204  *        a previous call to pvh_lock().  *locked_pvh will be updated if existing mappings
5205  *        need to be disconnected prior to retyping.
5206  * @param old_pte Returns the prior PTE contents, iff the PTE is successfully updated
5207  * @param options bitmask of PMAP_OPTIONS_* flags passed to pmap_enter_options().
5208  * @param mapping_type The type of the new mapping, this defines which SPTM frame type to use.
5209  *
5210  * @return SPTM_SUCCESS iff able to successfully update *pte_p to new_pte via sptm_map_page(),
5211  *         SPTM_MAP_VALID if an existing mapping was successfully upgraded via sptm_map_page(),
5212  *         SPTM_MAP_FLUSH_PENDING if the TLB flush of a previous mapping is still in-flight and
5213  *             the mapping operation should be retried, or if the mapping operation should be retried
5214  *             because we had to temporarily re-enable preemption which would invalidate caller-held
5215  *             per-CPU data.
5216  *         Otherwise an appropriate SPTM or TXM error code; in these cases the mapping should not be
5217  *             retried and the caller should return an error.
5218  */
5219 static inline sptm_return_t
5220 pmap_enter_pte(
5221 	pmap_t pmap,
5222 	pt_entry_t *pte_p,
5223 	pt_entry_t new_pte,
5224 	locked_pvh_t *locked_pvh,
5225 	pt_entry_t *old_pte,
5226 	vm_map_address_t v,
5227 	unsigned int options,
5228 	pmap_mapping_type_t mapping_type)
5229 {
5230 	sptm_pte_t prev_pte;
5231 	bool changed_wiring = false;
5232 
5233 	assert(pte_p != NULL);
5234 	assert(old_pte != NULL);
5235 
5236 	/* SPTM TODO: handle PAGE_RATIO_4 configurations if those devices remain supported. */
5237 
5238 	assert(get_preemption_level() > 0);
5239 	const pmap_paddr_t pa = pte_to_pa(new_pte) & ~PAGE_MASK;
5240 	sptm_frame_type_t prev_frame_type = XNU_DEFAULT;
5241 	sptm_frame_type_t new_frame_type = XNU_DEFAULT;
5242 
5243 	/*
5244 	 * If the caller specified a mapping type of PMAP_MAPPINGS_TYPE_INFER, then we
5245 	 * keep the existing logic of deriving the SPTM frame type from the XPRR permissions.
5246 	 *
5247 	 * If the caller specified another mapping type, we simply follow that. This refactor was
5248 	 * needed for the XNU_KERNEL_RESTRICTED work, and it also allows us to be more precise at
5249 	 * what we want. It's better to let the caller specify the mapping type rather than use the
5250 	 * permissions for that.
5251 	 *
5252 	 * In the future, we should move entirely to use pmap_mapping_type_t; see rdar://114886323.
5253 	 */
5254 	if (mapping_type != PMAP_MAPPING_TYPE_INFER) {
5255 		switch (mapping_type) {
5256 		case PMAP_MAPPING_TYPE_DEFAULT:
5257 			new_frame_type = (sptm_frame_type_t)mapping_type;
5258 			break;
5259 		case PMAP_MAPPING_TYPE_ROZONE:
5260 			assert(((pmap == kernel_pmap) && zone_spans_ro_va(v, v + pt_attr_page_size(pmap_get_pt_attr(pmap)))));
5261 			new_frame_type = (sptm_frame_type_t)mapping_type;
5262 			break;
5263 		case PMAP_MAPPING_TYPE_RESTRICTED:
5264 			if (use_xnu_restricted) {
5265 				new_frame_type = (sptm_frame_type_t)mapping_type;
5266 			} else {
5267 				new_frame_type = XNU_DEFAULT;
5268 			}
5269 			break;
5270 		default:
5271 			panic("invalid mapping type: %d", mapping_type);
5272 		}
5273 	} else if (__improbable(pte_to_xprr_perm(new_pte) == XPRR_USER_JIT_PERM)) {
5274 		/*
5275 		 * Always check for XPRR_USER_JIT_PERM before we check for anything else. When using
5276 		 * RWX permissions, the only allowed type is XNU_USER_JIT, regardless of any other
5277 		 * flags which the VM may have provided.
5278 		 *
5279 		 * TODO: Assert that the PMAP_OPTIONS_XNU_USER_DEBUG flag isn't set when entering
5280 		 * this case. We can't do this for now because this might trigger on some macOS
5281 		 * systems where applications use MAP_JIT with RW/RX permissions, and then later
5282 		 * switch to RWX (which will cause a switch to XNU_USER_JIT from XNU_USER_DEBUG
5283 		 * but the VM will still have PMAP_OPTIONS_XNU_USER_DEBUG set). If the VM can
5284 		 * catch this case, and remove PMAP_OPTIONS_XNU_USER_DEBUG when an application
5285 		 * switches to RWX, then we can start asserting this requirement.
5286 		 */
5287 		new_frame_type = XNU_USER_JIT;
5288 	} else if (__improbable(options & PMAP_OPTIONS_XNU_USER_DEBUG)) {
5289 		/*
5290 		 * Both XNU_USER_DEBUG and XNU_USER_EXEC allow RX permissions. Given that, we must
5291 		 * test for PMAP_OPTIONS_XNU_USER_DEBUG before we test for XNU_USER_EXEC since the
5292 		 * XNU_USER_DEBUG type overlays the XNU_USER_EXEC type.
5293 		 */
5294 		new_frame_type = XNU_USER_DEBUG;
5295 	} else if (pte_to_xprr_perm(new_pte) == XPRR_USER_RX_PERM) {
5296 		new_frame_type = XNU_USER_EXEC;
5297 	}
5298 
5299 	if (__improbable(new_frame_type != XNU_DEFAULT)) {
5300 		prev_frame_type = sptm_get_frame_type(pa);
5301 	}
5302 
5303 	if (__improbable(new_frame_type != prev_frame_type)) {
5304 		/**
5305 		 * Remove all existing mappings prior to retyping, so that we can safely retype without having to worry
5306 		 * about a concurrent operation on one of those mappings triggering an SPTM violation.  In particular,
5307 		 * pmap_remove() may clear a mapping to this page without holding its PVH lock.  This approach works
5308 		 * because we hold the PVH lock during this call, and any attempt to enter a new mapping for the page
5309 		 * will also need to grab the PVH lock and call this function.
5310 		 */
5311 		pmap_page_protect_options_with_flush_range((ppnum_t)atop(pa), VM_PROT_NONE,
5312 		    PMAP_OPTIONS_PPO_PENDING_RETYPE, locked_pvh, NULL);
5313 		/**
5314 		 * In the unlikely event that pmap_page_protect_options_with_flush_range() had to process
5315 		 * an excessively long PV list, it will have enabled preemption by placing the PVH lock
5316 		 * in sleep mode.  In this case, we may have been migrated to a different CPU, and caller
5317 		 * assumptions about the state of per-CPU data (such as per-CPU PVE availability) will no
5318 		 * longer hold true.  Ask the caller to retry by pretending we encountered a pending flush.
5319 		 */
5320 		if (__improbable(preemption_enabled())) {
5321 			return SPTM_MAP_FLUSH_PENDING;
5322 		}
5323 		sptm_retype_params_t retype_params = {.raw = SPTM_RETYPE_PARAMS_NULL};
5324 		/* Reload the existing frame type, as pmap_page_protect_options() may have changed it back to XNU_DEFAULT. */
5325 		prev_frame_type = sptm_get_frame_type(pa);
5326 		sptm_retype(pa, prev_frame_type, new_frame_type, retype_params);
5327 	}
5328 
5329 	const sptm_return_t sptm_status = sptm_map_page(pmap->ttep, v, new_pte);
5330 	if (__improbable((sptm_status != SPTM_SUCCESS) && (sptm_status != SPTM_MAP_VALID))) {
5331 		/*
5332 		 * We should always undo our previous retype, even if the SPTM returned SPTM_MAP_FLUSH_PENDING as
5333 		 * opposed to a TXM error.  In the case of SPTM_MAP_FLUSH_PENDING, pmap_enter() will drop the PVH
5334 		 * lock before turning around to retry the mapping operation.  It may then be possible for the
5335 		 * mapping state of the page to change such that our next attempt to map it will fail with a TXM
5336 		 * error, so if we were to leave the new type in place here we would then have lost our record
5337 		 * of the previous type and would effectively leave the page in an inconsistent state.
5338 		 */
5339 		if (__improbable(new_frame_type != prev_frame_type)) {
5340 			sptm_retype_params_t retype_params = {.raw = SPTM_RETYPE_PARAMS_NULL};
5341 			sptm_retype(pa, new_frame_type, prev_frame_type, retype_params);
5342 		}
5343 		return sptm_status;
5344 	}
5345 
5346 	*old_pte = prev_pte = PERCPU_GET(pmap_sptm_percpu)->sptm_prev_ptes[0];
5347 
5348 	if (prev_pte != new_pte) {
5349 		changed_wiring = pte_is_compressed(prev_pte, pte_p) ?
5350 		    (new_pte & ARM_PTE_WIRED) != 0 :
5351 		    (new_pte & ARM_PTE_WIRED) != (prev_pte & ARM_PTE_WIRED);
5352 
5353 		if ((pmap != kernel_pmap) && changed_wiring) {
5354 			pte_update_wiredcnt(pmap, pte_p, (new_pte & ARM_PTE_WIRED) != 0);
5355 		}
5356 
5357 		PMAP_TRACE(4 + pt_attr_leaf_level(pmap_get_pt_attr(pmap)), PMAP_CODE(PMAP__TTE),
5358 		    VM_KERNEL_ADDRHIDE(pmap), VM_KERNEL_ADDRHIDE(v),
5359 		    VM_KERNEL_ADDRHIDE(v + (pt_attr_page_size(pmap_get_pt_attr(pmap)) * PAGE_RATIO)), new_pte);
5360 	}
5361 
5362 	return sptm_status;
5363 }
5364 
5365 MARK_AS_PMAP_TEXT static pt_entry_t
5366 wimg_to_pte(unsigned int wimg, pmap_paddr_t pa)
5367 {
5368 	pt_entry_t pte;
5369 
5370 	switch (wimg & (VM_WIMG_MASK)) {
5371 	case VM_WIMG_IO:
5372 		// Map DRAM addresses with VM_WIMG_IO as Device-GRE instead of
5373 		// Device-nGnRnE. On H14+, accesses to them can be reordered by
5374 		// AP, while preserving the security benefits of using device
5375 		// mapping against side-channel attacks. On pre-H14 platforms,
5376 		// the accesses will still be strongly ordered.
5377 		if (is_dram_addr(pa)) {
5378 			pte = ARM_PTE_ATTRINDX(CACHE_ATTRINDX_POSTED_COMBINED_REORDERED);
5379 		} else {
5380 			pte = ARM_PTE_ATTRINDX(CACHE_ATTRINDX_DISABLE);
5381 #if HAS_FEAT_XS
5382 			pmap_io_range_t *io_rgn = pmap_find_io_attr(pa);
5383 			if (__improbable((io_rgn != NULL) && (io_rgn->wimg & PMAP_IO_RANGE_STRONG_SYNC))) {
5384 				pte = ARM_PTE_ATTRINDX(CACHE_ATTRINDX_DISABLE_XS);
5385 			}
5386 #endif /* HAS_FEAT_XS */
5387 		}
5388 		pte |= ARM_PTE_NX | ARM_PTE_PNX;
5389 		break;
5390 	case VM_WIMG_RT:
5391 		pte = ARM_PTE_ATTRINDX(CACHE_ATTRINDX_RT);
5392 		pte |= ARM_PTE_NX | ARM_PTE_PNX;
5393 		break;
5394 	case VM_WIMG_POSTED:
5395 		if (is_dram_addr(pa)) {
5396 			pte = ARM_PTE_ATTRINDX(CACHE_ATTRINDX_POSTED_COMBINED_REORDERED);
5397 		} else {
5398 			pte = ARM_PTE_ATTRINDX(CACHE_ATTRINDX_POSTED);
5399 		}
5400 		pte |= ARM_PTE_NX | ARM_PTE_PNX;
5401 		break;
5402 	case VM_WIMG_POSTED_REORDERED:
5403 		if (is_dram_addr(pa)) {
5404 			pte = ARM_PTE_ATTRINDX(CACHE_ATTRINDX_POSTED_COMBINED_REORDERED);
5405 		} else {
5406 			pte = ARM_PTE_ATTRINDX(CACHE_ATTRINDX_POSTED_REORDERED);
5407 		}
5408 		pte |= ARM_PTE_NX | ARM_PTE_PNX;
5409 		break;
5410 	case VM_WIMG_POSTED_COMBINED_REORDERED:
5411 		pte = ARM_PTE_ATTRINDX(CACHE_ATTRINDX_POSTED_COMBINED_REORDERED);
5412 #if HAS_FEAT_XS
5413 		if (!is_dram_addr(pa)) {
5414 			pte = ARM_PTE_ATTRINDX(CACHE_ATTRINDX_POSTED_COMBINED_REORDERED_XS);
5415 		}
5416 #endif /* HAS_FEAT_XS */
5417 		pte |= ARM_PTE_NX | ARM_PTE_PNX;
5418 		break;
5419 	case VM_WIMG_WCOMB:
5420 		pte = ARM_PTE_ATTRINDX(CACHE_ATTRINDX_WRITECOMB);
5421 		pte |= ARM_PTE_NX | ARM_PTE_PNX;
5422 		break;
5423 	case VM_WIMG_WTHRU:
5424 		pte = ARM_PTE_ATTRINDX(CACHE_ATTRINDX_WRITETHRU);
5425 		pte |= ARM_PTE_SH(SH_OUTER_MEMORY);
5426 		break;
5427 	case VM_WIMG_COPYBACK:
5428 		pte = ARM_PTE_ATTRINDX(CACHE_ATTRINDX_WRITEBACK);
5429 		pte |= ARM_PTE_SH(SH_OUTER_MEMORY);
5430 		break;
5431 	case VM_WIMG_INNERWBACK:
5432 		pte = ARM_PTE_ATTRINDX(CACHE_ATTRINDX_INNERWRITEBACK);
5433 		pte |= ARM_PTE_SH(SH_INNER_MEMORY);
5434 		break;
5435 	default:
5436 		pte = ARM_PTE_ATTRINDX(CACHE_ATTRINDX_DEFAULT);
5437 		pte |= ARM_PTE_SH(SH_OUTER_MEMORY);
5438 	}
5439 
5440 	return pte;
5441 }
5442 
5443 
5444 /*
5445  * Construct a PTE (and the physical page attributes) for the given virtual to
5446  * physical mapping.
5447  *
5448  * This function has no side effects and is safe to call so that it is safe to
5449  * call while attempting a pmap_enter transaction.
5450  */
5451 MARK_AS_PMAP_TEXT static pt_entry_t
5452 pmap_construct_pte(
5453 	const pmap_t pmap,
5454 	vm_map_address_t va,
5455 	pmap_paddr_t pa,
5456 	vm_prot_t prot,
5457 	vm_prot_t fault_type,
5458 	boolean_t wired,
5459 	const pt_attr_t* const pt_attr,
5460 	uint16_t *pp_attr_bits /* OUTPUT */
5461 	)
5462 {
5463 	bool set_NX = false, set_XO = false;
5464 	pt_entry_t pte = pa_to_pte(pa) | ARM_PTE_TYPE;
5465 	assert(pp_attr_bits != NULL);
5466 	*pp_attr_bits = 0;
5467 
5468 	if (wired) {
5469 		pte |= ARM_PTE_WIRED;
5470 	}
5471 
5472 #if DEVELOPMENT || DEBUG
5473 	if ((prot & VM_PROT_EXECUTE) || !nx_enabled || !pmap->nx_enabled)
5474 #else
5475 	if ((prot & VM_PROT_EXECUTE))
5476 #endif
5477 	{
5478 		set_NX = false;
5479 	} else {
5480 		set_NX = true;
5481 	}
5482 
5483 	if (prot == VM_PROT_EXECUTE) {
5484 		set_XO = true;
5485 
5486 	}
5487 
5488 	if (set_NX) {
5489 		pte |= pt_attr_leaf_xn(pt_attr);
5490 	} else {
5491 		if (pmap == kernel_pmap) {
5492 			pte |= ARM_PTE_NX;
5493 		} else {
5494 			pte |= pt_attr_leaf_x(pt_attr);
5495 		}
5496 	}
5497 
5498 	if (pmap == kernel_pmap) {
5499 #if __ARM_KERNEL_PROTECT__
5500 		pte |= ARM_PTE_NG;
5501 #endif /* __ARM_KERNEL_PROTECT__ */
5502 		if (prot & VM_PROT_WRITE) {
5503 			pte |= ARM_PTE_AP(AP_RWNA);
5504 			*pp_attr_bits |= PP_ATTR_MODIFIED | PP_ATTR_REFERENCED;
5505 		} else {
5506 			pte |= ARM_PTE_AP(AP_RONA);
5507 			*pp_attr_bits |= PP_ATTR_REFERENCED;
5508 		}
5509 	} else {
5510 		if (pmap->type != PMAP_TYPE_NESTED) {
5511 			pte |= ARM_PTE_NG;
5512 		} else if ((pmap->nested_region_unnested_table_bitmap)
5513 		    && (va >= pmap->nested_region_addr)
5514 		    && (va < (pmap->nested_region_addr + pmap->nested_region_size))) {
5515 			unsigned int index = (unsigned int)((va - pmap->nested_region_addr)  >> pt_attr_twig_shift(pt_attr));
5516 
5517 			if ((pmap->nested_region_unnested_table_bitmap)
5518 			    && bitmap_test(pmap->nested_region_unnested_table_bitmap, index)) {
5519 				pte |= ARM_PTE_NG;
5520 			}
5521 		}
5522 		if (prot & VM_PROT_WRITE) {
5523 			assert(pmap->type != PMAP_TYPE_NESTED);
5524 			if (pa_valid(pa) && (!ppattr_pa_test_bits(pa, PP_ATTR_MODIFIED))) {
5525 				if (fault_type & VM_PROT_WRITE) {
5526 					pte |= pt_attr_leaf_rw(pt_attr);
5527 					*pp_attr_bits |= PP_ATTR_REFERENCED | PP_ATTR_MODIFIED;
5528 				} else {
5529 					pte |= pt_attr_leaf_ro(pt_attr);
5530 					/*
5531 					 * Mark the page as MODFAULT so that a subsequent write
5532 					 * may be handled through arm_fast_fault().
5533 					 */
5534 					*pp_attr_bits |= PP_ATTR_REFERENCED | PP_ATTR_MODFAULT;
5535 					pte_set_was_writeable(pte, true);
5536 				}
5537 			} else {
5538 				pte |= pt_attr_leaf_rw(pt_attr);
5539 				*pp_attr_bits |= (PP_ATTR_REFERENCED | PP_ATTR_MODIFIED);
5540 			}
5541 		} else {
5542 			if (set_XO) {
5543 				pte |= pt_attr_leaf_rona(pt_attr);
5544 			} else {
5545 				pte |= pt_attr_leaf_ro(pt_attr);
5546 			}
5547 			*pp_attr_bits |= PP_ATTR_REFERENCED;
5548 		}
5549 	}
5550 
5551 	pte |= ARM_PTE_AF;
5552 	return pte;
5553 }
5554 
5555 MARK_AS_PMAP_TEXT kern_return_t
5556 pmap_enter_options_internal(
5557 	pmap_t pmap,
5558 	vm_map_address_t v,
5559 	pmap_paddr_t pa,
5560 	vm_prot_t prot,
5561 	vm_prot_t fault_type,
5562 	unsigned int flags,
5563 	boolean_t wired,
5564 	unsigned int options,
5565 	pmap_mapping_type_t mapping_type)
5566 {
5567 	ppnum_t         pn = (ppnum_t)atop(pa);
5568 	pt_entry_t      *pte_p;
5569 	unsigned int    wimg_bits;
5570 	bool            committed = false;
5571 	kern_return_t   kr = KERN_SUCCESS;
5572 	uint16_t pp_attr_bits;
5573 	volatile uint16_t *wiredcnt = NULL;
5574 	pv_free_list_t *local_pv_free;
5575 
5576 	validate_pmap_mutable(pmap);
5577 
5578 	/**
5579 	 * Prepare for the SPTM call early by prefetching the relavant FTEs. Cache misses
5580 	 * in SPTM accessing these turn out to contribute to a large portion of delay on
5581 	 * the critical path. Technically, sptm_prefetch_fte may not find an FTE associated
5582 	 * with pa and return LIBSPTM_FAILURE. However, we are okay with that as it's only
5583 	 * a best-effort performance optimization.
5584 	 */
5585 	sptm_prefetch_fte(pmap->ttep);
5586 	sptm_prefetch_fte(pa);
5587 
5588 	__unused const pt_attr_t * const pt_attr = pmap_get_pt_attr(pmap);
5589 
5590 	if ((v) & pt_attr_leaf_offmask(pt_attr)) {
5591 		panic("pmap_enter_options() pmap %p v 0x%llx",
5592 		    pmap, (uint64_t)v);
5593 	}
5594 
5595 	if (__improbable((pmap == kernel_pmap) && (v >= CPUWINDOWS_BASE) && (v < CPUWINDOWS_TOP))) {
5596 		panic("pmap_enter_options() kernel pmap %p v 0x%llx belongs to [CPUWINDOWS_BASE: 0x%llx, CPUWINDOWS_TOP: 0x%llx)",
5597 		    pmap, (uint64_t)v, (uint64_t)CPUWINDOWS_BASE, (uint64_t)CPUWINDOWS_TOP);
5598 	}
5599 
5600 	if ((pa) & pt_attr_leaf_offmask(pt_attr)) {
5601 		panic("pmap_enter_options() pmap %p pa 0x%llx",
5602 		    pmap, (uint64_t)pa);
5603 	}
5604 
5605 	/* The PA should not extend beyond the architected physical address space */
5606 	pa &= ARM_PTE_PAGE_MASK;
5607 
5608 	if ((prot & VM_PROT_EXECUTE) && (pmap == kernel_pmap)) {
5609 #if defined(KERNEL_INTEGRITY_CTRR) && defined(CONFIG_XNUPOST)
5610 		extern vm_offset_t ctrr_test_page;
5611 		if (__probable(v != ctrr_test_page))
5612 #endif
5613 		panic("pmap_enter_options(): attempt to add executable mapping to kernel_pmap");
5614 	}
5615 	assert(pn != vm_page_fictitious_addr);
5616 
5617 	pmap_lock(pmap, PMAP_LOCK_SHARED);
5618 
5619 	/*
5620 	 *	Expand pmap to include this pte.  Assume that
5621 	 *	pmap is always expanded to include enough hardware
5622 	 *	pages to map one VM page.
5623 	 */
5624 	while ((pte_p = pmap_pte(pmap, v)) == PT_ENTRY_NULL) {
5625 		/* Must unlock to expand the pmap. */
5626 		pmap_unlock(pmap, PMAP_LOCK_SHARED);
5627 
5628 		kr = pmap_expand(pmap, v, options, pt_attr_leaf_level(pt_attr));
5629 
5630 		if (kr != KERN_SUCCESS) {
5631 			return kr;
5632 		}
5633 
5634 		pmap_lock(pmap, PMAP_LOCK_SHARED);
5635 	}
5636 
5637 	if (options & PMAP_OPTIONS_NOENTER) {
5638 		pmap_unlock(pmap, PMAP_LOCK_SHARED);
5639 		return KERN_SUCCESS;
5640 	}
5641 
5642 	/*
5643 	 * Since we may not hold the pmap lock exclusive, updating the pte is
5644 	 * done via a cmpxchg loop.
5645 	 * We need to be careful about modifying non-local data structures before commiting
5646 	 * the new pte since we may need to re-do the transaction.
5647 	 */
5648 	const pt_entry_t prev_pte = os_atomic_load(pte_p, relaxed);
5649 
5650 	if (((prev_pte & ARM_PTE_TYPE_VALID) == ARM_PTE_TYPE) && (pte_to_pa(prev_pte) != pa)) {
5651 		/*
5652 		 * There is already a mapping here & it's for a different physical page.
5653 		 * First remove that mapping.
5654 		 * We assume that we can leave the pmap lock held for shared access rather
5655 		 * than exclusive access here, because we assume that the VM won't try to
5656 		 * simultaneously map the same VA to multiple different physical pages.
5657 		 * If that assumption is violated, sptm_map_page() will panic as the architecture
5658 		 * does not allow the output address of a mapping to be changed without a break-
5659 		 * before-make sequence.
5660 		 */
5661 		pmap_remove_range(pmap, v, v + PAGE_SIZE);
5662 	}
5663 
5664 	if (pmap != kernel_pmap) {
5665 		ptd_info_t *ptd_info = ptep_get_info(pte_p);
5666 		wiredcnt = &ptd_info->wiredcnt;
5667 	}
5668 
5669 	while (!committed) {
5670 		pt_entry_t spte = ARM_PTE_TYPE_FAULT;
5671 		pv_alloc_return_t pv_status = PV_ALLOC_SUCCESS;
5672 		bool skip_footprint_debit = false;
5673 
5674 		/*
5675 		 * The XO index is used for TPRO mappings. To avoid exposing them as --x,
5676 		 * the VM code tracks VM_MAP_TPRO requests and couples them with the proper
5677 		 * read-write protection. The PMAP layer though still needs to use the right
5678 		 * index, which is the older XO-now-TPRO one and that is specially selected
5679 		 * here thanks to PMAP_OPTIONS_MAP_TPRO.
5680 		 *
5681 		 * Note that pmap_construct_pte() may check the nested region ASID bitmap,
5682 		 * which needs to happen at every iteration of the commit loop in case we
5683 		 * previously dropped the pmap lock.
5684 		 */
5685 		pt_entry_t pte = pmap_construct_pte(pmap, v, pa,
5686 		    ((options & PMAP_OPTIONS_MAP_TPRO) ? VM_PROT_RORW_TP : prot), fault_type, wired, pt_attr, &pp_attr_bits);
5687 
5688 
5689 		if (pa_valid(pa)) {
5690 			unsigned int pai;
5691 			boolean_t   is_altacct = FALSE, is_internal = FALSE, is_reusable = FALSE, is_external = FALSE;
5692 
5693 			is_internal = FALSE;
5694 			is_altacct = FALSE;
5695 
5696 			pai = pa_index(pa);
5697 			locked_pvh_t locked_pvh;
5698 
5699 			if (__improbable(options & PMAP_OPTIONS_NOPREEMPT)) {
5700 				locked_pvh = pvh_lock_nopreempt(pai);
5701 			} else {
5702 				locked_pvh = pvh_lock(pai);
5703 			}
5704 
5705 			/*
5706 			 * Make sure that the current per-cpu PV free list has
5707 			 * enough entries (2 in the worst-case scenario) to handle the enter_pv
5708 			 * if the transaction succeeds. At this point, preemption has either
5709 			 * been disabled by the caller or by pvh_lock() above.
5710 			 * Note that we can still be interrupted, but a primary
5711 			 * interrupt handler can never enter the pmap.
5712 			 */
5713 			assert(get_preemption_level() > 0);
5714 			local_pv_free = &pmap_get_cpu_data()->pv_free;
5715 			const bool allocation_required = !pvh_test_type(locked_pvh.pvh, PVH_TYPE_NULL) &&
5716 			    !(pvh_test_type(locked_pvh.pvh, PVH_TYPE_PTEP) && pvh_ptep(locked_pvh.pvh) == pte_p);
5717 
5718 			if (__improbable(allocation_required && (local_pv_free->count < 2))) {
5719 				pv_entry_t *new_pve_p[2] = {PV_ENTRY_NULL};
5720 				int new_allocated_pves = 0;
5721 
5722 				while (new_allocated_pves < 2) {
5723 					local_pv_free = &pmap_get_cpu_data()->pv_free;
5724 					pv_status = pv_alloc(pmap, PMAP_LOCK_SHARED, options, &new_pve_p[new_allocated_pves], &locked_pvh, wiredcnt);
5725 					if (pv_status == PV_ALLOC_FAIL) {
5726 						break;
5727 					} else if (pv_status == PV_ALLOC_RETRY) {
5728 						/*
5729 						 * In the case that pv_alloc() had to grab a new page of PVEs,
5730 						 * it will have dropped the pmap lock while doing so.
5731 						 * On non-PPL devices, dropping the lock re-enables preemption so we may
5732 						 * be on a different CPU now.
5733 						 */
5734 						local_pv_free = &pmap_get_cpu_data()->pv_free;
5735 					} else {
5736 						/* If we've gotten this far then a node should've been allocated. */
5737 						assert(new_pve_p[new_allocated_pves] != PV_ENTRY_NULL);
5738 
5739 						new_allocated_pves++;
5740 					}
5741 				}
5742 
5743 				for (int i = 0; i < new_allocated_pves; i++) {
5744 					pv_free(new_pve_p[i]);
5745 				}
5746 			}
5747 
5748 			if (pv_status == PV_ALLOC_FAIL) {
5749 				pvh_unlock(&locked_pvh);
5750 				kr = KERN_RESOURCE_SHORTAGE;
5751 				break;
5752 			} else if (pv_status == PV_ALLOC_RETRY) {
5753 				pvh_unlock(&locked_pvh);
5754 				/* We dropped the pmap and PVH locks to allocate. Retry transaction. */
5755 				continue;
5756 			}
5757 
5758 			if ((flags & (VM_WIMG_MASK | VM_WIMG_USE_DEFAULT))) {
5759 				wimg_bits = (flags & (VM_WIMG_MASK | VM_WIMG_USE_DEFAULT));
5760 			} else {
5761 				wimg_bits = pmap_cache_attributes(pn);
5762 			}
5763 
5764 			/**
5765 			 * We may be retrying this operation after dropping the PVH lock.
5766 			 * Cache attributes for the physical page may have changed while the lock
5767 			 * was dropped, so update PTE cache attributes on each loop iteration.
5768 			 */
5769 			pte |= pmap_get_pt_ops(pmap)->wimg_to_pte(wimg_bits, pa);
5770 
5771 
5772 			const sptm_return_t sptm_status = pmap_enter_pte(pmap, pte_p, pte, &locked_pvh, &spte, v, options, mapping_type);
5773 			assert(committed == false);
5774 			if ((sptm_status == SPTM_SUCCESS) || (sptm_status == SPTM_MAP_VALID)) {
5775 				committed = true;
5776 			} else if (sptm_status == SPTM_MAP_FLUSH_PENDING) {
5777 				pvh_unlock(&locked_pvh);
5778 				continue;
5779 			} else if (sptm_status == SPTM_MAP_CODESIGN_ERROR) {
5780 				pvh_unlock(&locked_pvh);
5781 				kr = KERN_CODESIGN_ERROR;
5782 				break;
5783 			} else {
5784 				pvh_unlock(&locked_pvh);
5785 				kr = KERN_FAILURE;
5786 				break;
5787 			}
5788 			const bool had_valid_mapping = (sptm_status == SPTM_MAP_VALID);
5789 			/* End of transaction. Commit pv changes, pa bits, and memory accounting. */
5790 			if (!had_valid_mapping) {
5791 				pv_entry_t *new_pve_p = PV_ENTRY_NULL;
5792 				int pve_ptep_idx = 0;
5793 				pv_status = pmap_enter_pv(pmap, pte_p, options, PMAP_LOCK_SHARED, &locked_pvh, &new_pve_p, &pve_ptep_idx);
5794 				/* We did all the allocations up top. So this shouldn't be able to fail. */
5795 				if (pv_status != PV_ALLOC_SUCCESS) {
5796 					panic("%s: unexpected pmap_enter_pv ret code: %d. new_pve_p=%p pmap=%p",
5797 					    __func__, pv_status, new_pve_p, pmap);
5798 				}
5799 
5800 				if (pmap != kernel_pmap) {
5801 					if (options & PMAP_OPTIONS_INTERNAL) {
5802 						ppattr_pve_set_internal(pai, new_pve_p, pve_ptep_idx);
5803 						if ((options & PMAP_OPTIONS_ALT_ACCT) ||
5804 						    PMAP_FOOTPRINT_SUSPENDED(pmap)) {
5805 							/*
5806 							 * Make a note to ourselves that this
5807 							 * mapping is using alternative
5808 							 * accounting. We'll need this in order
5809 							 * to know which ledger to debit when
5810 							 * the mapping is removed.
5811 							 *
5812 							 * The altacct bit must be set while
5813 							 * the pv head is locked. Defer the
5814 							 * ledger accounting until after we've
5815 							 * dropped the lock.
5816 							 */
5817 							ppattr_pve_set_altacct(pai, new_pve_p, pve_ptep_idx);
5818 							is_altacct = TRUE;
5819 						}
5820 					}
5821 					if (ppattr_test_reusable(pai) &&
5822 					    !is_altacct) {
5823 						is_reusable = TRUE;
5824 					} else if (options & PMAP_OPTIONS_INTERNAL) {
5825 						is_internal = TRUE;
5826 					} else {
5827 						is_external = TRUE;
5828 					}
5829 				}
5830 			}
5831 
5832 			pvh_unlock(&locked_pvh);
5833 
5834 			if (pp_attr_bits != 0) {
5835 				ppattr_pa_set_bits(pa, pp_attr_bits);
5836 			}
5837 
5838 			if (!had_valid_mapping && (pmap != kernel_pmap)) {
5839 				pmap_ledger_credit(pmap, task_ledgers.phys_mem, pt_attr_page_size(pt_attr) * PAGE_RATIO);
5840 
5841 				if (is_internal) {
5842 					/*
5843 					 * Make corresponding adjustments to
5844 					 * phys_footprint statistics.
5845 					 */
5846 					pmap_ledger_credit(pmap, task_ledgers.internal, pt_attr_page_size(pt_attr) * PAGE_RATIO);
5847 					if (is_altacct) {
5848 						/*
5849 						 * If this page is internal and
5850 						 * in an IOKit region, credit
5851 						 * the task's total count of
5852 						 * dirty, internal IOKit pages.
5853 						 * It should *not* count towards
5854 						 * the task's total physical
5855 						 * memory footprint, because
5856 						 * this entire region was
5857 						 * already billed to the task
5858 						 * at the time the mapping was
5859 						 * created.
5860 						 *
5861 						 * Put another way, this is
5862 						 * internal++ and
5863 						 * alternate_accounting++, so
5864 						 * net effect on phys_footprint
5865 						 * is 0. That means: don't
5866 						 * touch phys_footprint here.
5867 						 */
5868 						pmap_ledger_credit(pmap, task_ledgers.alternate_accounting, pt_attr_page_size(pt_attr) * PAGE_RATIO);
5869 					} else {
5870 						if (pte_is_compressed(spte, pte_p) && !(spte & ARM_PTE_COMPRESSED_ALT)) {
5871 							/* Replacing a compressed page (with internal accounting). No change to phys_footprint. */
5872 							skip_footprint_debit = true;
5873 						} else {
5874 							pmap_ledger_credit(pmap, task_ledgers.phys_footprint, pt_attr_page_size(pt_attr) * PAGE_RATIO);
5875 						}
5876 					}
5877 				}
5878 				if (is_reusable) {
5879 					pmap_ledger_credit(pmap, task_ledgers.reusable, pt_attr_page_size(pt_attr) * PAGE_RATIO);
5880 				} else if (is_external) {
5881 					pmap_ledger_credit(pmap, task_ledgers.external, pt_attr_page_size(pt_attr) * PAGE_RATIO);
5882 				}
5883 			}
5884 		} else {
5885 			if (prot & VM_PROT_EXECUTE) {
5886 				kr = KERN_FAILURE;
5887 				break;
5888 			}
5889 
5890 			wimg_bits = pmap_cache_attributes(pn);
5891 			if ((flags & (VM_WIMG_MASK | VM_WIMG_USE_DEFAULT))) {
5892 				wimg_bits = (wimg_bits & (~VM_WIMG_MASK)) | (flags & (VM_WIMG_MASK | VM_WIMG_USE_DEFAULT));
5893 			}
5894 
5895 			pte |= pmap_get_pt_ops(pmap)->wimg_to_pte(wimg_bits, pa);
5896 
5897 
5898 			/**
5899 			 * pmap_enter_pte() expects to be called with preemption disabled so it can access
5900 			 * the per-CPU prev_ptes array.
5901 			 */
5902 			disable_preemption();
5903 			const sptm_return_t sptm_status = pmap_enter_pte(pmap, pte_p, pte, NULL, &spte, v, options, mapping_type);
5904 			enable_preemption();
5905 			assert(committed == false);
5906 			if ((sptm_status == SPTM_SUCCESS) || (sptm_status == SPTM_MAP_VALID)) {
5907 				committed = true;
5908 
5909 				/**
5910 				 * If there was already a valid pte here then we reuse its
5911 				 * reference on the ptd and drop the one that we took above.
5912 				 */
5913 			} else if (__improbable(sptm_status != SPTM_MAP_FLUSH_PENDING)) {
5914 				panic("%s: Unexpected SPTM return code %u for non-managed PA 0x%llx", __func__, (unsigned int)sptm_status, (unsigned long long)pa);
5915 			}
5916 		}
5917 		if (committed) {
5918 			if (pte_is_compressed(spte, pte_p)) {
5919 				assert(pmap != kernel_pmap);
5920 
5921 				/* One less "compressed" */
5922 				pmap_ledger_debit(pmap, task_ledgers.internal_compressed,
5923 				    pt_attr_page_size(pt_attr) * PAGE_RATIO);
5924 
5925 				if (spte & ARM_PTE_COMPRESSED_ALT) {
5926 					pmap_ledger_debit(pmap, task_ledgers.alternate_accounting_compressed, pt_attr_page_size(pt_attr) * PAGE_RATIO);
5927 				} else if (!skip_footprint_debit) {
5928 					/* Was part of the footprint */
5929 					pmap_ledger_debit(pmap, task_ledgers.phys_footprint, pt_attr_page_size(pt_attr) * PAGE_RATIO);
5930 				}
5931 			}
5932 		}
5933 	}
5934 
5935 	pmap_unlock(pmap, PMAP_LOCK_SHARED);
5936 
5937 	if (kr == KERN_CODESIGN_ERROR) {
5938 		/* Print any logs from TXM */
5939 		txm_print_logs();
5940 	}
5941 	return kr;
5942 }
5943 
5944 kern_return_t
5945 pmap_enter_options_addr(
5946 	pmap_t pmap,
5947 	vm_map_address_t v,
5948 	pmap_paddr_t pa,
5949 	vm_prot_t prot,
5950 	vm_prot_t fault_type,
5951 	unsigned int flags,
5952 	boolean_t wired,
5953 	unsigned int options,
5954 	__unused void   *arg,
5955 	pmap_mapping_type_t mapping_type)
5956 {
5957 	kern_return_t kr = KERN_FAILURE;
5958 
5959 
5960 	PMAP_TRACE(2, PMAP_CODE(PMAP__ENTER) | DBG_FUNC_START,
5961 	    VM_KERNEL_ADDRHIDE(pmap), VM_KERNEL_ADDRHIDE(v), pa, prot);
5962 
5963 	kr = pmap_enter_options_internal(pmap, v, pa, prot, fault_type, flags, wired, options, mapping_type);
5964 
5965 	PMAP_TRACE(2, PMAP_CODE(PMAP__ENTER) | DBG_FUNC_END, kr);
5966 
5967 	return kr;
5968 }
5969 
5970 kern_return_t
5971 pmap_enter_options(
5972 	pmap_t pmap,
5973 	vm_map_address_t v,
5974 	ppnum_t pn,
5975 	vm_prot_t prot,
5976 	vm_prot_t fault_type,
5977 	unsigned int flags,
5978 	boolean_t wired,
5979 	unsigned int options,
5980 	__unused void   *arg,
5981 	pmap_mapping_type_t mapping_type)
5982 {
5983 	return pmap_enter_options_addr(pmap, v, ((pmap_paddr_t)pn) << PAGE_SHIFT, prot,
5984 	           fault_type, flags, wired, options, arg, mapping_type);
5985 }
5986 
5987 /*
5988  *	Routine:	pmap_change_wiring
5989  *	Function:	Change the wiring attribute for a map/virtual-address
5990  *			pair.
5991  *	In/out conditions:
5992  *			The mapping must already exist in the pmap.
5993  */
5994 MARK_AS_PMAP_TEXT void
5995 pmap_change_wiring_internal(
5996 	pmap_t pmap,
5997 	vm_map_address_t v,
5998 	boolean_t wired)
5999 {
6000 	pt_entry_t     *pte_p, prev_pte;
6001 
6002 	validate_pmap_mutable(pmap);
6003 
6004 	pmap_lock(pmap, PMAP_LOCK_SHARED);
6005 
6006 	const pt_entry_t new_wiring = (wired ? ARM_PTE_WIRED : 0);
6007 
6008 	pte_p = pmap_pte(pmap, v);
6009 	if (pte_p == PT_ENTRY_NULL) {
6010 		if (!wired) {
6011 			/*
6012 			 * The PTE may have already been cleared by a disconnect/remove operation, and the L3 table
6013 			 * may have been freed by a remove operation.
6014 			 */
6015 			goto pmap_change_wiring_return;
6016 		} else {
6017 			panic("%s: Attempt to wire nonexistent PTE for pmap %p", __func__, pmap);
6018 		}
6019 	}
6020 
6021 	disable_preemption();
6022 	pmap_sptm_percpu_data_t *sptm_pcpu = PERCPU_GET(pmap_sptm_percpu);
6023 	sptm_pcpu->sptm_templates[0] = (*pte_p & ~ARM_PTE_WIRED) | new_wiring;
6024 
6025 	pmap_retype_epoch_enter();
6026 	sptm_update_region(pmap->ttep, v, 1, sptm_pcpu->sptm_templates_pa, SPTM_UPDATE_SW_WIRED);
6027 	pmap_retype_epoch_exit();
6028 
6029 	prev_pte = os_atomic_load(&sptm_pcpu->sptm_prev_ptes[0], relaxed);
6030 	enable_preemption();
6031 
6032 	if ((prev_pte & ARM_PTE_TYPE_MASK) == ARM_PTE_TYPE_FAULT) {
6033 		goto pmap_change_wiring_return;
6034 	}
6035 
6036 	if ((pmap != kernel_pmap) && (wired != pte_is_wired(prev_pte))) {
6037 		pte_update_wiredcnt(pmap, pte_p, wired);
6038 	}
6039 
6040 pmap_change_wiring_return:
6041 	pmap_unlock(pmap, PMAP_LOCK_SHARED);
6042 }
6043 
6044 void
6045 pmap_change_wiring(
6046 	pmap_t pmap,
6047 	vm_map_address_t v,
6048 	boolean_t wired)
6049 {
6050 	pmap_change_wiring_internal(pmap, v, wired);
6051 }
6052 
6053 MARK_AS_PMAP_TEXT pmap_paddr_t
6054 pmap_find_pa_internal(
6055 	pmap_t pmap,
6056 	addr64_t va)
6057 {
6058 	pmap_paddr_t    pa = 0;
6059 
6060 	validate_pmap(pmap);
6061 
6062 	if (pmap != kernel_pmap) {
6063 		pmap_lock(pmap, PMAP_LOCK_SHARED);
6064 	}
6065 
6066 	pa = pmap_vtophys(pmap, va);
6067 
6068 	if (pmap != kernel_pmap) {
6069 		pmap_unlock(pmap, PMAP_LOCK_SHARED);
6070 	}
6071 
6072 	return pa;
6073 }
6074 
6075 pmap_paddr_t
6076 pmap_find_pa_nofault(pmap_t pmap, addr64_t va)
6077 {
6078 	pmap_paddr_t pa = 0;
6079 
6080 	if (pmap == kernel_pmap) {
6081 		pa = mmu_kvtop(va);
6082 	} else if ((current_thread()->map) && (pmap == vm_map_pmap(current_thread()->map))) {
6083 		/*
6084 		 * Note that this doesn't account for PAN: mmu_uvtop() may return a valid
6085 		 * translation even if PAN would prevent kernel access through the translation.
6086 		 * It's therefore assumed the UVA will be accessed in a PAN-disabled context.
6087 		 */
6088 		pa = mmu_uvtop(va);
6089 	}
6090 	return pa;
6091 }
6092 
6093 pmap_paddr_t
6094 pmap_find_pa(
6095 	pmap_t pmap,
6096 	addr64_t va)
6097 {
6098 	pmap_paddr_t pa = pmap_find_pa_nofault(pmap, va);
6099 
6100 	if (pa != 0) {
6101 		return pa;
6102 	}
6103 
6104 	if (not_in_kdp) {
6105 		return pmap_find_pa_internal(pmap, va);
6106 	} else {
6107 		return pmap_vtophys(pmap, va);
6108 	}
6109 }
6110 
6111 ppnum_t
6112 pmap_find_phys_nofault(
6113 	pmap_t pmap,
6114 	addr64_t va)
6115 {
6116 	ppnum_t ppn;
6117 	ppn = atop(pmap_find_pa_nofault(pmap, va));
6118 	return ppn;
6119 }
6120 
6121 ppnum_t
6122 pmap_find_phys(
6123 	pmap_t pmap,
6124 	addr64_t va)
6125 {
6126 	ppnum_t ppn;
6127 	ppn = atop(pmap_find_pa(pmap, va));
6128 	return ppn;
6129 }
6130 
6131 /**
6132  * Translate a kernel virtual address into a physical address.
6133  *
6134  * @param va The kernel virtual address to translate. Does not work on user
6135  *           virtual addresses.
6136  *
6137  * @return The physical address if the translation was successful, or zero if
6138  *         no valid mappings were found for the given virtual address.
6139  */
6140 pmap_paddr_t
6141 kvtophys(vm_offset_t va)
6142 {
6143 	sptm_paddr_t pa;
6144 
6145 	if (sptm_kvtophys(va, &pa) != LIBSPTM_SUCCESS) {
6146 		return 0;
6147 	}
6148 
6149 	return pa;
6150 }
6151 
6152 /**
6153  * Variant of kvtophys that can't fail. If no mapping is found or the mapping
6154  * points to a non-kernel-managed physical page, then this call will panic().
6155  *
6156  * @note The output of this function is guaranteed to be a kernel-managed
6157  *       physical page, which means it's safe to pass the output directly to
6158  *       pa_index() to create a physical address index for various pmap data
6159  *       structures.
6160  *
6161  * @param va The kernel virtual address to translate. Does not work on user
6162  *           virtual addresses.
6163  *
6164  * @return The translated physical address for the given virtual address.
6165  */
6166 pmap_paddr_t
6167 kvtophys_nofail(vm_offset_t va)
6168 {
6169 	pmap_paddr_t pa;
6170 
6171 	if (__improbable(sptm_kvtophys(va, &pa) != LIBSPTM_SUCCESS)) {
6172 		panic("%s: VA->PA translation failed for va %p", __func__, (void *)va);
6173 	}
6174 
6175 	return pa;
6176 }
6177 
6178 pmap_paddr_t
6179 pmap_vtophys(
6180 	pmap_t pmap,
6181 	addr64_t va)
6182 {
6183 	if ((va < pmap->min) || (va >= pmap->max)) {
6184 		return 0;
6185 	}
6186 
6187 	const pt_attr_t * const pt_attr = pmap_get_pt_attr(pmap);
6188 
6189 	tt_entry_t * ttp = NULL;
6190 	tt_entry_t * ttep = NULL;
6191 	tt_entry_t   tte = ARM_TTE_EMPTY;
6192 	pmap_paddr_t pa = 0;
6193 	unsigned int cur_level;
6194 
6195 	ttp = pmap->tte;
6196 
6197 	for (cur_level = pt_attr_root_level(pt_attr); cur_level <= pt_attr_leaf_level(pt_attr); cur_level++) {
6198 		ttep = &ttp[ttn_index(pt_attr, va, cur_level)];
6199 
6200 		tte = *ttep;
6201 
6202 		const uint64_t valid_mask = pt_attr->pta_level_info[cur_level].valid_mask;
6203 		const uint64_t type_mask = pt_attr->pta_level_info[cur_level].type_mask;
6204 		const uint64_t type_block = pt_attr->pta_level_info[cur_level].type_block;
6205 		const uint64_t offmask = pt_attr->pta_level_info[cur_level].offmask;
6206 
6207 		if ((tte & valid_mask) != valid_mask) {
6208 			return (pmap_paddr_t) 0;
6209 		}
6210 
6211 		/* This detects both leaf entries and intermediate block mappings. */
6212 		if ((tte & type_mask) == type_block) {
6213 			pa = ((tte & ARM_TTE_PA_MASK & ~offmask) | (va & offmask));
6214 			break;
6215 		}
6216 
6217 		ttp = (tt_entry_t*)phystokv(tte & ARM_TTE_TABLE_MASK);
6218 	}
6219 
6220 	return pa;
6221 }
6222 
6223 /*
6224  *	pmap_init_pte_page - Initialize a page table page.
6225  */
6226 MARK_AS_PMAP_TEXT void
6227 pmap_init_pte_page(
6228 	pmap_t pmap,
6229 	pt_entry_t *pte_p,
6230 	vm_offset_t va,
6231 	unsigned int ttlevel,
6232 	boolean_t alloc_ptd)
6233 {
6234 	pt_desc_t   *ptdp = NULL;
6235 	unsigned int pai = pa_index(kvtophys_nofail((vm_offset_t)pte_p));
6236 	const uintptr_t pvh = pai_to_pvh(pai);
6237 
6238 	if (pvh_test_type(pvh, PVH_TYPE_NULL)) {
6239 		if (alloc_ptd) {
6240 			/*
6241 			 * This path should only be invoked from arm_vm_init.  If we are emulating 16KB pages
6242 			 * on 4KB hardware, we may already have allocated a page table descriptor for a
6243 			 * bootstrap request, so we check for an existing PTD here.
6244 			 */
6245 			ptdp = ptd_alloc(pmap, PMAP_PAGE_ALLOCATE_NOWAIT);
6246 			if (ptdp == NULL) {
6247 				panic("%s: unable to allocate PTD", __func__);
6248 			}
6249 			locked_pvh_t locked_pvh = pvh_lock(pai);
6250 			pvh_update_head(&locked_pvh, ptdp, PVH_TYPE_PTDP);
6251 			pvh_unlock(&locked_pvh);
6252 		} else {
6253 			panic("pmap_init_pte_page(): no PTD for pte_p %p", pte_p);
6254 		}
6255 	} else if (pvh_test_type(pvh, PVH_TYPE_PTDP)) {
6256 		ptdp = pvh_ptd(pvh);
6257 	} else {
6258 		panic("pmap_init_pte_page(): invalid PVH type for pte_p %p", pte_p);
6259 	}
6260 
6261 	// pagetable zero-fill and barrier should be guaranteed by the SPTM
6262 	ptd_info_init(ptdp, pmap, va, ttlevel, pte_p);
6263 }
6264 
6265 /*
6266  * This function guarantees that a pmap has the necessary page tables in place
6267  * to map the specified VA.  If necessary, it will allocate new tables at any
6268  * non-root level in the hierarchy (the root table is always already allocated
6269  * and stored in the pmap).
6270  *
6271  * @note This function is expected to be called without any pmap or PVH lock
6272  *       held.
6273  *
6274  * @note It is possible for an L3 table newly allocated by this function to be
6275  *       deleted by another thread before control returns to the caller, iff that
6276  *       table is an ordinary userspace table.  Callers that use this function
6277  *       to allocate new user L3 tables are therefore expected to keep calling
6278  *       this function until they observe a successful L3 PTE lookup with the pmap
6279  *       lock held.  As long as it does not drop the pmap lock, the caller may
6280  *       then safely use the looked-up L3 table.  See the use of this function in
6281  *       pmap_enter_options_internal() for an example.
6282  *
6283  * @param pmap The pmap for which to ensure mapping space is present.
6284  * @param v The virtual address for which to ensure mapping space is present
6285  *          in [pmap].
6286  * @param options Flags to pass to pmap_tt_allocate() if a new table needs to be
6287  *                allocated.  The only valid option is PMAP_OPTIONS_NOWAIT, which
6288  *                specifies that the allocation must not block.
6289  * @param level The maximum paging level for which to ensure a table is present.
6290  *
6291  * @return KERN_INVALID_ADDRESS if [v] is outside the pmap's mappable range,
6292  *         KERN_RESOURCE_SHORTAGE if a new table can't be allocated,
6293  *         KERN_SUCCESS otherwise.
6294  */
6295 MARK_AS_PMAP_TEXT static kern_return_t
6296 pmap_expand(
6297 	pmap_t pmap,
6298 	vm_map_address_t vaddr,
6299 	unsigned int options,
6300 	unsigned int level)
6301 {
6302 	__unused const pt_attr_t * const pt_attr = pmap_get_pt_attr(pmap);
6303 
6304 	if (__improbable((vaddr < pmap->min) || (vaddr >= pmap->max))) {
6305 		return KERN_INVALID_ADDRESS;
6306 	}
6307 	pmap_paddr_t pa;
6308 	const uint64_t pmap_page_size = pt_attr_page_size(pt_attr);
6309 	const uint64_t table_align_mask = (PAGE_SIZE / pmap_page_size) - 1;
6310 	unsigned int ttlevel = pt_attr_root_level(pt_attr);
6311 	tt_entry_t *table_ttep = pmap->tte;
6312 	tt_entry_t *ttep;
6313 	tt_entry_t old_tte = ARM_TTE_EMPTY;
6314 
6315 	pa = 0x0ULL;
6316 
6317 	for (; ttlevel < level; ttlevel++) {
6318 		/**
6319 		 * If the previous iteration didn't allocate a new table, obtain the table from the previous TTE.
6320 		 * Doing this step at the beginning of the loop instead of the end (which would make it part of
6321 		 * the prior iteration) avoids the possibility of executing this step to extract an L3 table KVA
6322 		 * from an L2 TTE, which would be useless because there would be no next iteration to make use
6323 		 * of the table KVA.
6324 		 */
6325 		if (table_ttep == NULL) {
6326 			assert((old_tte & (ARM_TTE_TYPE_MASK | ARM_TTE_VALID)) == (ARM_TTE_TYPE_TABLE | ARM_TTE_VALID));
6327 			table_ttep = (tt_entry_t*)phystokv(old_tte & ARM_TTE_TABLE_MASK);
6328 		}
6329 
6330 		vm_map_address_t v = pt_attr_align_va(pt_attr, ttlevel, vaddr);
6331 
6332 		/**
6333 		 * We don't need to hold the pmap lock while walking the paging hierarchy.  Only L3 tables are
6334 		 * allowed to be dynamically removed, and only for regular user pmaps at that.  We may allocate
6335 		 * a new L3 table below, but we will only access L0-L2 tables, so there's no risk of a table
6336 		 * being deleted while we are using it for the next level(s) of lookup.
6337 		 */
6338 		ttep = &table_ttep[ttn_index(pt_attr, vaddr, ttlevel)];
6339 		old_tte = os_atomic_load(ttep, relaxed);
6340 		table_ttep = NULL;
6341 		if ((old_tte & (ARM_TTE_TYPE_MASK | ARM_TTE_VALID)) != (ARM_TTE_TYPE_TABLE | ARM_TTE_VALID)) {
6342 			tt_entry_t new_tte, *new_ttep;
6343 			while (pmap_tt_allocate(pmap, &new_ttep, ttlevel + 1, options | PMAP_PAGE_NOZEROFILL) != KERN_SUCCESS) {
6344 				if (options & PMAP_OPTIONS_NOWAIT) {
6345 					return KERN_RESOURCE_SHORTAGE;
6346 				}
6347 				VM_PAGE_WAIT();
6348 			}
6349 			/* Grab the pmap lock to ensure we don't try to concurrently map different tables at the same TTE. */
6350 			pmap_lock(pmap, PMAP_LOCK_EXCLUSIVE);
6351 			old_tte = os_atomic_load(ttep, relaxed);
6352 			if ((old_tte & (ARM_TTE_TYPE_MASK | ARM_TTE_VALID)) != (ARM_TTE_TYPE_TABLE | ARM_TTE_VALID)) {
6353 				pmap_init_pte_page(pmap, (pt_entry_t *) new_ttep, v, ttlevel + 1, FALSE);
6354 				pa = kvtophys_nofail((vm_offset_t)new_ttep);
6355 				/*
6356 				 * If the table is going to map a kernel RO zone VA region, then we must
6357 				 * upgrade its SPTM type to XNU_PAGE_TABLE_ROZONE.  The SPTM's type system
6358 				 * requires the table to be transitioned through XNU_DEFAULT for refcount
6359 				 * enforcement, which is fine since this path is expected to execute only
6360 				 * once during boot.
6361 				 */
6362 				if (__improbable(ttlevel == pt_attr_twig_level(pt_attr)) &&
6363 				    (pmap == kernel_pmap) && zone_spans_ro_va(vaddr, vaddr + PAGE_SIZE)) {
6364 					sptm_retype_params_t retype_params = {.raw = SPTM_RETYPE_PARAMS_NULL};
6365 					sptm_retype(pa, XNU_PAGE_TABLE, XNU_DEFAULT, retype_params);
6366 					retype_params.level = (sptm_pt_level_t)pt_attr_leaf_level(pt_attr);
6367 					sptm_retype(pa, XNU_DEFAULT, XNU_PAGE_TABLE_ROZONE, retype_params);
6368 				}
6369 				new_tte = (pa & ARM_TTE_TABLE_MASK) | ARM_TTE_TYPE_TABLE | ARM_TTE_VALID;
6370 				sptm_map_table(pmap->ttep, v, (sptm_pt_level_t)ttlevel, new_tte);
6371 				PMAP_TRACE(4 + ttlevel, PMAP_CODE(PMAP__TTE), VM_KERNEL_ADDRHIDE(pmap), VM_KERNEL_ADDRHIDE(v & ~pt_attr_ln_offmask(pt_attr, ttlevel)),
6372 				    VM_KERNEL_ADDRHIDE((v & ~pt_attr_ln_offmask(pt_attr, ttlevel)) + pt_attr_ln_size(pt_attr, ttlevel)), new_tte);
6373 				/**
6374 				 * If we need to set up multiple TTEs mapping different parts of the same page
6375 				 * (e.g. because we're carving multiple 4K page tables out of a 16K native page,
6376 				 * determine which of the grouped TTEs is the one that we need to follow for the
6377 				 * next level of the table walk.
6378 				 */
6379 				table_ttep = new_ttep + ((((uintptr_t)ttep / sizeof(tt_entry_t)) & table_align_mask) *
6380 				    (pmap_page_size / sizeof(tt_entry_t)));
6381 				pa = 0x0ULL;
6382 				new_ttep = (tt_entry_t *)NULL;
6383 			}
6384 			pmap_unlock(pmap, PMAP_LOCK_EXCLUSIVE);
6385 
6386 			if (new_ttep != (tt_entry_t *)NULL) {
6387 				pmap_tt_deallocate(pmap, new_ttep, ttlevel + 1);
6388 				new_ttep = (tt_entry_t *)NULL;
6389 			}
6390 		}
6391 	}
6392 
6393 	return KERN_SUCCESS;
6394 }
6395 
6396 /*
6397  *	Routine:	pmap_gc
6398  *	Function:
6399  *              Pmap garbage collection
6400  *		Called by the pageout daemon when pages are scarce.
6401  *
6402  */
6403 void
6404 pmap_gc(void)
6405 {
6406 	/*
6407 	 * TODO: as far as I can tell this has never been implemented to do anything meaninful.
6408 	 * We can't just destroy any old pmap on the chance that it may be active on a CPU
6409 	 * or may contain wired mappings.  However, it may make sense to scan the pmap VM
6410 	 * object here, and for each page consult the SPTM frame table and if necessary
6411 	 * the PTD in the PV head table.  If the frame table indicates the page is a leaf
6412 	 * page table page and the PTD indicates it has no wired mappings, we can call
6413 	 * pmap_remove() on the VA region mapped by the page and therein return the page
6414 	 * to the VM.
6415 	 */
6416 }
6417 
6418 /*
6419  *      By default, don't attempt pmap GC more frequently
6420  *      than once / 1 minutes.
6421  */
6422 
6423 void
6424 compute_pmap_gc_throttle(
6425 	void *arg __unused)
6426 {
6427 }
6428 
6429 /*
6430  * pmap_attribute_cache_sync(vm_offset_t pa)
6431  *
6432  * Invalidates all of the instruction cache on a physical page and
6433  * pushes any dirty data from the data cache for the same physical page
6434  */
6435 
6436 kern_return_t
6437 pmap_attribute_cache_sync(
6438 	ppnum_t pp,
6439 	vm_size_t size,
6440 	__unused vm_machine_attribute_t attribute,
6441 	__unused vm_machine_attribute_val_t * value)
6442 {
6443 	if (size > PAGE_SIZE) {
6444 		panic("pmap_attribute_cache_sync size: 0x%llx", (uint64_t)size);
6445 	} else {
6446 		cache_sync_page(pp);
6447 	}
6448 
6449 	return KERN_SUCCESS;
6450 }
6451 
6452 /*
6453  * pmap_sync_page_data_phys(ppnum_t pp)
6454  *
6455  * Invalidates all of the instruction cache on a physical page and
6456  * pushes any dirty data from the data cache for the same physical page.
6457  * Not required on SPTM systems, because the SPTM automatically performs
6458  * the invalidate operation when retyping to one of the types that allow
6459  * for executable permissions.
6460  */
6461 void
6462 pmap_sync_page_data_phys(
6463 	__unused ppnum_t pp)
6464 {
6465 	return;
6466 }
6467 
6468 /*
6469  * pmap_sync_page_attributes_phys(ppnum_t pp)
6470  *
6471  * Write back and invalidate all cachelines on a physical page.
6472  */
6473 void
6474 pmap_sync_page_attributes_phys(
6475 	ppnum_t pp)
6476 {
6477 	flush_dcache((vm_offset_t) (pp << PAGE_SHIFT), PAGE_SIZE, TRUE);
6478 }
6479 
6480 #if CONFIG_COREDUMP
6481 /* temporary workaround */
6482 boolean_t
6483 coredumpok(
6484 	vm_map_t map,
6485 	mach_vm_offset_t va)
6486 {
6487 	pt_entry_t     *pte_p;
6488 	pt_entry_t      spte;
6489 
6490 	pte_p = pmap_pte(map->pmap, va);
6491 	if (0 == pte_p) {
6492 		return FALSE;
6493 	}
6494 	if (vm_map_entry_has_device_pager(map, va)) {
6495 		return FALSE;
6496 	}
6497 	spte = *pte_p;
6498 	return (spte & ARM_PTE_ATTRINDXMASK) == ARM_PTE_ATTRINDX(CACHE_ATTRINDX_DEFAULT);
6499 }
6500 #endif
6501 
6502 void
6503 fillPage(
6504 	ppnum_t pn,
6505 	unsigned int fill)
6506 {
6507 	unsigned int   *addr;
6508 	int             count;
6509 
6510 	addr = (unsigned int *) phystokv(ptoa(pn));
6511 	count = PAGE_SIZE / sizeof(unsigned int);
6512 	while (count--) {
6513 		*addr++ = fill;
6514 	}
6515 }
6516 
6517 extern void     mapping_set_mod(ppnum_t pn);
6518 
6519 void
6520 mapping_set_mod(
6521 	ppnum_t pn)
6522 {
6523 	pmap_set_modify(pn);
6524 }
6525 
6526 extern void     mapping_set_ref(ppnum_t pn);
6527 
6528 void
6529 mapping_set_ref(
6530 	ppnum_t pn)
6531 {
6532 	pmap_set_reference(pn);
6533 }
6534 
6535 /*
6536  * Clear specified attribute bits.
6537  *
6538  * Try to force an arm_fast_fault() for all mappings of
6539  * the page - to force attributes to be set again at fault time.
6540  * If the forcing succeeds, clear the cached bits at the head.
6541  * Otherwise, something must have been wired, so leave the cached
6542  * attributes alone.
6543  */
6544 MARK_AS_PMAP_TEXT static void
6545 phys_attribute_clear_with_flush_range(
6546 	ppnum_t         pn,
6547 	unsigned int    bits,
6548 	int             options,
6549 	void            *arg,
6550 	pmap_tlb_flush_range_t *flush_range)
6551 {
6552 	pmap_paddr_t    pa = ptoa(pn);
6553 	vm_prot_t       allow_mode = VM_PROT_ALL;
6554 
6555 	if ((arg != NULL) || (flush_range != NULL)) {
6556 		options = options & ~PMAP_OPTIONS_NOFLUSH;
6557 	}
6558 
6559 	if (__improbable((options & PMAP_OPTIONS_FF_WIRED) != 0)) {
6560 		panic("phys_attribute_clear(%#010x,%#010x,%#010x,%p,%p): "
6561 		    "invalid options",
6562 		    pn, bits, options, arg, flush_range);
6563 	}
6564 
6565 	if (__improbable((bits & PP_ATTR_MODIFIED) &&
6566 	    (options & PMAP_OPTIONS_NOFLUSH))) {
6567 		panic("phys_attribute_clear(%#010x,%#010x,%#010x,%p,%p): "
6568 		    "should not clear 'modified' without flushing TLBs",
6569 		    pn, bits, options, arg, flush_range);
6570 	}
6571 
6572 	assert(pn != vm_page_fictitious_addr);
6573 
6574 	if (options & PMAP_OPTIONS_CLEAR_WRITE) {
6575 		assert(bits == PP_ATTR_MODIFIED);
6576 
6577 		pmap_page_protect_options_with_flush_range(pn, (VM_PROT_ALL & ~VM_PROT_WRITE), options, NULL, flush_range);
6578 		/*
6579 		 * We short circuit this case; it should not need to
6580 		 * invoke arm_force_fast_fault, so just clear the modified bit.
6581 		 * pmap_page_protect has taken care of resetting
6582 		 * the state so that we'll see the next write as a fault to
6583 		 * the VM (i.e. we don't want a fast fault).
6584 		 */
6585 		ppattr_pa_clear_bits(pa, (pp_attr_t)bits);
6586 		return;
6587 	}
6588 	if (bits & PP_ATTR_REFERENCED) {
6589 		allow_mode &= ~(VM_PROT_READ | VM_PROT_EXECUTE);
6590 	}
6591 	if (bits & PP_ATTR_MODIFIED) {
6592 		allow_mode &= ~VM_PROT_WRITE;
6593 	}
6594 
6595 	if (bits == PP_ATTR_NOENCRYPT) {
6596 		/*
6597 		 * We short circuit this case; it should not need to
6598 		 * invoke arm_force_fast_fault, so just clear and
6599 		 * return.  On ARM, this bit is just a debugging aid.
6600 		 */
6601 		ppattr_pa_clear_bits(pa, (pp_attr_t)bits);
6602 		return;
6603 	}
6604 
6605 	arm_force_fast_fault_with_flush_range(pn, allow_mode, options, NULL, (pp_attr_t)bits, flush_range);
6606 }
6607 
6608 MARK_AS_PMAP_TEXT void
6609 phys_attribute_clear_internal(
6610 	ppnum_t         pn,
6611 	unsigned int    bits,
6612 	int             options,
6613 	void            *arg)
6614 {
6615 	phys_attribute_clear_with_flush_range(pn, bits, options, arg, NULL);
6616 }
6617 
6618 #if __ARM_RANGE_TLBI__
6619 
6620 MARK_AS_PMAP_TEXT static vm_map_address_t
6621 phys_attribute_clear_twig_internal(
6622 	pmap_t pmap,
6623 	vm_map_address_t start,
6624 	vm_map_address_t end,
6625 	unsigned int bits,
6626 	unsigned int options,
6627 	pmap_tlb_flush_range_t *flush_range)
6628 {
6629 	pmap_assert_locked(pmap, PMAP_LOCK_SHARED);
6630 	const pt_attr_t * const pt_attr = pmap_get_pt_attr(pmap);
6631 	assert(end >= start);
6632 	assert((end - start) <= pt_attr_twig_size(pt_attr));
6633 	const uint64_t pmap_page_size = pt_attr_page_size(pt_attr);
6634 	vm_map_address_t va = start;
6635 	pt_entry_t     *pte_p, *start_pte_p, *end_pte_p, *curr_pte_p;
6636 	tt_entry_t     *tte_p;
6637 	tte_p = pmap_tte(pmap, start);
6638 
6639 	/**
6640 	 * It's possible that this portion of our VA region has never been paged in, in which case
6641 	 * there may not be a valid twig or leaf table here.
6642 	 */
6643 	if ((tte_p == (tt_entry_t *) NULL) || ((*tte_p & ARM_TTE_TYPE_MASK) != ARM_TTE_TYPE_TABLE)) {
6644 		assert(flush_range->pending_region_entries == 0);
6645 		return end;
6646 	}
6647 
6648 	pte_p = (pt_entry_t *) ttetokv(*tte_p);
6649 
6650 	start_pte_p = &pte_p[pte_index(pt_attr, start)];
6651 	end_pte_p = start_pte_p + ((end - start) >> pt_attr_leaf_shift(pt_attr));
6652 	assert(end_pte_p >= start_pte_p);
6653 	for (curr_pte_p = start_pte_p; curr_pte_p < end_pte_p; curr_pte_p++, va += pmap_page_size) {
6654 		if (flush_range->pending_region_entries == 0) {
6655 			flush_range->pending_region_start = va;
6656 		} else {
6657 			assertf((flush_range->pending_region_start +
6658 			    (flush_range->pending_region_entries * pmap_page_size)) == va,
6659 			    "pending_region_start 0x%llx + 0x%lx pages != va 0%llx",
6660 			    (unsigned long long)flush_range->pending_region_start,
6661 			    (unsigned long)flush_range->pending_region_entries,
6662 			    (unsigned long long)va);
6663 		}
6664 		flush_range->current_ptep = curr_pte_p;
6665 		const pt_entry_t spte = os_atomic_load(curr_pte_p, relaxed);
6666 		const pmap_paddr_t pa = pte_to_pa(spte);
6667 		if (((spte & ARM_PTE_TYPE_MASK) != ARM_PTE_TYPE_FAULT) && pa_valid(pa)) {
6668 			/* The PTE maps a managed page, so do the appropriate PV list-based permission changes. */
6669 			const ppnum_t pn = (ppnum_t) atop(pa);
6670 			phys_attribute_clear_with_flush_range(pn, bits, options, NULL, flush_range);
6671 			if (__probable(flush_range->region_entry_added)) {
6672 				flush_range->region_entry_added = false;
6673 			} else {
6674 				/**
6675 				 * It's possible that some other thread removed the mapping between our check
6676 				 * of the PTE above and taking the PVH lock in the
6677 				 * phys_attribute_clear_with_flush_range() path.  In that case we have a
6678 				 * discontinuity in the region to update, so just submit any pending region
6679 				 * templates and start a new region op on the next iteration.
6680 				 */
6681 				pmap_multipage_op_submit_region(flush_range);
6682 			}
6683 		} else if (__improbable((spte & ARM_PTE_TYPE_MASK) == ARM_PTE_TYPE_FAULT)) {
6684 			/**
6685 			 * We've found an invalid mapping, so we have a discontinuity in the the region to
6686 			 * update.  Handle this by submitting any pending region templates and starting a new
6687 			 * region on the next iteration.  In theory we could instead handle this by installing
6688 			 * a "safe" (AF bit cleared, minimal permissions) PTE template; the SPTM would just
6689 			 * ignore the update on finding an invalid mapping in the PTE.  But we don't know
6690 			 * what a "safe" template will be in all cases: for example, JIT regions require all
6691 			 * mapping to either be invalid or to have full RWX permissions.
6692 			 */
6693 			pmap_multipage_op_submit_region(flush_range);
6694 		} else if (pmap_insert_flush_range_template(spte, flush_range)) {
6695 			/**
6696 			 * We've found a mapping to a non-managed page, so just insert the existing
6697 			 * PTE into the pending region ops since we don't manage attributes for non-managed
6698 			 * pages.
6699 			 * If pmap_insert_flush_range_template() returns true, indicating that it reached
6700 			 * the mapping limit and submitted the SPTM call, then we also submit any pending
6701 			 * disjoint ops.  Having pending operations in either category will keep preemption
6702 			 * disabled, and we want to ensure that we can at least temporarily
6703 			 * re-enable preemption every SPTM_MAPPING_LIMIT mappings.
6704 			 */
6705 			pmap_multipage_op_submit_disjoint(0, flush_range);
6706 		}
6707 		if (((flush_range->processed_entries + flush_range->pending_disjoint_entries +
6708 		    flush_range->pending_region_entries) >= SPTM_MAPPING_LIMIT) &&
6709 		    pmap_pending_preemption()) {
6710 			pmap_multipage_op_submit(flush_range);
6711 			assert(preemption_enabled());
6712 		}
6713 	}
6714 
6715 	/* SPTM region ops can't span L3 table boundaries, so submit any pending region templates now. */
6716 	pmap_multipage_op_submit_region(flush_range);
6717 	return end;
6718 }
6719 
6720 MARK_AS_PMAP_TEXT vm_map_address_t
6721 phys_attribute_clear_range_internal(
6722 	pmap_t pmap,
6723 	vm_map_address_t start,
6724 	vm_map_address_t end,
6725 	unsigned int bits,
6726 	unsigned int options)
6727 {
6728 	if (__improbable(end < start)) {
6729 		panic("%s: invalid address range %p, %p", __func__, (void*)start, (void*)end);
6730 	}
6731 	validate_pmap_mutable(pmap);
6732 
6733 	vm_map_address_t va = start;
6734 	pmap_tlb_flush_range_t flush_range = {
6735 		.ptfr_pmap = pmap,
6736 		.ptfr_start = start,
6737 		.ptfr_end = end,
6738 		.current_ptep = NULL,
6739 		.pending_region_start = 0,
6740 		.pending_region_entries = 0,
6741 		.region_entry_added = false,
6742 		.current_header = NULL,
6743 		.current_header_first_mapping_index = 0,
6744 		.processed_entries = 0,
6745 		.pending_disjoint_entries = 0,
6746 		.ptfr_flush_needed = false
6747 	};
6748 
6749 	pmap_lock(pmap, PMAP_LOCK_SHARED);
6750 	const pt_attr_t * const pt_attr = pmap_get_pt_attr(pmap);
6751 
6752 	while (va < end) {
6753 		vm_map_address_t curr_end;
6754 
6755 		curr_end = ((va + pt_attr_twig_size(pt_attr)) & ~pt_attr_twig_offmask(pt_attr));
6756 		if (curr_end > end) {
6757 			curr_end = end;
6758 		}
6759 
6760 		va = phys_attribute_clear_twig_internal(pmap, va, curr_end, bits, options, &flush_range);
6761 	}
6762 	pmap_multipage_op_submit(&flush_range);
6763 	pmap_unlock(pmap, PMAP_LOCK_SHARED);
6764 	assert((flush_range.pending_disjoint_entries == 0) && (flush_range.pending_region_entries == 0));
6765 	if (flush_range.ptfr_flush_needed) {
6766 		pmap_get_pt_ops(pmap)->flush_tlb_region_async(
6767 			flush_range.ptfr_start,
6768 			flush_range.ptfr_end - flush_range.ptfr_start,
6769 			flush_range.ptfr_pmap,
6770 			true);
6771 		sync_tlb_flush();
6772 	}
6773 	return va;
6774 }
6775 
6776 static void
6777 phys_attribute_clear_range(
6778 	pmap_t pmap,
6779 	vm_map_address_t start,
6780 	vm_map_address_t end,
6781 	unsigned int bits,
6782 	unsigned int options)
6783 {
6784 	/*
6785 	 * We allow single-page requests to execute non-preemptibly,
6786 	 * as it doesn't make sense to sample AST_URGENT for a single-page
6787 	 * operation, and there are a couple of special use cases that
6788 	 * require a non-preemptible single-page operation.
6789 	 */
6790 	if ((end - start) > (pt_attr_page_size(pmap_get_pt_attr(pmap)) * PAGE_RATIO)) {
6791 		pmap_verify_preemptible();
6792 	}
6793 	__assert_only const int preemption_level = get_preemption_level();
6794 
6795 	PMAP_TRACE(3, PMAP_CODE(PMAP__ATTRIBUTE_CLEAR_RANGE) | DBG_FUNC_START, bits);
6796 
6797 	phys_attribute_clear_range_internal(pmap, start, end, bits, options);
6798 
6799 	PMAP_TRACE(3, PMAP_CODE(PMAP__ATTRIBUTE_CLEAR_RANGE) | DBG_FUNC_END);
6800 
6801 	assert(preemption_level == get_preemption_level());
6802 }
6803 #endif /* __ARM_RANGE_TLBI__ */
6804 
6805 static void
6806 phys_attribute_clear(
6807 	ppnum_t         pn,
6808 	unsigned int    bits,
6809 	int             options,
6810 	void            *arg)
6811 {
6812 	/*
6813 	 * Do we really want this tracepoint?  It will be extremely chatty.
6814 	 * Also, should we have a corresponding trace point for the set path?
6815 	 */
6816 	PMAP_TRACE(3, PMAP_CODE(PMAP__ATTRIBUTE_CLEAR) | DBG_FUNC_START, pn, bits);
6817 
6818 	phys_attribute_clear_internal(pn, bits, options, arg);
6819 
6820 	PMAP_TRACE(3, PMAP_CODE(PMAP__ATTRIBUTE_CLEAR) | DBG_FUNC_END);
6821 }
6822 
6823 /*
6824  *	Set specified attribute bits.
6825  *
6826  *	Set cached value in the pv head because we have
6827  *	no per-mapping hardware support for referenced and
6828  *	modify bits.
6829  */
6830 MARK_AS_PMAP_TEXT void
6831 phys_attribute_set_internal(
6832 	ppnum_t pn,
6833 	unsigned int bits)
6834 {
6835 	pmap_paddr_t    pa = ptoa(pn);
6836 	assert(pn != vm_page_fictitious_addr);
6837 
6838 	ppattr_pa_set_bits(pa, (uint16_t)bits);
6839 
6840 	return;
6841 }
6842 
6843 static void
6844 phys_attribute_set(
6845 	ppnum_t pn,
6846 	unsigned int bits)
6847 {
6848 	phys_attribute_set_internal(pn, bits);
6849 }
6850 
6851 
6852 /*
6853  *	Check specified attribute bits.
6854  *
6855  *	use the software cached bits (since no hw support).
6856  */
6857 static boolean_t
6858 phys_attribute_test(
6859 	ppnum_t pn,
6860 	unsigned int bits)
6861 {
6862 	pmap_paddr_t    pa = ptoa(pn);
6863 	assert(pn != vm_page_fictitious_addr);
6864 	return ppattr_pa_test_bits(pa, (pp_attr_t)bits);
6865 }
6866 
6867 
6868 /*
6869  *	Set the modify/reference bits on the specified physical page.
6870  */
6871 void
6872 pmap_set_modify(ppnum_t pn)
6873 {
6874 	phys_attribute_set(pn, PP_ATTR_MODIFIED);
6875 }
6876 
6877 
6878 /*
6879  *	Clear the modify bits on the specified physical page.
6880  */
6881 void
6882 pmap_clear_modify(
6883 	ppnum_t pn)
6884 {
6885 	phys_attribute_clear(pn, PP_ATTR_MODIFIED, 0, NULL);
6886 }
6887 
6888 
6889 /*
6890  *	pmap_is_modified:
6891  *
6892  *	Return whether or not the specified physical page is modified
6893  *	by any physical maps.
6894  */
6895 boolean_t
6896 pmap_is_modified(
6897 	ppnum_t pn)
6898 {
6899 	return phys_attribute_test(pn, PP_ATTR_MODIFIED);
6900 }
6901 
6902 
6903 /*
6904  *	Set the reference bit on the specified physical page.
6905  */
6906 static void
6907 pmap_set_reference(
6908 	ppnum_t pn)
6909 {
6910 	phys_attribute_set(pn, PP_ATTR_REFERENCED);
6911 }
6912 
6913 /*
6914  *	Clear the reference bits on the specified physical page.
6915  */
6916 void
6917 pmap_clear_reference(
6918 	ppnum_t pn)
6919 {
6920 	phys_attribute_clear(pn, PP_ATTR_REFERENCED, 0, NULL);
6921 }
6922 
6923 
6924 /*
6925  *	pmap_is_referenced:
6926  *
6927  *	Return whether or not the specified physical page is referenced
6928  *	by any physical maps.
6929  */
6930 boolean_t
6931 pmap_is_referenced(
6932 	ppnum_t pn)
6933 {
6934 	return phys_attribute_test(pn, PP_ATTR_REFERENCED);
6935 }
6936 
6937 /*
6938  * pmap_get_refmod(phys)
6939  *  returns the referenced and modified bits of the specified
6940  *  physical page.
6941  */
6942 unsigned int
6943 pmap_get_refmod(
6944 	ppnum_t pn)
6945 {
6946 	return ((phys_attribute_test(pn, PP_ATTR_MODIFIED)) ? VM_MEM_MODIFIED : 0)
6947 	       | ((phys_attribute_test(pn, PP_ATTR_REFERENCED)) ? VM_MEM_REFERENCED : 0);
6948 }
6949 
6950 static inline unsigned int
6951 pmap_clear_refmod_mask_to_modified_bits(const unsigned int mask)
6952 {
6953 	return ((mask & VM_MEM_MODIFIED) ? PP_ATTR_MODIFIED : 0) |
6954 	       ((mask & VM_MEM_REFERENCED) ? PP_ATTR_REFERENCED : 0);
6955 }
6956 
6957 /*
6958  * pmap_clear_refmod(phys, mask)
6959  *  clears the referenced and modified bits as specified by the mask
6960  *  of the specified physical page.
6961  */
6962 void
6963 pmap_clear_refmod_options(
6964 	ppnum_t         pn,
6965 	unsigned int    mask,
6966 	unsigned int    options,
6967 	void            *arg)
6968 {
6969 	unsigned int    bits;
6970 
6971 	bits = pmap_clear_refmod_mask_to_modified_bits(mask);
6972 	phys_attribute_clear(pn, bits, options, arg);
6973 }
6974 
6975 /*
6976  * Perform pmap_clear_refmod_options on a virtual address range.
6977  * The operation will be performed in bulk & tlb flushes will be coalesced
6978  * if possible.
6979  *
6980  * Returns true if the operation is supported on this platform.
6981  * If this function returns false, the operation is not supported and
6982  * nothing has been modified in the pmap.
6983  */
6984 bool
6985 pmap_clear_refmod_range_options(
6986 	pmap_t pmap __unused,
6987 	vm_map_address_t start __unused,
6988 	vm_map_address_t end __unused,
6989 	unsigned int mask __unused,
6990 	unsigned int options __unused)
6991 {
6992 #if __ARM_RANGE_TLBI__
6993 	unsigned int    bits;
6994 	bits = pmap_clear_refmod_mask_to_modified_bits(mask);
6995 	phys_attribute_clear_range(pmap, start, end, bits, options);
6996 	return true;
6997 #else /* __ARM_RANGE_TLBI__ */
6998 #pragma unused(pmap, start, end, mask, options)
6999 	/*
7000 	 * This operation allows the VM to bulk modify refmod bits on a virtually
7001 	 * contiguous range of addresses. This is large performance improvement on
7002 	 * platforms that support ranged tlbi instructions. But on older platforms,
7003 	 * we can only flush per-page or the entire asid. So we currently
7004 	 * only support this operation on platforms that support ranged tlbi.
7005 	 * instructions. On other platforms, we require that
7006 	 * the VM modify the bits on a per-page basis.
7007 	 */
7008 	return false;
7009 #endif /* __ARM_RANGE_TLBI__ */
7010 }
7011 
7012 void
7013 pmap_clear_refmod(
7014 	ppnum_t pn,
7015 	unsigned int mask)
7016 {
7017 	pmap_clear_refmod_options(pn, mask, 0, NULL);
7018 }
7019 
7020 unsigned int
7021 pmap_disconnect_options(
7022 	ppnum_t pn,
7023 	unsigned int options,
7024 	void *arg)
7025 {
7026 	if ((options & PMAP_OPTIONS_COMPRESSOR_IFF_MODIFIED)) {
7027 		/*
7028 		 * On ARM, the "modified" bit is managed by software, so
7029 		 * we know up-front if the physical page is "modified",
7030 		 * without having to scan all the PTEs pointing to it.
7031 		 * The caller should have made the VM page "busy" so noone
7032 		 * should be able to establish any new mapping and "modify"
7033 		 * the page behind us.
7034 		 */
7035 		if (pmap_is_modified(pn)) {
7036 			/*
7037 			 * The page has been modified and will be sent to
7038 			 * the VM compressor.
7039 			 */
7040 			options |= PMAP_OPTIONS_COMPRESSOR;
7041 		} else {
7042 			/*
7043 			 * The page hasn't been modified and will be freed
7044 			 * instead of compressed.
7045 			 */
7046 		}
7047 	}
7048 
7049 	/* disconnect the page */
7050 	pmap_page_protect_options(pn, 0, options, arg);
7051 
7052 	/* return ref/chg status */
7053 	return pmap_get_refmod(pn);
7054 }
7055 
7056 /*
7057  *	Routine:
7058  *		pmap_disconnect
7059  *
7060  *	Function:
7061  *		Disconnect all mappings for this page and return reference and change status
7062  *		in generic format.
7063  *
7064  */
7065 unsigned int
7066 pmap_disconnect(
7067 	ppnum_t pn)
7068 {
7069 	pmap_page_protect(pn, 0);       /* disconnect the page */
7070 	return pmap_get_refmod(pn);   /* return ref/chg status */
7071 }
7072 
7073 boolean_t
7074 pmap_has_managed_page(ppnum_t first, ppnum_t last)
7075 {
7076 	if (ptoa(first) >= vm_last_phys) {
7077 		return FALSE;
7078 	}
7079 	if (ptoa(last) < vm_first_phys) {
7080 		return FALSE;
7081 	}
7082 
7083 	return TRUE;
7084 }
7085 
7086 /*
7087  * The state maintained by the noencrypt functions is used as a
7088  * debugging aid on ARM.  This incurs some overhead on the part
7089  * of the caller.  A special case check in phys_attribute_clear
7090  * (the most expensive path) currently minimizes this overhead,
7091  * but stubbing these functions out on RELEASE kernels yields
7092  * further wins.
7093  */
7094 boolean_t
7095 pmap_is_noencrypt(
7096 	ppnum_t pn)
7097 {
7098 #if DEVELOPMENT || DEBUG
7099 	boolean_t result = FALSE;
7100 
7101 	if (!pa_valid(ptoa(pn))) {
7102 		return FALSE;
7103 	}
7104 
7105 	result = (phys_attribute_test(pn, PP_ATTR_NOENCRYPT));
7106 
7107 	return result;
7108 #else
7109 #pragma unused(pn)
7110 	return FALSE;
7111 #endif
7112 }
7113 
7114 void
7115 pmap_set_noencrypt(
7116 	ppnum_t pn)
7117 {
7118 #if DEVELOPMENT || DEBUG
7119 	if (!pa_valid(ptoa(pn))) {
7120 		return;
7121 	}
7122 
7123 	phys_attribute_set(pn, PP_ATTR_NOENCRYPT);
7124 #else
7125 #pragma unused(pn)
7126 #endif
7127 }
7128 
7129 void
7130 pmap_clear_noencrypt(
7131 	ppnum_t pn)
7132 {
7133 #if DEVELOPMENT || DEBUG
7134 	if (!pa_valid(ptoa(pn))) {
7135 		return;
7136 	}
7137 
7138 	phys_attribute_clear(pn, PP_ATTR_NOENCRYPT, 0, NULL);
7139 #else
7140 #pragma unused(pn)
7141 #endif
7142 }
7143 
7144 void
7145 pmap_lock_phys_page(ppnum_t pn)
7146 {
7147 	unsigned int    pai;
7148 	pmap_paddr_t    phys = ptoa(pn);
7149 
7150 	if (pa_valid(phys)) {
7151 		pai = pa_index(phys);
7152 		__unused const locked_pvh_t locked_pvh = pvh_lock(pai);
7153 	} else {
7154 		simple_lock(&phys_backup_lock, LCK_GRP_NULL);
7155 	}
7156 }
7157 
7158 
7159 void
7160 pmap_unlock_phys_page(ppnum_t pn)
7161 {
7162 	unsigned int    pai;
7163 	pmap_paddr_t    phys = ptoa(pn);
7164 
7165 	if (pa_valid(phys)) {
7166 		pai = pa_index(phys);
7167 		locked_pvh_t locked_pvh = {.pvh = pai_to_pvh(pai), .pai = pai};
7168 		pvh_unlock(&locked_pvh);
7169 	} else {
7170 		simple_unlock(&phys_backup_lock);
7171 	}
7172 }
7173 
7174 MARK_AS_PMAP_TEXT void
7175 pmap_clear_user_ttb_internal(void)
7176 {
7177 	set_mmu_ttb(invalid_ttep & TTBR_BADDR_MASK);
7178 }
7179 
7180 void
7181 pmap_clear_user_ttb(void)
7182 {
7183 	PMAP_TRACE(3, PMAP_CODE(PMAP__CLEAR_USER_TTB) | DBG_FUNC_START, NULL, 0, 0);
7184 	pmap_clear_user_ttb_internal();
7185 	PMAP_TRACE(3, PMAP_CODE(PMAP__CLEAR_USER_TTB) | DBG_FUNC_END);
7186 }
7187 
7188 /**
7189  * Set up a "fast fault", or a page fault that won't go through the VM layer on
7190  * a page. This is primarily used to manage ref/mod bits in software. Depending
7191  * on the value of allow_mode, the next read and/or write of the page will fault
7192  * and the ref/mod bits will be updated.
7193  *
7194  * @param ppnum Page number to set up a fast fault on.
7195  * @param allow_mode VM_PROT_NONE will cause the next read and write access to
7196  *                   fault.
7197  *                   VM_PROT_READ will only cause the next write access to fault.
7198  *                   Other values are undefined.
7199  * @param options PMAP_OPTIONS_NOFLUSH indicates TLBI flush is not needed.
7200  *                PMAP_OPTIONS_FF_WIRED forces a fast fault even on wired pages.
7201  *                PMAP_OPTIONS_SET_REUSABLE/PMAP_OPTIONS_CLEAR_REUSABLE updates
7202  *                the global reusable bit of the page.
7203  * @param locked_pvh If non-NULL, this indicates the PVH lock for [ppnum] is already locked
7204  *                   by the caller.  This is an input/output parameter which may be updated
7205  *                   to reflect a new PV head value to be passed to a later call to pvh_unlock().
7206  * @param bits_to_clear Mask of additional pp_attr_t bits to clear for the physical
7207  *                      page, iff this function completes successfully and returns
7208  *                      TRUE.  This is typically some combination of
7209  *                      the referenced, modified, and noencrypt bits.
7210  * @param flush_range When present, this function will skip the TLB flush for the
7211  *                    mappings that are covered by the range, leaving that to be
7212  *                    done later by the caller.  It may also avoid submitting mapping
7213  *                    updates directly to the SPTM, instead accumulating them in a
7214  *                    per-CPU array to be submitted later by the caller.
7215  *
7216  * @return TRUE if the fast fault was successfully configured for all mappings
7217  *         of the page, FALSE otherwise (e.g. if wired mappings are present and
7218  *         PMAP_OPTIONS_FF_WIRED was not passed).
7219  *
7220  * @note PMAP_OPTIONS_NOFLUSH and flush_range cannot both be specified.
7221  *
7222  * @warning PMAP_OPTIONS_FF_WIRED should only be used with pages accessible from
7223  *          EL0.  The kernel may assume that accesses to wired, kernel-owned pages
7224  *          won't fault.
7225  */
7226 MARK_AS_PMAP_TEXT static boolean_t
7227 arm_force_fast_fault_with_flush_range(
7228 	ppnum_t         ppnum,
7229 	vm_prot_t       allow_mode,
7230 	int             options,
7231 	locked_pvh_t   *locked_pvh,
7232 	pp_attr_t       bits_to_clear,
7233 	pmap_tlb_flush_range_t *flush_range)
7234 {
7235 	pmap_paddr_t     phys = ptoa(ppnum);
7236 	pv_entry_t      *pve_p;
7237 	pt_entry_t      *pte_p;
7238 	unsigned int     pai;
7239 	boolean_t        result;
7240 	unsigned int     num_mappings = 0, num_skipped_mappings = 0;
7241 	bool             ref_fault;
7242 	bool             mod_fault;
7243 	bool             clear_write_fault = false;
7244 	bool             ref_aliases_mod = false;
7245 
7246 	assert(ppnum != vm_page_fictitious_addr);
7247 
7248 	/**
7249 	 * Assert that PMAP_OPTIONS_NOFLUSH and flush_range cannot both be specified.
7250 	 *
7251 	 * PMAP_OPTIONS_NOFLUSH indicates there is no need of flushing the TLB in the entire operation, and
7252 	 * flush_range indicates the caller requests deferral of the TLB flushing. Fundemantally, the two
7253 	 * semantics conflict with each other, so assert they are not both true.
7254 	 */
7255 	assert(!(flush_range && (options & PMAP_OPTIONS_NOFLUSH)));
7256 
7257 	if (!pa_valid(phys)) {
7258 		return FALSE;   /* Not a managed page. */
7259 	}
7260 
7261 	result = TRUE;
7262 	ref_fault = false;
7263 	mod_fault = false;
7264 	pai = pa_index(phys);
7265 	locked_pvh_t local_locked_pvh = {.pvh = 0};
7266 	if (__probable(locked_pvh == NULL)) {
7267 		if (flush_range != NULL) {
7268 			/**
7269 			 * If we're partway through processing a multi-page batched call,
7270 			 * preemption will already be disabled so we can't simply call
7271 			 * pvh_lock() which may block.  Instead, we first try to acquire
7272 			 * the lock without waiting, which in most cases should succeed.
7273 			 * If it fails, we submit the pending batched operations to re-
7274 			 * enable preemption and then acquire the lock normally.
7275 			 */
7276 			local_locked_pvh = pvh_try_lock(pai);
7277 			if (__improbable(!pvh_try_lock_success(&local_locked_pvh))) {
7278 				pmap_multipage_op_submit(flush_range);
7279 				local_locked_pvh = pvh_lock(pai);
7280 			}
7281 		} else {
7282 			local_locked_pvh = pvh_lock(pai);
7283 		}
7284 	} else {
7285 		local_locked_pvh = *locked_pvh;
7286 		assert(pai == local_locked_pvh.pai);
7287 	}
7288 	assert(local_locked_pvh.pvh != 0);
7289 	pvh_assert_locked(pai);
7290 
7291 	pte_p = PT_ENTRY_NULL;
7292 	pve_p = PV_ENTRY_NULL;
7293 	if (pvh_test_type(local_locked_pvh.pvh, PVH_TYPE_PTEP)) {
7294 		pte_p = pvh_ptep(local_locked_pvh.pvh);
7295 	} else if (pvh_test_type(local_locked_pvh.pvh, PVH_TYPE_PVEP)) {
7296 		pve_p = pvh_pve_list(local_locked_pvh.pvh);
7297 	} else if (__improbable(!pvh_test_type(local_locked_pvh.pvh, PVH_TYPE_NULL))) {
7298 		panic("%s: invalid PV head 0x%llx for PA 0x%llx", __func__, (uint64_t)local_locked_pvh.pvh, (uint64_t)phys);
7299 	}
7300 
7301 	const bool is_reusable = ppattr_test_reusable(pai);
7302 
7303 	bool pvh_lock_sleep_mode_needed = false;
7304 	pmap_sptm_percpu_data_t *sptm_pcpu = NULL;
7305 	sptm_disjoint_op_t *sptm_ops = NULL;
7306 
7307 	/**
7308 	 * This would also work as a block, with the above variables declared using the
7309 	 * __block qualifier, but the extra runtime overhead of block syntax (e.g.
7310 	 * dereferencing __block variables through stack forwarding pointers) isn't needed
7311 	 * here, as we never need to use this code sequence as a closure.
7312 	 */
7313 	#define FFF_PERCPU_INIT() do { \
7314 	        disable_preemption(); \
7315 	        sptm_pcpu = PERCPU_GET(pmap_sptm_percpu); \
7316 	        sptm_ops = sptm_pcpu->sptm_ops; \
7317 	} while (0)
7318 
7319 	FFF_PERCPU_INIT();
7320 
7321 	int pve_ptep_idx = 0;
7322 
7323 	/**
7324 	 * With regard to TLBI, there are three cases:
7325 	 *
7326 	 * 1. PMAP_OPTIONS_NOFLUSH is specified. In such case, SPTM doesn't need to flush TLB and neither does pmap.
7327 	 * 2. PMAP_OPTIONS_NOFLUSH is not specified, but flush_range is, indicating the caller intends to flush TLB
7328 	 *    itself (with range TLBI). In such case, we check the flush_range limits and only issue the TLBI if a
7329 	 *    mapping is out of the range.
7330 	 * 3. Neither PMAP_OPTIONS_NOFLUSH nor a valid flush_range pointer is specified. In such case, we should just
7331 	 *    let SPTM handle TLBI flushing.
7332 	 */
7333 	const bool defer_tlbi = (options & PMAP_OPTIONS_NOFLUSH) || flush_range;
7334 	const uint32_t sptm_update_options = SPTM_UPDATE_PERMS_AND_WAS_WRITABLE | SPTM_UPDATE_AF | (defer_tlbi ? SPTM_UPDATE_DEFER_TLBI : 0);
7335 
7336 	while ((pve_p != PV_ENTRY_NULL) || (pte_p != PT_ENTRY_NULL)) {
7337 		pt_entry_t       spte;
7338 		pt_entry_t       tmplate;
7339 
7340 		if (__improbable(pvh_lock_sleep_mode_needed)) {
7341 			assert((num_mappings == 0) && (num_skipped_mappings == 0));
7342 			/**
7343 			 * Undo the explicit preemption disable done in the last call to FFF_PER_CPU_INIT().
7344 			 * If the PVH lock is placed in sleep mode, we can't rely on it to disable preemption,
7345 			 * so we need these explicit preemption twiddles to ensure we don't get migrated off-
7346 			 * core while processing SPTM per-CPU data.  At the same time, we also want preemption
7347 			 * to briefly be re-enabled every SPTM_MAPPING_LIMIT mappings so that any pending
7348 			 * urgent ASTs can be handled.
7349 			 */
7350 			enable_preemption();
7351 			pvh_lock_enter_sleep_mode(&local_locked_pvh);
7352 			pvh_lock_sleep_mode_needed = false;
7353 			FFF_PERCPU_INIT();
7354 		}
7355 
7356 		if (pve_p != PV_ENTRY_NULL) {
7357 			pte_p = pve_get_ptep(pve_p, pve_ptep_idx);
7358 			if (pte_p == PT_ENTRY_NULL) {
7359 				goto fff_skip_pve;
7360 			}
7361 		}
7362 
7363 #ifdef PVH_FLAG_IOMMU
7364 		if (pvh_ptep_is_iommu(pte_p)) {
7365 			++num_skipped_mappings;
7366 			goto fff_skip_pve;
7367 		}
7368 #endif
7369 		spte = os_atomic_load(pte_p, relaxed);
7370 		if (pte_is_compressed(spte, pte_p)) {
7371 			panic("pte is COMPRESSED: pte_p=%p ppnum=0x%x", pte_p, ppnum);
7372 		}
7373 
7374 		pt_desc_t *ptdp = NULL;
7375 		pmap_t pmap = NULL;
7376 		vm_map_address_t va = 0;
7377 
7378 		if ((flush_range != NULL) && (pte_p == flush_range->current_ptep)) {
7379 			/**
7380 			 * If the current mapping matches the flush range's current iteration position,
7381 			 * there's no need to do the work of getting the PTD.  We already know the pmap,
7382 			 * and the VA is implied by flush_range->pending_region_start.
7383 			 */
7384 			pmap = flush_range->ptfr_pmap;
7385 		} else {
7386 			ptdp = ptep_get_ptd(pte_p);
7387 			pmap = ptdp->pmap;
7388 			va = ptd_get_va(ptdp, pte_p);
7389 			assert(va >= pmap->min && va < pmap->max);
7390 		}
7391 
7392 		bool skip_pte = pte_is_wired(spte) &&
7393 		    ((options & PMAP_OPTIONS_FF_WIRED) == 0);
7394 
7395 		if (skip_pte) {
7396 			result = FALSE;
7397 		}
7398 
7399 		// A concurrent pmap_remove() may have cleared the PTE
7400 		if (__improbable((spte & ARM_PTE_TYPE_MASK) == ARM_PTE_TYPE_FAULT)) {
7401 			skip_pte = true;
7402 		}
7403 
7404 		/**
7405 		 * If the PTD is NULL, we're adding the current mapping to the pending region templates instead of the
7406 		 * pending disjoint ops, so we don't need to do flush range disjoint op management.
7407 		 */
7408 		if ((flush_range != NULL) && (ptdp != NULL) && !skip_pte) {
7409 			/**
7410 			 * Insert a "header" entry for this physical page into the SPTM disjoint ops array.
7411 			 * We do this in three cases:
7412 			 * 1) We're at the beginning of the SPTM ops array (num_mappings == 0, flush_range->pending_disjoint_entries == 0).
7413 			 * 2) We may not be at the beginning of the SPTM ops array, but we are about to add the first operation
7414 			 *    for this physical page (num_mappings == 0, flush_range->pending_disjoint_entries == ?).
7415 			 * 3) We need to change the options passed to the SPTM for a run of one or more mappings.  Specifically,
7416 			 *    if we encounter a run of mappings that reside outside the VA region of our flush_range, or that
7417 			 *    belong to a pmap other than the one targeted by our flush_range, we should ask the SPTM to flush
7418 			 *    the TLB for us (i.e., clear SPTM_UPDATE_DEFER_TLBI), but only for those specific mappings.
7419 			 */
7420 			uint32_t per_mapping_sptm_update_options = sptm_update_options;
7421 			if ((flush_range->ptfr_pmap != pmap) || (va >= flush_range->ptfr_end) || (va < flush_range->ptfr_start)) {
7422 				per_mapping_sptm_update_options &= ~SPTM_UPDATE_DEFER_TLBI;
7423 			}
7424 			if ((num_mappings == 0) ||
7425 			    (flush_range->current_header->per_paddr_header.options != per_mapping_sptm_update_options)) {
7426 				if (pmap_multipage_op_add_page(phys, &num_mappings, per_mapping_sptm_update_options, flush_range)) {
7427 					/**
7428 					 * If we needed to submit the pending disjoint ops to make room for the new page,
7429 					 * flush any pending region ops to reenable preemption and restart the loop with
7430 					 * the lock in sleep mode.  This prevents preemption from being held disabled
7431 					 * for an arbitrary amount of time in the pathological case in which we have
7432 					 * both pending region ops and an excessively long PV list that repeatedly
7433 					 * requires new page headers with SPTM_MAPPING_LIMIT - 1 entries already pending.
7434 					 */
7435 					pmap_multipage_op_submit_region(flush_range);
7436 					assert(num_mappings == 0);
7437 					num_skipped_mappings = 0;
7438 					pvh_lock_sleep_mode_needed = true;
7439 					continue;
7440 				}
7441 			}
7442 		}
7443 
7444 		const pt_attr_t * const pt_attr = pmap_get_pt_attr(pmap);
7445 
7446 		/* update pmap stats and ledgers */
7447 		const bool is_internal = ppattr_pve_is_internal(pai, pve_p, pve_ptep_idx);
7448 		const bool is_altacct = ppattr_pve_is_altacct(pai, pve_p, pve_ptep_idx);
7449 		if (is_altacct) {
7450 			/*
7451 			 * We do not track "reusable" status for
7452 			 * "alternate accounting" mappings.
7453 			 */
7454 		} else if ((options & PMAP_OPTIONS_CLEAR_REUSABLE) &&
7455 		    is_reusable &&
7456 		    is_internal &&
7457 		    pmap != kernel_pmap) {
7458 			/* one less "reusable" */
7459 			pmap_ledger_debit(pmap, task_ledgers.reusable, pt_attr_page_size(pt_attr) * PAGE_RATIO);
7460 			/* one more "internal" */
7461 			pmap_ledger_credit(pmap, task_ledgers.internal, pt_attr_page_size(pt_attr) * PAGE_RATIO);
7462 			pmap_ledger_credit(pmap, task_ledgers.phys_footprint, pt_attr_page_size(pt_attr) * PAGE_RATIO);
7463 
7464 			/*
7465 			 * Since the page is being marked non-reusable, we assume that it will be
7466 			 * modified soon.  Avoid the cost of another trap to handle the fast
7467 			 * fault when we next write to this page.
7468 			 */
7469 			clear_write_fault = true;
7470 		} else if ((options & PMAP_OPTIONS_SET_REUSABLE) &&
7471 		    !is_reusable &&
7472 		    is_internal &&
7473 		    pmap != kernel_pmap) {
7474 			/* one more "reusable" */
7475 			pmap_ledger_credit(pmap, task_ledgers.reusable, pt_attr_page_size(pt_attr) * PAGE_RATIO);
7476 			pmap_ledger_debit(pmap, task_ledgers.internal, pt_attr_page_size(pt_attr) * PAGE_RATIO);
7477 			pmap_ledger_debit(pmap, task_ledgers.phys_footprint, pt_attr_page_size(pt_attr) * PAGE_RATIO);
7478 		}
7479 
7480 		if (skip_pte) {
7481 			++num_skipped_mappings;
7482 			goto fff_skip_pve;
7483 		}
7484 
7485 		tmplate = spte;
7486 
7487 		if ((allow_mode & VM_PROT_READ) != VM_PROT_READ) {
7488 			/* read protection sets the pte to fault */
7489 			tmplate =  tmplate & ~ARM_PTE_AF;
7490 			ref_fault = true;
7491 		}
7492 		if ((allow_mode & VM_PROT_WRITE) != VM_PROT_WRITE) {
7493 			/* take away write permission if set */
7494 			if (pmap == kernel_pmap) {
7495 				if ((tmplate & ARM_PTE_APMASK) == ARM_PTE_AP(AP_RWNA)) {
7496 					tmplate = ((tmplate & ~ARM_PTE_APMASK) | ARM_PTE_AP(AP_RONA));
7497 					pte_set_was_writeable(tmplate, true);
7498 					mod_fault = true;
7499 				}
7500 			} else {
7501 				if ((tmplate & ARM_PTE_APMASK) == pt_attr_leaf_rw(pt_attr)) {
7502 					tmplate = ((tmplate & ~ARM_PTE_APMASK) | pt_attr_leaf_ro(pt_attr));
7503 					pte_set_was_writeable(tmplate, true);
7504 					mod_fault = true;
7505 				}
7506 			}
7507 		}
7508 
7509 		if (ptdp != NULL) {
7510 			sptm_ops[num_mappings].root_pt_paddr = pmap->ttep;
7511 			sptm_ops[num_mappings].vaddr = va;
7512 			sptm_ops[num_mappings].pte_template = tmplate;
7513 			++num_mappings;
7514 		} else if (pmap_insert_flush_range_template(tmplate, flush_range)) {
7515 			/**
7516 			 * We submit both the pending disjoint and pending region ops whenever
7517 			 * either category reaches the mapping limit.  Having pending operations
7518 			 * in either category will keep preemption disabled, and we want to ensure
7519 			 * that we can at least temporarily re-enable preemption roughly every
7520 			 * SPTM_MAPPING_LIMIT mappings.
7521 			 */
7522 			pmap_multipage_op_submit_disjoint(num_mappings, flush_range);
7523 			pvh_lock_sleep_mode_needed = true;
7524 			num_mappings = num_skipped_mappings = 0;
7525 		}
7526 fff_skip_pve:
7527 		if ((num_mappings + num_skipped_mappings) >= SPTM_MAPPING_LIMIT) {
7528 			if (flush_range != NULL) {
7529 				/* See comment above for why we submit both disjoint and region ops when we hit the limit. */
7530 				pmap_multipage_op_submit_disjoint(num_mappings, flush_range);
7531 				pmap_multipage_op_submit_region(flush_range);
7532 			} else if (num_mappings > 0) {
7533 				sptm_update_disjoint(phys, sptm_pcpu->sptm_ops_pa, num_mappings, sptm_update_options);
7534 			}
7535 			pvh_lock_sleep_mode_needed = true;
7536 			num_mappings = num_skipped_mappings = 0;
7537 		}
7538 		pte_p = PT_ENTRY_NULL;
7539 		if ((pve_p != PV_ENTRY_NULL) && (++pve_ptep_idx == PTE_PER_PVE)) {
7540 			pve_ptep_idx = 0;
7541 			pve_p = pve_next(pve_p);
7542 		}
7543 	}
7544 
7545 	if (num_mappings != 0) {
7546 		sptm_return_t sptm_ret;
7547 
7548 		if (flush_range == NULL) {
7549 			sptm_ret = sptm_update_disjoint(phys, sptm_pcpu->sptm_ops_pa, num_mappings, sptm_update_options);
7550 		} else {
7551 			/* Resync the pending mapping state in flush_range with our local state. */
7552 			assert(num_mappings >= flush_range->pending_disjoint_entries);
7553 			flush_range->pending_disjoint_entries = num_mappings;
7554 		}
7555 	}
7556 
7557 	/**
7558 	 * Undo the explicit disable_preemption() done in FFF_PERCPU_INIT().
7559 	 * Note that enable_preemption() decrements a per-thread counter, so if
7560 	 * we happen to still hold the PVH lock in spin mode then preemption won't
7561 	 * actually be re-enabled until we drop the lock (which also decrements
7562 	 * the per-thread counter.
7563 	 */
7564 	enable_preemption();
7565 
7566 	/*
7567 	 * If we are using the same approach for ref and mod
7568 	 * faults on this PTE, do not clear the write fault;
7569 	 * this would cause both ref and mod to be set on the
7570 	 * page again, and prevent us from taking ANY read/write
7571 	 * fault on the mapping.
7572 	 */
7573 	if (clear_write_fault && !ref_aliases_mod) {
7574 		arm_clear_fast_fault(ppnum, VM_PROT_WRITE, local_locked_pvh.pvh, PT_ENTRY_NULL, 0);
7575 	}
7576 
7577 	pp_attr_t attrs_to_clear = (result ? bits_to_clear : 0);
7578 	pp_attr_t attrs_to_set = 0;
7579 	/* update global "reusable" status for this page */
7580 	if ((options & PMAP_OPTIONS_CLEAR_REUSABLE) && is_reusable) {
7581 		attrs_to_clear |= PP_ATTR_REUSABLE;
7582 	} else if ((options & PMAP_OPTIONS_SET_REUSABLE) && !is_reusable) {
7583 		attrs_to_set |= PP_ATTR_REUSABLE;
7584 	}
7585 
7586 	if (mod_fault) {
7587 		attrs_to_set |= PP_ATTR_MODFAULT;
7588 	}
7589 	if (ref_fault) {
7590 		attrs_to_set |= PP_ATTR_REFFAULT;
7591 	}
7592 
7593 	if (attrs_to_set | attrs_to_clear) {
7594 		ppattr_modify_bits(pai, attrs_to_clear, attrs_to_set);
7595 	}
7596 
7597 	if (__probable(locked_pvh == NULL)) {
7598 		pvh_unlock(&local_locked_pvh);
7599 	} else {
7600 		*locked_pvh = local_locked_pvh;
7601 	}
7602 	if ((flush_range != NULL) && !preemption_enabled()) {
7603 		flush_range->processed_entries += num_skipped_mappings;
7604 	}
7605 	return result;
7606 }
7607 
7608 MARK_AS_PMAP_TEXT boolean_t
7609 arm_force_fast_fault_internal(
7610 	ppnum_t         ppnum,
7611 	vm_prot_t       allow_mode,
7612 	int             options)
7613 {
7614 	if (__improbable((options & (PMAP_OPTIONS_FF_LOCKED | PMAP_OPTIONS_FF_WIRED | PMAP_OPTIONS_NOFLUSH)) != 0)) {
7615 		panic("arm_force_fast_fault(0x%x, 0x%x, 0x%x): invalid options", ppnum, allow_mode, options);
7616 	}
7617 	return arm_force_fast_fault_with_flush_range(ppnum, allow_mode, options, NULL, 0, NULL);
7618 }
7619 
7620 /*
7621  *	Routine:	arm_force_fast_fault
7622  *
7623  *	Function:
7624  *		Force all mappings for this page to fault according
7625  *		to the access modes allowed, so we can gather ref/modify
7626  *		bits again.
7627  */
7628 
7629 boolean_t
7630 arm_force_fast_fault(
7631 	ppnum_t         ppnum,
7632 	vm_prot_t       allow_mode,
7633 	int             options,
7634 	__unused void   *arg)
7635 {
7636 	pmap_paddr_t    phys = ptoa(ppnum);
7637 
7638 	assert(ppnum != vm_page_fictitious_addr);
7639 
7640 	if (!pa_valid(phys)) {
7641 		return FALSE;   /* Not a managed page. */
7642 	}
7643 
7644 	return arm_force_fast_fault_internal(ppnum, allow_mode, options);
7645 }
7646 
7647 /**
7648  * Clear pending force fault for at most SPTM_MAPPING_LIMIT mappings for this
7649  * page based on the observed fault type, and update the appropriate ref/modify
7650  * bits for the physical page. This typically involves adding write permissions
7651  * back for write faults and setting the Access Flag for both read/write faults
7652  * (since the lack of those things is what caused the fault in the first place).
7653  *
7654  * @note Only SPTM_MAPPING_LIMIT number of mappings can be modified in a single
7655  *       arm_clear_fast_fault() call to prevent excessive PVH lock contention as
7656  *       the PVH lock should be held for `ppnum` already. If a fault is
7657  *       subsequently taken on a mapping we haven't processed, arm_fast_fault()
7658  *       will call this function with a non-NULL pte_p to perform a targeted
7659  *       fixup.
7660  *
7661  * @param ppnum Page number of the page to clear a pending force fault on.
7662  * @param fault_type The type of access/fault that triggered us wanting to clear
7663  *                   the pending force fault status. This determines how we
7664  *                   modify the PTE to not cause a fault in the future and also
7665  *                   whether we mark the PTE as referenced or modified.
7666  *                   Typically a write fault would cause the page to be marked
7667  *                   as referenced and modified, and a read fault would only
7668  *                   cause the page to be marked as referenced.
7669  * @param pvh pv_head_table entry value for [ppnum] returned by a previous call
7670  *            to pvh_lock().
7671  * @param pte_p If this value is non-PT_ENTRY_NULL then only this specified PTE
7672  *              will be modified. If it is PT_ENTRY_NULL, then every mapping to
7673  *              `ppnum` will be modified.
7674  * @param attrs_to_clear Mask of additional pp_attr_t bits to clear for the physical
7675  *                       page upon completion of this function.  This is typically
7676  *                       some combination of the REFFAULT and MODFAULT bits.
7677  *
7678  * @return TRUE if any PTEs were modified, FALSE otherwise.
7679  */
7680 MARK_AS_PMAP_TEXT static boolean_t
7681 arm_clear_fast_fault(
7682 	ppnum_t ppnum,
7683 	vm_prot_t fault_type,
7684 	uintptr_t pvh,
7685 	pt_entry_t *pte_p,
7686 	pp_attr_t attrs_to_clear)
7687 {
7688 	const pmap_paddr_t pa = ptoa(ppnum);
7689 	pv_entry_t     *pve_p;
7690 	boolean_t       result;
7691 	unsigned int    num_mappings = 0, num_skipped_mappings = 0;
7692 	pp_attr_t       attrs_to_set = 0;
7693 
7694 	assert(ppnum != vm_page_fictitious_addr);
7695 
7696 	if (!pa_valid(pa)) {
7697 		return FALSE;   /* Not a managed page. */
7698 	}
7699 
7700 	result = FALSE;
7701 	pve_p = PV_ENTRY_NULL;
7702 	if (pte_p == PT_ENTRY_NULL) {
7703 		if (pvh_test_type(pvh, PVH_TYPE_PTEP)) {
7704 			pte_p = pvh_ptep(pvh);
7705 		} else if (pvh_test_type(pvh, PVH_TYPE_PVEP)) {
7706 			pve_p = pvh_pve_list(pvh);
7707 		} else if (__improbable(!pvh_test_type(pvh, PVH_TYPE_NULL))) {
7708 			panic("%s: invalid PV head 0x%llx for PA 0x%llx", __func__, (uint64_t)pvh, (uint64_t)pa);
7709 		}
7710 	}
7711 
7712 	disable_preemption();
7713 	pmap_sptm_percpu_data_t *sptm_pcpu = PERCPU_GET(pmap_sptm_percpu);
7714 	sptm_disjoint_op_t *sptm_ops = sptm_pcpu->sptm_ops;
7715 
7716 	int pve_ptep_idx = 0;
7717 
7718 	while ((pve_p != PV_ENTRY_NULL) || (pte_p != PT_ENTRY_NULL)) {
7719 		pt_entry_t spte;
7720 		pt_entry_t tmplate;
7721 
7722 		if (pve_p != PV_ENTRY_NULL) {
7723 			pte_p = pve_get_ptep(pve_p, pve_ptep_idx);
7724 			if (pte_p == PT_ENTRY_NULL) {
7725 				goto cff_skip_pve;
7726 			}
7727 		}
7728 
7729 #ifdef PVH_FLAG_IOMMU
7730 		if (pvh_ptep_is_iommu(pte_p)) {
7731 			++num_skipped_mappings;
7732 			goto cff_skip_pve;
7733 		}
7734 #endif
7735 		spte = os_atomic_load(pte_p, relaxed);
7736 		// A concurrent pmap_remove() may have cleared the PTE
7737 		if (__improbable((spte & ARM_PTE_TYPE_MASK) == ARM_PTE_TYPE_FAULT)) {
7738 			++num_skipped_mappings;
7739 			goto cff_skip_pve;
7740 		}
7741 
7742 		const pt_desc_t * const ptdp = ptep_get_ptd(pte_p);
7743 		const pmap_t pmap = ptdp->pmap;
7744 		const vm_map_address_t va = ptd_get_va(ptdp, pte_p);
7745 
7746 		assert(va >= pmap->min && va < pmap->max);
7747 
7748 		tmplate = spte;
7749 
7750 		if ((fault_type & VM_PROT_WRITE) && (pte_was_writeable(spte))) {
7751 			{
7752 				if (pmap == kernel_pmap) {
7753 					tmplate = ((spte & ~ARM_PTE_APMASK) | ARM_PTE_AP(AP_RWNA));
7754 				} else {
7755 					assert(pmap->type != PMAP_TYPE_NESTED);
7756 					tmplate = ((spte & ~ARM_PTE_APMASK) | pt_attr_leaf_rw(pmap_get_pt_attr(pmap)));
7757 				}
7758 			}
7759 
7760 			tmplate |= ARM_PTE_AF;
7761 
7762 			pte_set_was_writeable(tmplate, false);
7763 			attrs_to_set |= (PP_ATTR_REFERENCED | PP_ATTR_MODIFIED);
7764 		} else if ((fault_type & VM_PROT_READ) && ((spte & ARM_PTE_AF) != ARM_PTE_AF)) {
7765 			tmplate = spte | ARM_PTE_AF;
7766 
7767 			{
7768 				attrs_to_set |= PP_ATTR_REFERENCED;
7769 			}
7770 		}
7771 
7772 		assert(spte != ARM_PTE_TYPE_FAULT);
7773 
7774 		if (spte != tmplate) {
7775 			sptm_ops[num_mappings].root_pt_paddr = pmap->ttep;
7776 			sptm_ops[num_mappings].vaddr = va;
7777 			sptm_ops[num_mappings].pte_template = tmplate;
7778 			++num_mappings;
7779 			result = TRUE;
7780 		}
7781 
7782 cff_skip_pve:
7783 		if ((num_mappings + num_skipped_mappings) == SPTM_MAPPING_LIMIT) {
7784 			if (num_mappings != 0) {
7785 				sptm_update_disjoint(pa, sptm_pcpu->sptm_ops_pa, num_mappings,
7786 				    SPTM_UPDATE_PERMS_AND_WAS_WRITABLE | SPTM_UPDATE_AF);
7787 				num_mappings = 0;
7788 			}
7789 			/*
7790 			 * We've reached the limit of mappings that can be processed in a single arm_clear_fast_fault()
7791 			 * call.  Bail out here to avoid excessive PVH lock duration on the fault path.  If a fault is
7792 			 * subsequently taken on a mapping we haven't processed, arm_fast_fault() will call this
7793 			 * function with a non-NULL pte_p to perform a targeted fixup.
7794 			 */
7795 			break;
7796 		}
7797 
7798 		pte_p = PT_ENTRY_NULL;
7799 		if ((pve_p != PV_ENTRY_NULL) && (++pve_ptep_idx == PTE_PER_PVE)) {
7800 			pve_ptep_idx = 0;
7801 			pve_p = pve_next(pve_p);
7802 		}
7803 	}
7804 
7805 	if (num_mappings != 0) {
7806 		assert(result == TRUE);
7807 		sptm_update_disjoint(pa, sptm_pcpu->sptm_ops_pa, num_mappings,
7808 		    SPTM_UPDATE_PERMS_AND_WAS_WRITABLE | SPTM_UPDATE_AF);
7809 	}
7810 
7811 	if (attrs_to_set | attrs_to_clear) {
7812 		ppattr_modify_bits(pa_index(pa), attrs_to_clear, attrs_to_set);
7813 	}
7814 	enable_preemption();
7815 
7816 	return result;
7817 }
7818 
7819 /*
7820  * Determine if the fault was induced by software tracking of
7821  * modify/reference bits.  If so, re-enable the mapping (and set
7822  * the appropriate bits).
7823  *
7824  * Returns KERN_SUCCESS if the fault was induced and was
7825  * successfully handled.
7826  *
7827  * Returns KERN_FAILURE if the fault was not induced and
7828  * the function was unable to deal with it.
7829  *
7830  * Returns KERN_PROTECTION_FAILURE if the pmap layer explictly
7831  * disallows this type of access.
7832  */
7833 MARK_AS_PMAP_TEXT kern_return_t
7834 arm_fast_fault_internal(
7835 	pmap_t pmap,
7836 	vm_map_address_t va,
7837 	vm_prot_t fault_type,
7838 	__unused bool was_af_fault,
7839 	__unused bool from_user)
7840 {
7841 	kern_return_t   result = KERN_FAILURE;
7842 	pt_entry_t     *ptep;
7843 	pt_entry_t      spte = ARM_PTE_TYPE_FAULT;
7844 	locked_pvh_t    locked_pvh = {.pvh = 0};
7845 	unsigned int    pai;
7846 	pmap_paddr_t    pa;
7847 	validate_pmap_mutable(pmap);
7848 
7849 	if (__probable(preemption_enabled())) {
7850 		pmap_lock(pmap, PMAP_LOCK_SHARED);
7851 	} else if (__improbable(!pmap_try_lock(pmap, PMAP_LOCK_SHARED))) {
7852 		/**
7853 		 * In certain cases, arm_fast_fault() may be invoked with preemption disabled
7854 		 * on the copyio path.  In theses cases the (in-kernel) caller expects that any
7855 		 * faults taken against the user address may not be handled successfully
7856 		 * (vm_fault() allows non-preemptible callers with the possibility that the
7857 		 * fault may not be successfully handled) and will result in the copyio operation
7858 		 * returning EFAULT.  It is then the caller's responsibility to retry the copyio
7859 		 * operation in a preemptible context.
7860 		 *
7861 		 * For these cases attempting to acquire the sleepable lock will panic, so
7862 		 * we simply make a best effort and return failure just as the VM does if we
7863 		 * can't acquire the lock without sleeping.
7864 		 */
7865 		return result;
7866 	}
7867 
7868 	/*
7869 	 * If the entry doesn't exist, is completely invalid, or is already
7870 	 * valid, we can't fix it here.
7871 	 */
7872 
7873 	const uint64_t pmap_page_size = pt_attr_page_size(pmap_get_pt_attr(pmap)) * PAGE_RATIO;
7874 	ptep = pmap_pte(pmap, va & ~(pmap_page_size - 1));
7875 	if (ptep != PT_ENTRY_NULL) {
7876 		while (true) {
7877 			spte = os_atomic_load(ptep, relaxed);
7878 
7879 			pa = pte_to_pa(spte);
7880 
7881 			if ((spte == ARM_PTE_TYPE_FAULT) ||
7882 			    pte_is_compressed(spte, ptep)) {
7883 				pmap_unlock(pmap, PMAP_LOCK_SHARED);
7884 				return result;
7885 			}
7886 
7887 			if (!pa_valid(pa)) {
7888 				const sptm_frame_type_t frame_type = sptm_get_frame_type(pa);
7889 				if (frame_type == XNU_PROTECTED_IO) {
7890 					result = KERN_PROTECTION_FAILURE;
7891 				}
7892 				pmap_unlock(pmap, PMAP_LOCK_SHARED);
7893 				return result;
7894 			}
7895 			pai = pa_index(pa);
7896 			/**
7897 			 * Check for preemption disablement and in that case use pvh_try_lock()
7898 			 * for the same reason we use pmap_try_lock() above.
7899 			 */
7900 			if (__probable(preemption_enabled())) {
7901 				locked_pvh = pvh_lock(pai);
7902 			} else {
7903 				locked_pvh = pvh_try_lock(pai);
7904 				if (__improbable(!pvh_try_lock_success(&locked_pvh))) {
7905 					pmap_unlock(pmap, PMAP_LOCK_SHARED);
7906 					return result;
7907 				}
7908 			}
7909 			assert(locked_pvh.pvh != 0);
7910 			if (os_atomic_load(ptep, relaxed) == spte) {
7911 				/*
7912 				 * Double-check the spte value, as we care about the AF bit.
7913 				 * It's also possible that pmap_page_protect() transitioned the
7914 				 * PTE to compressed/empty before we grabbed the PVH lock.
7915 				 */
7916 				break;
7917 			}
7918 			pvh_unlock(&locked_pvh);
7919 		}
7920 	} else {
7921 		pmap_unlock(pmap, PMAP_LOCK_SHARED);
7922 		return result;
7923 	}
7924 
7925 
7926 	if (result == KERN_SUCCESS) {
7927 		goto ff_cleanup;
7928 	}
7929 
7930 	pp_attr_t attrs = os_atomic_load(&pp_attr_table[pai], relaxed);
7931 	if ((attrs & PP_ATTR_REFFAULT) || ((fault_type & VM_PROT_WRITE) && (attrs & PP_ATTR_MODFAULT))) {
7932 		/*
7933 		 * An attempted access will always clear ref/mod fault state, as
7934 		 * appropriate for the fault type.  arm_clear_fast_fault will
7935 		 * update the associated PTEs for the page as appropriate; if
7936 		 * any PTEs are updated, we redrive the access.  If the mapping
7937 		 * does not actually allow for the attempted access, the
7938 		 * following fault will (hopefully) fail to update any PTEs, and
7939 		 * thus cause arm_fast_fault to decide that it failed to handle
7940 		 * the fault.
7941 		 */
7942 		pp_attr_t attrs_to_clear = 0;
7943 		if (attrs & PP_ATTR_REFFAULT) {
7944 			attrs_to_clear |= PP_ATTR_REFFAULT;
7945 		}
7946 		if ((fault_type & VM_PROT_WRITE) && (attrs & PP_ATTR_MODFAULT)) {
7947 			attrs_to_clear |= PP_ATTR_MODFAULT;
7948 		}
7949 
7950 		if (arm_clear_fast_fault((ppnum_t)atop(pa), fault_type, locked_pvh.pvh, PT_ENTRY_NULL, attrs_to_clear)) {
7951 			/*
7952 			 * Should this preserve KERN_PROTECTION_FAILURE?  The
7953 			 * cost of not doing so is a another fault in a case
7954 			 * that should already result in an exception.
7955 			 */
7956 			result = KERN_SUCCESS;
7957 		}
7958 	}
7959 
7960 	/*
7961 	 * If the PTE already has sufficient permissions, we can report the fault as handled.
7962 	 * This may happen, for example, if multiple threads trigger roughly simultaneous faults
7963 	 * on mappings of the same page
7964 	 */
7965 	if ((result == KERN_FAILURE) && (spte & ARM_PTE_AF)) {
7966 		uintptr_t ap_ro, ap_rw, ap_x;
7967 		if (pmap == kernel_pmap) {
7968 			ap_ro = ARM_PTE_AP(AP_RONA);
7969 			ap_rw = ARM_PTE_AP(AP_RWNA);
7970 			ap_x = ARM_PTE_NX;
7971 		} else {
7972 			ap_ro = pt_attr_leaf_ro(pmap_get_pt_attr(pmap));
7973 			ap_rw = pt_attr_leaf_rw(pmap_get_pt_attr(pmap));
7974 			ap_x = pt_attr_leaf_x(pmap_get_pt_attr(pmap));
7975 		}
7976 		/*
7977 		 * NOTE: this doesn't currently handle user-XO mappings. Depending upon the
7978 		 * hardware they may be xPRR-protected, in which case they'll be handled
7979 		 * by the is_pte_xprr_protected() case above.  Additionally, the exception
7980 		 * handling path currently does not call arm_fast_fault() without at least
7981 		 * VM_PROT_READ in fault_type.
7982 		 */
7983 		if (((spte & ARM_PTE_APMASK) == ap_rw) ||
7984 		    (!(fault_type & VM_PROT_WRITE) && ((spte & ARM_PTE_APMASK) == ap_ro))) {
7985 			if (!(fault_type & VM_PROT_EXECUTE) || ((spte & ARM_PTE_XMASK) == ap_x)) {
7986 				result = KERN_SUCCESS;
7987 			}
7988 		}
7989 	}
7990 
7991 	if ((result == KERN_FAILURE) && arm_clear_fast_fault((ppnum_t)atop(pa), fault_type, locked_pvh.pvh, ptep, 0)) {
7992 		/*
7993 		 * A prior arm_clear_fast_fault() operation may have returned early due to
7994 		 * another pending PV list operation or an excessively large PV list.
7995 		 * Attempt a targeted fixup of the PTE that caused the fault to avoid repeatedly
7996 		 * taking a fault on the same mapping.
7997 		 */
7998 		result = KERN_SUCCESS;
7999 	}
8000 
8001 ff_cleanup:
8002 
8003 	pvh_unlock(&locked_pvh);
8004 	pmap_unlock(pmap, PMAP_LOCK_SHARED);
8005 	return result;
8006 }
8007 
8008 kern_return_t
8009 arm_fast_fault(
8010 	pmap_t pmap,
8011 	vm_map_address_t va,
8012 	vm_prot_t fault_type,
8013 	bool was_af_fault,
8014 	__unused bool from_user)
8015 {
8016 	kern_return_t   result = KERN_FAILURE;
8017 
8018 	if (va < pmap->min || va >= pmap->max) {
8019 		return result;
8020 	}
8021 
8022 	PMAP_TRACE(3, PMAP_CODE(PMAP__FAST_FAULT) | DBG_FUNC_START,
8023 	    VM_KERNEL_ADDRHIDE(pmap), VM_KERNEL_ADDRHIDE(va), fault_type,
8024 	    from_user);
8025 
8026 
8027 	result = arm_fast_fault_internal(pmap, va, fault_type, was_af_fault, from_user);
8028 
8029 	PMAP_TRACE(3, PMAP_CODE(PMAP__FAST_FAULT) | DBG_FUNC_END, result);
8030 
8031 	return result;
8032 }
8033 
8034 void
8035 pmap_copy_page(
8036 	ppnum_t psrc,
8037 	ppnum_t pdst)
8038 {
8039 	bcopy_phys((addr64_t) (ptoa(psrc)),
8040 	    (addr64_t) (ptoa(pdst)),
8041 	    PAGE_SIZE);
8042 }
8043 
8044 
8045 /*
8046  *	pmap_copy_page copies the specified (machine independent) pages.
8047  */
8048 void
8049 pmap_copy_part_page(
8050 	ppnum_t psrc,
8051 	vm_offset_t src_offset,
8052 	ppnum_t pdst,
8053 	vm_offset_t dst_offset,
8054 	vm_size_t len)
8055 {
8056 	bcopy_phys((addr64_t) (ptoa(psrc) + src_offset),
8057 	    (addr64_t) (ptoa(pdst) + dst_offset),
8058 	    len);
8059 }
8060 
8061 
8062 /*
8063  *	pmap_zero_page zeros the specified (machine independent) page.
8064  */
8065 void
8066 pmap_zero_page(
8067 	ppnum_t pn)
8068 {
8069 	assert(pn != vm_page_fictitious_addr);
8070 	bzero_phys((addr64_t) ptoa(pn), PAGE_SIZE);
8071 }
8072 
8073 /*
8074  *	pmap_zero_part_page
8075  *	zeros the specified (machine independent) part of a page.
8076  */
8077 void
8078 pmap_zero_part_page(
8079 	ppnum_t pn,
8080 	vm_offset_t offset,
8081 	vm_size_t len)
8082 {
8083 	assert(pn != vm_page_fictitious_addr);
8084 	assert(offset + len <= PAGE_SIZE);
8085 	bzero_phys((addr64_t) (ptoa(pn) + offset), len);
8086 }
8087 
8088 void
8089 pmap_map_globals(
8090 	void)
8091 {
8092 	pt_entry_t      pte;
8093 
8094 	pte = pa_to_pte(kvtophys_nofail((vm_offset_t)&lowGlo)) | AP_RONA | ARM_PTE_NX | ARM_PTE_PNX | ARM_PTE_AF | ARM_PTE_TYPE;
8095 #if __ARM_KERNEL_PROTECT__
8096 	pte |= ARM_PTE_NG;
8097 #endif /* __ARM_KERNEL_PROTECT__ */
8098 	pte |= ARM_PTE_ATTRINDX(CACHE_ATTRINDX_WRITEBACK);
8099 	pte |= ARM_PTE_SH(SH_OUTER_MEMORY);
8100 	sptm_map_page(kernel_pmap->ttep, LOWGLOBAL_ALIAS, pte);
8101 
8102 #if KASAN
8103 	kasan_notify_address(LOWGLOBAL_ALIAS, PAGE_SIZE);
8104 #endif
8105 }
8106 
8107 vm_offset_t
8108 pmap_cpu_windows_copy_addr(int cpu_num, unsigned int index)
8109 {
8110 	if (__improbable(index >= CPUWINDOWS_MAX)) {
8111 		panic("%s: invalid index %u", __func__, index);
8112 	}
8113 	return (vm_offset_t)(CPUWINDOWS_BASE + (PAGE_SIZE * ((CPUWINDOWS_MAX * cpu_num) + index)));
8114 }
8115 
8116 MARK_AS_PMAP_TEXT unsigned int
8117 pmap_map_cpu_windows_copy_internal(
8118 	ppnum_t pn,
8119 	vm_prot_t prot,
8120 	unsigned int wimg_bits)
8121 {
8122 	pt_entry_t      *ptep = NULL, pte;
8123 	pmap_cpu_data_t *pmap_cpu_data = pmap_get_cpu_data();
8124 	unsigned int    cpu_num;
8125 	unsigned int    cpu_window_index;
8126 	vm_offset_t     cpu_copywindow_vaddr = 0;
8127 	bool            need_strong_sync = false;
8128 
8129 	assert(get_preemption_level() > 0);
8130 	cpu_num = pmap_cpu_data->cpu_number;
8131 
8132 	for (cpu_window_index = 0; cpu_window_index < CPUWINDOWS_MAX; cpu_window_index++) {
8133 		cpu_copywindow_vaddr = pmap_cpu_windows_copy_addr(cpu_num, cpu_window_index);
8134 		ptep = pmap_pte(kernel_pmap, cpu_copywindow_vaddr);
8135 		assert(!pte_is_compressed(*ptep, ptep));
8136 		if (*ptep == ARM_PTE_TYPE_FAULT) {
8137 			break;
8138 		}
8139 	}
8140 	if (__improbable(cpu_window_index == CPUWINDOWS_MAX)) {
8141 		panic("%s: out of windows", __func__);
8142 	}
8143 
8144 	pte = pa_to_pte(ptoa(pn)) | ARM_PTE_TYPE | ARM_PTE_AF | ARM_PTE_NX | ARM_PTE_PNX;
8145 #if __ARM_KERNEL_PROTECT__
8146 	pte |= ARM_PTE_NG;
8147 #endif /* __ARM_KERNEL_PROTECT__ */
8148 	pte |= wimg_to_pte(wimg_bits, ptoa(pn));
8149 
8150 	if (prot & VM_PROT_WRITE) {
8151 		pte |= ARM_PTE_AP(AP_RWNA);
8152 	} else {
8153 		pte |= ARM_PTE_AP(AP_RONA);
8154 	}
8155 
8156 	/*
8157 	 * It's expected to be safe for an interrupt handler to nest copy-window usage with the
8158 	 * active thread on a CPU, as long as a sufficient number of copy windows are available.
8159 	 * --If the interrupt handler executes before the active thread creates the per-CPU mapping,
8160 	 *   or after the active thread completely removes the mapping, it may use the same mapping
8161 	 *   but will finish execution and tear down the mapping without the thread needing to know.
8162 	 * --If the interrupt handler executes after the active thread creates the per-CPU mapping,
8163 	 *   it will observe the valid mapping and use a different copy window.
8164 	 * --If the interrupt handler executes after the active thread clears the PTE in
8165 	 *   pmap_unmap_cpu_windows_copy() but before the active thread flushes the TLB, the code
8166 	 *   for computing cpu_window_index above will observe the PTE_INVALID_IN_FLIGHT token set
8167 	 *   by the SPTM, and will select a different index.
8168 	 */
8169 	const sptm_return_t sptm_status = sptm_map_page(kernel_pmap->ttep, cpu_copywindow_vaddr, pte);
8170 	if (__improbable(sptm_status != SPTM_SUCCESS)) {
8171 		panic("%s: failed to map CPU copy-window VA 0x%llx with SPTM status %d",
8172 		    __func__, (unsigned long long)cpu_copywindow_vaddr, sptm_status);
8173 	}
8174 
8175 	/*
8176 	 * Clean up any pending strong TLB flush for the same window in a thread we may have
8177 	 * interrupted.
8178 	 */
8179 	if (__improbable(pmap_cpu_data->copywindow_strong_sync[cpu_window_index])) {
8180 		arm64_sync_tlb(true);
8181 	}
8182 	pmap_cpu_data->copywindow_strong_sync[cpu_window_index] = need_strong_sync;
8183 
8184 	return cpu_window_index;
8185 }
8186 
8187 unsigned int
8188 pmap_map_cpu_windows_copy(
8189 	ppnum_t pn,
8190 	vm_prot_t prot,
8191 	unsigned int wimg_bits)
8192 {
8193 	return pmap_map_cpu_windows_copy_internal(pn, prot, wimg_bits);
8194 }
8195 
8196 MARK_AS_PMAP_TEXT void
8197 pmap_unmap_cpu_windows_copy_internal(
8198 	unsigned int index)
8199 {
8200 	unsigned int    cpu_num;
8201 	vm_offset_t     cpu_copywindow_vaddr = 0;
8202 	pmap_cpu_data_t *pmap_cpu_data = pmap_get_cpu_data();
8203 
8204 	assert(index < CPUWINDOWS_MAX);
8205 	assert(get_preemption_level() > 0);
8206 
8207 	cpu_num = pmap_cpu_data->cpu_number;
8208 
8209 	cpu_copywindow_vaddr = pmap_cpu_windows_copy_addr(cpu_num, index);
8210 	/* Issue full-system DSB to ensure prior operations on the per-CPU window
8211 	 * (which are likely to have been on I/O memory) are complete before
8212 	 * tearing down the mapping. */
8213 	__builtin_arm_dsb(DSB_SY);
8214 	sptm_unmap_region(kernel_pmap->ttep, cpu_copywindow_vaddr, 1, 0);
8215 	if (__improbable(pmap_cpu_data->copywindow_strong_sync[index])) {
8216 		arm64_sync_tlb(true);
8217 		pmap_cpu_data->copywindow_strong_sync[index] = false;
8218 	}
8219 }
8220 
8221 void
8222 pmap_unmap_cpu_windows_copy(
8223 	unsigned int index)
8224 {
8225 	return pmap_unmap_cpu_windows_copy_internal(index);
8226 }
8227 
8228 /*
8229  * Indicate that a pmap is intended to be used as a nested pmap
8230  * within one or more larger address spaces.  This must be set
8231  * before pmap_nest() is called with this pmap as the 'subordinate'.
8232  */
8233 MARK_AS_PMAP_TEXT void
8234 pmap_set_nested_internal(
8235 	pmap_t pmap)
8236 {
8237 	validate_pmap_mutable(pmap);
8238 	const pt_attr_t * const pt_attr = pmap_get_pt_attr(pmap);
8239 	if (__improbable(pmap->type != PMAP_TYPE_USER)) {
8240 		panic("%s: attempt to nest unsupported pmap %p of type 0x%hhx",
8241 		    __func__, pmap, pmap->type);
8242 	}
8243 	pmap->type = PMAP_TYPE_NESTED;
8244 	sptm_retype_params_t retype_params = {.raw = SPTM_RETYPE_PARAMS_NULL};
8245 	retype_params.attr_idx = (pt_attr_page_size(pt_attr) == 4096) ? SPTM_PT_GEOMETRY_4K : SPTM_PT_GEOMETRY_16K;
8246 	pmap_txm_acquire_exclusive_lock(pmap);
8247 	sptm_retype(pmap->ttep, XNU_USER_ROOT_TABLE, XNU_SHARED_ROOT_TABLE, retype_params);
8248 	pmap_txm_release_exclusive_lock(pmap);
8249 	pmap_get_pt_ops(pmap)->free_id(pmap);
8250 }
8251 
8252 void
8253 pmap_set_nested(
8254 	pmap_t pmap)
8255 {
8256 	pmap_set_nested_internal(pmap);
8257 }
8258 
8259 bool
8260 pmap_is_nested(
8261 	pmap_t pmap)
8262 {
8263 	return pmap->type == PMAP_TYPE_NESTED;
8264 }
8265 
8266 /*
8267  * pmap_trim_range(pmap, start, end)
8268  *
8269  * pmap  = pmap to operate on
8270  * start = start of the range
8271  * end   = end of the range
8272  *
8273  * Attempts to deallocate TTEs for the given range in the nested range.
8274  */
8275 MARK_AS_PMAP_TEXT static void
8276 pmap_trim_range(
8277 	pmap_t pmap,
8278 	addr64_t start,
8279 	addr64_t end)
8280 {
8281 	addr64_t cur;
8282 	addr64_t nested_region_start;
8283 	addr64_t nested_region_end;
8284 	addr64_t adjusted_start;
8285 	addr64_t adjusted_end;
8286 	addr64_t adjust_offmask;
8287 	tt_entry_t * tte_p;
8288 	__unused const pt_attr_t * const pt_attr = pmap_get_pt_attr(pmap);
8289 
8290 	if (__improbable(end < start)) {
8291 		panic("%s: invalid address range, "
8292 		    "pmap=%p, start=%p, end=%p",
8293 		    __func__,
8294 		    pmap, (void*)start, (void*)end);
8295 	}
8296 
8297 	nested_region_start = pmap->nested_region_addr;
8298 	nested_region_end = nested_region_start + pmap->nested_region_size;
8299 
8300 	if (__improbable((start < nested_region_start) || (end > nested_region_end))) {
8301 		panic("%s: range outside nested region %p-%p, "
8302 		    "pmap=%p, start=%p, end=%p",
8303 		    __func__, (void *)nested_region_start, (void *)nested_region_end,
8304 		    pmap, (void*)start, (void*)end);
8305 	}
8306 
8307 	/* Contract the range to TT page boundaries. */
8308 	const uint64_t page_ratio = PAGE_SIZE / pt_attr_page_size(pt_attr);
8309 
8310 	adjust_offmask = pt_attr_leaf_table_offmask(pt_attr) * page_ratio;
8311 	adjusted_start = ((start + adjust_offmask) & ~adjust_offmask);
8312 	adjusted_end = end & ~adjust_offmask;
8313 
8314 	/* Iterate over the range, trying to remove TTEs. */
8315 	for (cur = adjusted_start; (cur < adjusted_end) && (cur >= adjusted_start); cur += (pt_attr_twig_size(pt_attr) * page_ratio)) {
8316 		pmap_lock(pmap, PMAP_LOCK_EXCLUSIVE);
8317 
8318 		tte_p = pmap_tte(pmap, cur);
8319 
8320 		if ((tte_p != NULL) && ((*tte_p & ARM_TTE_TYPE_MASK) == ARM_TTE_TYPE_TABLE)) {
8321 			/* pmap_tte_deallocate()/pmap_tte_trim() will drop the pmap lock */
8322 			if ((pmap->type == PMAP_TYPE_NESTED) && (sptm_get_page_table_refcnt(tte_to_pa(*tte_p)) == 0)) {
8323 				/* Deallocate for the nested map. */
8324 				pmap_tte_deallocate(pmap, cur, tte_p, pt_attr_twig_level(pt_attr));
8325 			} else if (pmap->type == PMAP_TYPE_USER) {
8326 				/**
8327 				 * Just remove for the parent map. If the leaf table pointed
8328 				 * to by the TTE being removed (owned by the nested pmap)
8329 				 * has any mappings, then this call will panic. This
8330 				 * enforces the policy that tables being trimmed must be
8331 				 * empty to prevent possible use-after-free attacks.
8332 				 */
8333 				pmap_tte_trim(pmap, cur, tte_p);
8334 			} else {
8335 				panic("%s: Unsupported pmap type for nesting %p %d", __func__, pmap, pmap->type);
8336 			}
8337 		} else {
8338 			pmap_unlock(pmap, PMAP_LOCK_EXCLUSIVE);
8339 		}
8340 	}
8341 }
8342 
8343 /*
8344  * pmap_trim_internal(grand, subord, vstart, size)
8345  *
8346  * grand  = pmap subord is nested in
8347  * subord = nested pmap
8348  * vstart = start of the used range in grand
8349  * size   = size of the used range
8350  *
8351  * Attempts to trim the shared region page tables down to only cover the given
8352  * range in subord and grand.
8353  */
8354 MARK_AS_PMAP_TEXT void
8355 pmap_trim_internal(
8356 	pmap_t grand,
8357 	pmap_t subord,
8358 	addr64_t vstart,
8359 	uint64_t size)
8360 {
8361 	addr64_t vend;
8362 	addr64_t adjust_offmask;
8363 
8364 	if (__improbable(os_add_overflow(vstart, size, &vend))) {
8365 		panic("%s: grand addr wraps around, "
8366 		    "grand=%p, subord=%p, vstart=%p, size=%#llx",
8367 		    __func__, grand, subord, (void*)vstart, size);
8368 	}
8369 
8370 	validate_pmap_mutable(grand);
8371 	validate_pmap(subord);
8372 
8373 	__unused const pt_attr_t * const pt_attr = pmap_get_pt_attr(grand);
8374 
8375 	pmap_lock(subord, PMAP_LOCK_EXCLUSIVE);
8376 
8377 	if (__improbable(subord->type != PMAP_TYPE_NESTED)) {
8378 		panic("%s: subord is of non-nestable type 0x%hhx, "
8379 		    "grand=%p, subord=%p, vstart=%p, size=%#llx",
8380 		    __func__, subord->type, grand, subord, (void*)vstart, size);
8381 	}
8382 
8383 	if (__improbable(grand->type != PMAP_TYPE_USER)) {
8384 		panic("%s: grand is of unsupprted type 0x%hhx for nesting, "
8385 		    "grand=%p, subord=%p, vstart=%p, size=%#llx",
8386 		    __func__, grand->type, grand, subord, (void*)vstart, size);
8387 	}
8388 
8389 	if (__improbable(grand->nested_pmap != subord)) {
8390 		panic("%s: grand->nested != subord, "
8391 		    "grand=%p, subord=%p, vstart=%p, size=%#llx",
8392 		    __func__, grand, subord, (void*)vstart, size);
8393 	}
8394 
8395 	if (__improbable((size != 0) &&
8396 	    ((vstart < grand->nested_region_addr) || (vend > (grand->nested_region_addr + grand->nested_region_size))))) {
8397 		panic("%s: grand range not in nested region, "
8398 		    "grand=%p, subord=%p, vstart=%p, size=%#llx",
8399 		    __func__, grand, subord, (void*)vstart, size);
8400 	}
8401 
8402 
8403 	if (!grand->nested_has_no_bounds_ref) {
8404 		assert(subord->nested_bounds_set);
8405 
8406 		if (!grand->nested_bounds_set) {
8407 			/* Inherit the bounds from subord. */
8408 			grand->nested_region_true_start = subord->nested_region_true_start;
8409 			grand->nested_region_true_end = subord->nested_region_true_end;
8410 			grand->nested_bounds_set = true;
8411 		}
8412 
8413 		pmap_unlock(subord, PMAP_LOCK_EXCLUSIVE);
8414 		return;
8415 	}
8416 
8417 	if ((!subord->nested_bounds_set) && size) {
8418 		const uint64_t page_ratio = PAGE_SIZE / pt_attr_page_size(pt_attr);
8419 		adjust_offmask = pt_attr_leaf_table_offmask(pt_attr) * page_ratio;
8420 
8421 		subord->nested_region_true_start = vstart;
8422 		subord->nested_region_true_end = vend;
8423 		subord->nested_region_true_start &= ~adjust_offmask;
8424 
8425 		if (__improbable(os_add_overflow(subord->nested_region_true_end, adjust_offmask, &subord->nested_region_true_end))) {
8426 			panic("%s: padded true end wraps around, "
8427 			    "grand=%p, subord=%p, vstart=%p, size=%#llx",
8428 			    __func__, grand, subord, (void*)vstart, size);
8429 		}
8430 
8431 		subord->nested_region_true_end &= ~adjust_offmask;
8432 		subord->nested_bounds_set = true;
8433 	}
8434 
8435 	if (subord->nested_bounds_set) {
8436 		/* Inherit the bounds from subord. */
8437 		grand->nested_region_true_start = subord->nested_region_true_start;
8438 		grand->nested_region_true_end = subord->nested_region_true_end;
8439 		grand->nested_bounds_set = true;
8440 
8441 		/* If we know the bounds, we can trim the pmap. */
8442 		grand->nested_has_no_bounds_ref = false;
8443 		pmap_unlock(subord, PMAP_LOCK_EXCLUSIVE);
8444 	} else {
8445 		/* Don't trim if we don't know the bounds. */
8446 		pmap_unlock(subord, PMAP_LOCK_EXCLUSIVE);
8447 		return;
8448 	}
8449 
8450 	/* Trim grand to only cover the given range. */
8451 	pmap_trim_range(grand, grand->nested_region_addr, grand->nested_region_true_start);
8452 	pmap_trim_range(grand, grand->nested_region_true_end, (grand->nested_region_addr + grand->nested_region_size));
8453 
8454 	/* Try to trim subord. */
8455 	pmap_trim_subord(subord);
8456 }
8457 
8458 MARK_AS_PMAP_TEXT static void
8459 pmap_trim_self(pmap_t pmap)
8460 {
8461 	if (pmap->nested_has_no_bounds_ref && pmap->nested_pmap) {
8462 		/* If we have a no bounds ref, we need to drop it. */
8463 		pmap_lock(pmap->nested_pmap, PMAP_LOCK_SHARED);
8464 		pmap->nested_has_no_bounds_ref = false;
8465 		boolean_t nested_bounds_set = pmap->nested_pmap->nested_bounds_set;
8466 		vm_map_offset_t nested_region_true_start = pmap->nested_pmap->nested_region_true_start;
8467 		vm_map_offset_t nested_region_true_end = pmap->nested_pmap->nested_region_true_end;
8468 		pmap_unlock(pmap->nested_pmap, PMAP_LOCK_SHARED);
8469 
8470 		if (nested_bounds_set) {
8471 			pmap_trim_range(pmap, pmap->nested_region_addr, nested_region_true_start);
8472 			pmap_trim_range(pmap, nested_region_true_end, (pmap->nested_region_addr + pmap->nested_region_size));
8473 		}
8474 		/*
8475 		 * Try trimming the nested pmap, in case we had the
8476 		 * last reference.
8477 		 */
8478 		pmap_trim_subord(pmap->nested_pmap);
8479 	}
8480 }
8481 
8482 /*
8483  * pmap_trim_subord(grand, subord)
8484  *
8485  * grand  = pmap that we have nested subord in
8486  * subord = nested pmap we are attempting to trim
8487  *
8488  * Trims subord if possible
8489  */
8490 MARK_AS_PMAP_TEXT static void
8491 pmap_trim_subord(pmap_t subord)
8492 {
8493 	bool contract_subord = false;
8494 
8495 	pmap_lock(subord, PMAP_LOCK_EXCLUSIVE);
8496 
8497 	subord->nested_no_bounds_refcnt--;
8498 
8499 	if ((subord->nested_no_bounds_refcnt == 0) && (subord->nested_bounds_set)) {
8500 		/* If this was the last no bounds reference, trim subord. */
8501 		contract_subord = true;
8502 	}
8503 
8504 	pmap_unlock(subord, PMAP_LOCK_EXCLUSIVE);
8505 
8506 	if (contract_subord) {
8507 		pmap_trim_range(subord, subord->nested_region_addr, subord->nested_region_true_start);
8508 		pmap_trim_range(subord, subord->nested_region_true_end, subord->nested_region_addr + subord->nested_region_size);
8509 	}
8510 }
8511 
8512 void
8513 pmap_trim(
8514 	pmap_t grand,
8515 	pmap_t subord,
8516 	addr64_t vstart,
8517 	uint64_t size)
8518 {
8519 	pmap_trim_internal(grand, subord, vstart, size);
8520 }
8521 
8522 #if HAS_APPLE_PAC
8523 
8524 void *
8525 pmap_sign_user_ptr(void *value, ptrauth_key key, uint64_t discriminator, uint64_t jop_key)
8526 {
8527 	void *res = NULL;
8528 	const boolean_t current_intr_state = ml_set_interrupts_enabled(FALSE);
8529 
8530 	uint64_t saved_jop_state = ml_enable_user_jop_key(jop_key);
8531 	__compiler_materialize_and_prevent_reordering_on(value);
8532 	res = sptm_sign_user_pointer(value, key, discriminator, jop_key);
8533 	__compiler_materialize_and_prevent_reordering_on(res);
8534 	ml_disable_user_jop_key(jop_key, saved_jop_state);
8535 
8536 	ml_set_interrupts_enabled(current_intr_state);
8537 
8538 	return res;
8539 }
8540 
8541 void *
8542 pmap_auth_user_ptr(void *value, ptrauth_key key, uint64_t discriminator, uint64_t jop_key)
8543 {
8544 	void *res = NULL;
8545 	const boolean_t current_intr_state = ml_set_interrupts_enabled(FALSE);
8546 
8547 	uint64_t saved_jop_state = ml_enable_user_jop_key(jop_key);
8548 	__compiler_materialize_and_prevent_reordering_on(value);
8549 	res = sptm_auth_user_pointer(value, key, discriminator, jop_key);
8550 	__compiler_materialize_and_prevent_reordering_on(res);
8551 	ml_disable_user_jop_key(jop_key, saved_jop_state);
8552 
8553 	ml_set_interrupts_enabled(current_intr_state);
8554 
8555 	return res;
8556 }
8557 #endif /* HAS_APPLE_PAC */
8558 
8559 /*
8560  *	kern_return_t pmap_nest(grand, subord, vstart, size)
8561  *
8562  *	grand  = the pmap that we will nest subord into
8563  *	subord = the pmap that goes into the grand
8564  *	vstart  = start of range in pmap to be inserted
8565  *	size   = Size of nest area (up to 16TB)
8566  *
8567  *	Inserts a pmap into another.  This is used to implement shared segments.
8568  *
8569  */
8570 
8571 /**
8572  * Embeds a range of mappings from one pmap ('subord') into another ('grand')
8573  * by inserting the twig-level TTEs from 'subord' directly into 'grand'.
8574  * This function operates in 3 main phases:
8575  * 1. Bookkeeping to ensure tracking structures for the nested region are set up.
8576  * 2. Expansion of subord to ensure the required leaf-level page table pages for
8577  *    the mapping range are present in subord.
8578  * 3. Expansion of grand to ensure the required twig-level page table pages for
8579  *    the mapping range are present in grand.
8580  * 4. Invoke sptm_nest_region() to copy the relevant TTEs from subord to grand.
8581  *
8582  * This function may return early due to pending AST_URGENT preemption; if so
8583  * it will indicate the need to be re-entered.
8584  *
8585  * @param grand pmap to insert the TTEs into.  Must be a user pmap.
8586  * @param subord pmap from which to extract the TTEs.  Must be a nested pmap.
8587  * @param vstart twig-aligned virtual address for the beginning of the nesting range
8588  * @param size twig-aligned size of the nesting range
8589  *
8590  * @return KERN_RESOURCE_SHORTAGE on allocation failure, KERN_SUCCESS otherwise
8591  */
8592 MARK_AS_PMAP_TEXT kern_return_t
8593 pmap_nest_internal(
8594 	pmap_t grand,
8595 	pmap_t subord,
8596 	addr64_t vstart,
8597 	uint64_t size)
8598 {
8599 	kern_return_t kr = KERN_SUCCESS;
8600 	vm_map_offset_t vaddr;
8601 	tt_entry_t     *stte_p;
8602 	tt_entry_t     *gtte_p;
8603 	bitmap_t       *nested_region_unnested_table_bitmap;
8604 	int             expand_options = 0;
8605 	bool            deref_subord = true;
8606 
8607 	addr64_t vend;
8608 	if (__improbable(os_add_overflow(vstart, size, &vend))) {
8609 		panic("%s: %p grand addr wraps around: 0x%llx + 0x%llx", __func__, grand, vstart, size);
8610 	}
8611 
8612 	validate_pmap_mutable(grand);
8613 	validate_pmap(subord);
8614 	os_ref_retain_raw(&subord->ref_count, &pmap_refgrp);
8615 
8616 	const pt_attr_t * const pt_attr = pmap_get_pt_attr(grand);
8617 	if (__improbable(pmap_get_pt_attr(subord) != pt_attr)) {
8618 		panic("%s: attempt to nest pmap %p into pmap %p with mismatched attributes", __func__, subord, grand);
8619 	}
8620 
8621 	if (__improbable(((size | vstart) &
8622 	    (pt_attr_leaf_table_offmask(pt_attr))) != 0x0ULL)) {
8623 		panic("pmap_nest() pmap %p unaligned nesting request 0x%llx, 0x%llx",
8624 		    grand, vstart, size);
8625 	}
8626 
8627 	if (__improbable(subord->type != PMAP_TYPE_NESTED)) {
8628 		panic("%s: subordinate pmap %p is of non-nestable type 0x%hhx", __func__, subord, subord->type);
8629 	}
8630 
8631 	if (__improbable(grand->type != PMAP_TYPE_USER)) {
8632 		panic("%s: grand pmap %p is of unsupported type 0x%hhx for nesting", __func__, grand, grand->type);
8633 	}
8634 
8635 	/**
8636 	 * Use an acquire barrier to ensure that subsequent loads of nested_region_* fields are not
8637 	 * speculated ahead of the load of nested_region_unnested_table_bitmap, so that if we observe a non-NULL
8638 	 * nested_region_unnested_table_bitmap then we can be sure the other fields have been initialized as well.
8639 	 */
8640 	if (os_atomic_load(&subord->nested_region_unnested_table_bitmap, acquire) == NULL) {
8641 		uint64_t nested_region_unnested_table_bits = size >> pt_attr_twig_shift(pt_attr);
8642 
8643 		if (__improbable((nested_region_unnested_table_bits > UINT_MAX))) {
8644 			panic("%s: bitmap allocation size %llu will truncate, "
8645 			    "grand=%p, subord=%p, vstart=0x%llx, size=%llx",
8646 			    __func__, nested_region_unnested_table_bits,
8647 			    grand, subord, vstart, size);
8648 		}
8649 
8650 		nested_region_unnested_table_bitmap = bitmap_alloc((uint) nested_region_unnested_table_bits);
8651 
8652 		pmap_lock(subord, PMAP_LOCK_EXCLUSIVE);
8653 		if (subord->nested_region_unnested_table_bitmap == NULL) {
8654 			subord->nested_region_addr = vstart;
8655 			subord->nested_region_size = (mach_vm_offset_t) size;
8656 			sptm_configure_shared_region(subord->ttep, vstart, size >> pt_attr->pta_page_shift);
8657 
8658 			/**
8659 			 * Ensure that the rest of the subord->nested_region_* fields are
8660 			 * initialized and visible before setting the nested_region_unnested_table_bitmap
8661 			 * field (which is used as the flag to say that the rest are initialized).
8662 			 */
8663 			os_atomic_store(&subord->nested_region_unnested_table_bitmap, nested_region_unnested_table_bitmap, release);
8664 			nested_region_unnested_table_bitmap = NULL;
8665 		}
8666 		pmap_unlock(subord, PMAP_LOCK_EXCLUSIVE);
8667 		if (nested_region_unnested_table_bitmap != NULL) {
8668 			bitmap_free(nested_region_unnested_table_bitmap, nested_region_unnested_table_bits);
8669 		}
8670 	}
8671 
8672 	assertf(subord->nested_region_addr == vstart, "%s: pmap %p nested region addr 0x%llx doesn't match vstart 0x%llx",
8673 	    __func__, subord, (unsigned long long)subord->nested_region_addr, (unsigned long long)vstart);
8674 	assertf(subord->nested_region_size == size, "%s: pmap %p nested region size 0x%llx doesn't match size 0x%llx",
8675 	    __func__, subord, (unsigned long long)subord->nested_region_size, (unsigned long long)size);
8676 
8677 	pmap_lock(subord, PMAP_LOCK_EXCLUSIVE);
8678 
8679 	if (os_atomic_cmpxchg(&grand->nested_pmap, PMAP_NULL, subord, relaxed)) {
8680 		/*
8681 		 * If this is grand's first nesting operation, keep the reference on subord.
8682 		 * It will be released by pmap_destroy_internal() when grand is destroyed.
8683 		 */
8684 		deref_subord = false;
8685 
8686 		if (!subord->nested_bounds_set) {
8687 			/*
8688 			 * We are nesting without the shared regions bounds
8689 			 * being known.  We'll have to trim the pmap later.
8690 			 */
8691 			grand->nested_has_no_bounds_ref = true;
8692 			subord->nested_no_bounds_refcnt++;
8693 		}
8694 
8695 		grand->nested_region_addr = vstart;
8696 		grand->nested_region_size = (mach_vm_offset_t) size;
8697 	} else {
8698 		if (__improbable(grand->nested_pmap != subord)) {
8699 			panic("pmap_nest() pmap %p has a nested pmap", grand);
8700 		} else if (__improbable(grand->nested_region_addr > vstart)) {
8701 			panic("pmap_nest() pmap %p : attempt to nest outside the nested region", grand);
8702 		} else if ((grand->nested_region_addr + grand->nested_region_size) < vend) {
8703 			grand->nested_region_size = (mach_vm_offset_t)(vstart - grand->nested_region_addr + size);
8704 		}
8705 	}
8706 
8707 	vaddr = vstart;
8708 	if (vaddr < subord->nested_region_true_start) {
8709 		vaddr = subord->nested_region_true_start;
8710 	}
8711 
8712 	addr64_t true_end = vend;
8713 	if (true_end > subord->nested_region_true_end) {
8714 		true_end = subord->nested_region_true_end;
8715 	}
8716 
8717 	while (vaddr < true_end) {
8718 		stte_p = pmap_tte(subord, vaddr);
8719 		if (stte_p == PT_ENTRY_NULL || *stte_p == ARM_TTE_EMPTY) {
8720 			pmap_unlock(subord, PMAP_LOCK_EXCLUSIVE);
8721 			kr = pmap_expand(subord, vaddr, expand_options, pt_attr_leaf_level(pt_attr));
8722 
8723 			if (kr != KERN_SUCCESS) {
8724 				pmap_lock(grand, PMAP_LOCK_EXCLUSIVE);
8725 				goto done;
8726 			}
8727 
8728 			pmap_lock(subord, PMAP_LOCK_EXCLUSIVE);
8729 		}
8730 		vaddr += pt_attr_twig_size(pt_attr);
8731 	}
8732 
8733 	/*
8734 	 * copy TTEs from subord pmap into grand pmap
8735 	 */
8736 
8737 	vaddr = (vm_map_offset_t) vstart;
8738 	if (vaddr < subord->nested_region_true_start) {
8739 		vaddr = subord->nested_region_true_start;
8740 	}
8741 
8742 	pmap_unlock(subord, PMAP_LOCK_EXCLUSIVE);
8743 	pmap_lock(grand, PMAP_LOCK_EXCLUSIVE);
8744 
8745 	while (vaddr < true_end) {
8746 		gtte_p = pmap_tte(grand, vaddr);
8747 		if (gtte_p == PT_ENTRY_NULL) {
8748 			pmap_unlock(grand, PMAP_LOCK_EXCLUSIVE);
8749 			kr = pmap_expand(grand, vaddr, expand_options, pt_attr_twig_level(pt_attr));
8750 			pmap_lock(grand, PMAP_LOCK_EXCLUSIVE);
8751 
8752 			if (kr != KERN_SUCCESS) {
8753 				goto done;
8754 			}
8755 		}
8756 
8757 		vaddr += pt_attr_twig_size(pt_attr);
8758 	}
8759 
8760 	vaddr = (vm_map_offset_t) vstart;
8761 
8762 	/*
8763 	 * It is possible to have a preempted nest operation execute concurrently
8764 	 * with a trim operation that sets nested_region_true_start.  In this case,
8765 	 * update the nesting bounds.  This is useful both as a performance
8766 	 * optimization and to prevent an attempt to nest a just-trimmed TTE,
8767 	 * which will trigger an SPTM violation.
8768 	 * Note that pmap_trim() may concurrently update grand's bounds as we are
8769 	 * making these checks, but in that case pmap_trim_range() has not yet
8770 	 * been called on grand and will wait for us to drop grand's lock, so it
8771 	 * should see any TTEs we've nested here and clear them appropriately.
8772 	 */
8773 	if (vaddr < subord->nested_region_true_start) {
8774 		vaddr = subord->nested_region_true_start;
8775 	}
8776 	if (vaddr < grand->nested_region_true_start) {
8777 		vaddr = grand->nested_region_true_start;
8778 	}
8779 	if (true_end > subord->nested_region_true_end) {
8780 		true_end = subord->nested_region_true_end;
8781 	}
8782 	if (true_end > grand->nested_region_true_end) {
8783 		true_end = grand->nested_region_true_end;
8784 	}
8785 
8786 	while (vaddr < true_end) {
8787 		/*
8788 		 * The SPTM requires the run of TTE updates to all reside within the same L2 page, so the region
8789 		 * we supply to the SPTM can't span multiple L1 TTEs.
8790 		 */
8791 		vm_map_offset_t vlim = ((vaddr + pt_attr_ln_size(pt_attr, PMAP_TT_L1_LEVEL)) & ~pt_attr_ln_offmask(pt_attr, PMAP_TT_L1_LEVEL));
8792 		if (vlim > true_end) {
8793 			vlim = true_end;
8794 		}
8795 		pmap_txm_acquire_exclusive_lock(grand);
8796 		pmap_txm_acquire_shared_lock(subord);
8797 		sptm_nest_region(grand->ttep, subord->ttep, vaddr, (vlim - vaddr) >> pt_attr->pta_page_shift);
8798 		pmap_txm_release_shared_lock(subord);
8799 		pmap_txm_release_exclusive_lock(grand);
8800 		vaddr = vlim;
8801 	}
8802 
8803 done:
8804 	pmap_unlock(grand, PMAP_LOCK_EXCLUSIVE);
8805 	if (deref_subord) {
8806 		pmap_destroy_internal(subord);
8807 	}
8808 
8809 	return kr;
8810 }
8811 
8812 kern_return_t
8813 pmap_nest(
8814 	pmap_t grand,
8815 	pmap_t subord,
8816 	addr64_t vstart,
8817 	uint64_t size)
8818 {
8819 	kern_return_t kr = KERN_SUCCESS;
8820 
8821 	PMAP_TRACE(2, PMAP_CODE(PMAP__NEST) | DBG_FUNC_START,
8822 	    VM_KERNEL_ADDRHIDE(grand), VM_KERNEL_ADDRHIDE(subord),
8823 	    VM_KERNEL_ADDRHIDE(vstart));
8824 
8825 	pmap_verify_preemptible();
8826 	kr = pmap_nest_internal(grand, subord, vstart, size);
8827 
8828 	PMAP_TRACE(2, PMAP_CODE(PMAP__NEST) | DBG_FUNC_END, kr);
8829 
8830 	return kr;
8831 }
8832 
8833 /*
8834  *	kern_return_t pmap_unnest(grand, vaddr)
8835  *
8836  *	grand  = the pmap that will have the virtual range unnested
8837  *	vaddr  = start of range in pmap to be unnested
8838  *	size   = size of range in pmap to be unnested
8839  *
8840  */
8841 
8842 kern_return_t
8843 pmap_unnest(
8844 	pmap_t grand,
8845 	addr64_t vaddr,
8846 	uint64_t size)
8847 {
8848 	return pmap_unnest_options(grand, vaddr, size, 0);
8849 }
8850 
8851 /**
8852  * Undoes a prior pmap_nest() operation by removing a range of nesting mappings
8853  * from a top-level pmap ('grand').  The corresponding mappings in the nested
8854  * pmap will be marked non-global to avoid TLB conflicts with pmaps that may
8855  * still have the region nested.  The mappings in 'grand' will be left empty
8856  * with the assumption that they will be demand-filled by subsequent access faults.
8857  *
8858  * This function operates in 2 main phases:
8859  * 1. Iteration over the nested pmap's mappings for the specified range to mark
8860  *    them non-global.
8861  * 2. Calling the SPTM to clear the twig-level TTEs for the address range in grand.
8862  *
8863  * This function may return early due to pending AST_URGENT preemption; if so
8864  * it will indicate the need to be re-entered.
8865  *
8866  * @param grand pmap from which to unnest mappings
8867  * @param vaddr twig-aligned virtual address for the beginning of the nested range
8868  * @param size twig-aligned size of the nested range
8869  * @param option Extra control flags; may contain PMAP_UNNEST_CLEAN to indicate that
8870  *        grand is being torn down and step 1) above is not needed.
8871  */
8872 MARK_AS_PMAP_TEXT void
8873 pmap_unnest_options_internal(
8874 	pmap_t grand,
8875 	addr64_t vaddr,
8876 	uint64_t size,
8877 	unsigned int option)
8878 {
8879 	vm_map_offset_t start;
8880 	vm_map_offset_t addr;
8881 	unsigned int    current_index;
8882 	unsigned int    start_index;
8883 	unsigned int    max_index;
8884 
8885 	addr64_t vend;
8886 	addr64_t true_end;
8887 	if (__improbable(os_add_overflow(vaddr, size, &vend))) {
8888 		panic("%s: %p vaddr wraps around: 0x%llx + 0x%llx", __func__, grand, vaddr, size);
8889 	}
8890 
8891 	validate_pmap_mutable(grand);
8892 
8893 	__unused const pt_attr_t * const pt_attr = pmap_get_pt_attr(grand);
8894 
8895 	if (__improbable(((size | vaddr) & pt_attr_twig_offmask(pt_attr)) != 0x0ULL)) {
8896 		panic("%s: unaligned base address 0x%llx or size 0x%llx", __func__,
8897 		    (unsigned long long)vaddr, (unsigned long long)size);
8898 	}
8899 
8900 	if (__improbable(grand->nested_pmap == NULL)) {
8901 		panic("%s: %p has no nested pmap", __func__, grand);
8902 	}
8903 
8904 	true_end = vend;
8905 	if (true_end > grand->nested_pmap->nested_region_true_end) {
8906 		true_end = grand->nested_pmap->nested_region_true_end;
8907 	}
8908 
8909 	if ((option & PMAP_UNNEST_CLEAN) == 0) {
8910 		if ((vaddr < grand->nested_region_addr) || (vend > (grand->nested_region_addr + grand->nested_region_size))) {
8911 			panic("%s: %p: unnest request to not-fully-nested region [%p, %p)", __func__, grand, (void*)vaddr, (void*)vend);
8912 		}
8913 
8914 		/*
8915 		 * SPTM TODO: I suspect we may be able to hold the nested pmap lock shared here.
8916 		 * We would need to use atomic_bitmap_set below where we currently use bitmap_test + bitmap_set.
8917 		 * The risk is that a concurrent pmap_enter() against the nested pmap could observe the relevant
8918 		 * bit in the nested region bitmap to be clear, but could then create the (global) mapping after
8919 		 * we've made our SPTM sweep below to set NG.  In that case we could end up with a mix of global
8920 		 * and non-global mappings for the same VA region and thus a TLB conflict.  I'm uncertain if the
8921 		 * VM would allow these operation to happen concurrently.  Even if it does, we could still do
8922 		 * something fancier here such as waiting for concurrent pmap_enter() to drain after updating
8923 		 * the bitmap.
8924 		 */
8925 		pmap_lock(grand->nested_pmap, PMAP_LOCK_EXCLUSIVE);
8926 
8927 		disable_preemption();
8928 		pmap_sptm_percpu_data_t *sptm_pcpu = PERCPU_GET(pmap_sptm_percpu);
8929 		unsigned int num_mappings = 0;
8930 		start = vaddr;
8931 		if (start < grand->nested_pmap->nested_region_true_start) {
8932 			start = grand->nested_pmap->nested_region_true_start;
8933 		}
8934 		start_index = (unsigned int)((start - grand->nested_region_addr) >> pt_attr_twig_shift(pt_attr));
8935 		max_index = (unsigned int)((true_end - grand->nested_region_addr) >> pt_attr_twig_shift(pt_attr));
8936 
8937 		for (current_index = start_index, addr = start; current_index < max_index; current_index++) {
8938 			pt_entry_t  *bpte, *cpte;
8939 
8940 			vm_map_offset_t vlim = (addr + pt_attr_twig_size(pt_attr)) & ~pt_attr_twig_offmask(pt_attr);
8941 
8942 			bpte = pmap_pte(grand->nested_pmap, addr);
8943 
8944 			if (!bitmap_test(grand->nested_pmap->nested_region_unnested_table_bitmap, current_index)) {
8945 				/*
8946 				 * We've marked the 'twig' region as being unnested.  Every mapping entered within
8947 				 * the nested pmap in this region will now be marked non-global.
8948 				 */
8949 				bitmap_set(grand->nested_pmap->nested_region_unnested_table_bitmap, current_index);
8950 				for (cpte = bpte; (bpte != NULL) && (addr < vlim); cpte += PAGE_RATIO) {
8951 					pt_entry_t  spte = os_atomic_load(cpte, relaxed);
8952 
8953 					if ((spte & ARM_PTE_TYPE_MASK) != ARM_PTE_TYPE_FAULT) {
8954 						spte |= ARM_PTE_NG;
8955 					}
8956 
8957 					addr += (pt_attr_page_size(pt_attr) * PAGE_RATIO);
8958 
8959 					sptm_pcpu->sptm_templates[num_mappings] = spte;
8960 					++num_mappings;
8961 
8962 					if (num_mappings == SPTM_MAPPING_LIMIT) {
8963 						pmap_retype_epoch_enter();
8964 						sptm_update_region(grand->nested_pmap->ttep, start, num_mappings,
8965 						    sptm_pcpu->sptm_templates_pa, SPTM_UPDATE_NG);
8966 						pmap_retype_epoch_exit();
8967 						enable_preemption();
8968 						num_mappings = 0;
8969 						start = addr;
8970 						disable_preemption();
8971 						sptm_pcpu = PERCPU_GET(pmap_sptm_percpu);
8972 					}
8973 				}
8974 			}
8975 			/**
8976 			 * The SPTM does not allow region updates to span multiple leaf page tables, so request
8977 			 * any remaining updates up to vlim before moving to the next page table page.
8978 			 */
8979 			if (num_mappings != 0) {
8980 				pmap_retype_epoch_enter();
8981 				sptm_update_region(grand->nested_pmap->ttep, start, num_mappings,
8982 				    sptm_pcpu->sptm_templates_pa, SPTM_UPDATE_NG);
8983 				pmap_retype_epoch_exit();
8984 				enable_preemption();
8985 				num_mappings = 0;
8986 				disable_preemption();
8987 				sptm_pcpu = PERCPU_GET(pmap_sptm_percpu);
8988 			}
8989 			addr = start = vlim;
8990 		}
8991 
8992 		if (num_mappings != 0) {
8993 			pmap_retype_epoch_enter();
8994 			sptm_update_region(grand->nested_pmap->ttep, start, num_mappings,
8995 			    sptm_pcpu->sptm_templates_pa, SPTM_UPDATE_NG);
8996 			pmap_retype_epoch_exit();
8997 		}
8998 
8999 		enable_preemption();
9000 		pmap_unlock(grand->nested_pmap, PMAP_LOCK_EXCLUSIVE);
9001 	}
9002 
9003 	/*
9004 	 * invalidate all pdes for segment at vaddr in pmap grand
9005 	 */
9006 	addr = vaddr;
9007 
9008 	pmap_lock(grand, PMAP_LOCK_EXCLUSIVE);
9009 
9010 	if (addr < grand->nested_pmap->nested_region_true_start) {
9011 		addr = grand->nested_pmap->nested_region_true_start;
9012 	}
9013 
9014 	if (true_end > grand->nested_pmap->nested_region_true_end) {
9015 		true_end = grand->nested_pmap->nested_region_true_end;
9016 	}
9017 
9018 	while (addr < true_end) {
9019 		vm_map_offset_t vlim = ((addr + pt_attr_ln_size(pt_attr, PMAP_TT_L1_LEVEL)) & ~pt_attr_ln_offmask(pt_attr, PMAP_TT_L1_LEVEL));
9020 		if (vlim > true_end) {
9021 			vlim = true_end;
9022 		}
9023 		sptm_unnest_region(grand->ttep, grand->nested_pmap->ttep, addr, (vlim - addr) >> pt_attr->pta_page_shift);
9024 		addr = vlim;
9025 	}
9026 
9027 	pmap_unlock(grand, PMAP_LOCK_EXCLUSIVE);
9028 }
9029 
9030 kern_return_t
9031 pmap_unnest_options(
9032 	pmap_t grand,
9033 	addr64_t vaddr,
9034 	uint64_t size,
9035 	unsigned int option)
9036 {
9037 	PMAP_TRACE(2, PMAP_CODE(PMAP__UNNEST) | DBG_FUNC_START,
9038 	    VM_KERNEL_ADDRHIDE(grand), VM_KERNEL_ADDRHIDE(vaddr));
9039 
9040 	pmap_verify_preemptible();
9041 	pmap_unnest_options_internal(grand, vaddr, size, option);
9042 
9043 	PMAP_TRACE(2, PMAP_CODE(PMAP__UNNEST) | DBG_FUNC_END, KERN_SUCCESS);
9044 
9045 	return KERN_SUCCESS;
9046 }
9047 
9048 boolean_t
9049 pmap_adjust_unnest_parameters(
9050 	__unused pmap_t p,
9051 	__unused vm_map_offset_t *s,
9052 	__unused vm_map_offset_t *e)
9053 {
9054 	return TRUE; /* to get to log_unnest_badness()... */
9055 }
9056 
9057 #if PMAP_FORK_NEST
9058 /**
9059  * Perform any necessary pre-nesting of the parent's shared region at fork()
9060  * time.
9061  *
9062  * @note This should only be called from vm_map_fork().
9063  *
9064  * @param old_pmap The pmap of the parent task.
9065  * @param new_pmap The pmap of the child task.
9066  * @param nesting_start An output parameter that is updated with the start
9067  *                      address of the range that was pre-nested
9068  * @param nesting_end An output parameter that is updated with the end
9069  *                      address of the range that was pre-nested
9070  *
9071  * @return KERN_SUCCESS if the pre-nesting was succesfully completed.
9072  *         KERN_INVALID_ARGUMENT if the arguments were not valid.
9073  */
9074 kern_return_t
9075 pmap_fork_nest(
9076 	pmap_t old_pmap,
9077 	pmap_t new_pmap,
9078 	vm_map_offset_t *nesting_start,
9079 	vm_map_offset_t *nesting_end)
9080 {
9081 	if (old_pmap == NULL || new_pmap == NULL) {
9082 		return KERN_INVALID_ARGUMENT;
9083 	}
9084 	if (old_pmap->nested_pmap == NULL) {
9085 		return KERN_SUCCESS;
9086 	}
9087 	pmap_nest(new_pmap,
9088 	    old_pmap->nested_pmap,
9089 	    old_pmap->nested_region_addr,
9090 	    old_pmap->nested_region_size);
9091 	assertf(new_pmap->nested_pmap == old_pmap->nested_pmap &&
9092 	    new_pmap->nested_region_addr == old_pmap->nested_region_addr &&
9093 	    new_pmap->nested_region_size == old_pmap->nested_region_size,
9094 	    "nested new (%p,0x%llx,0x%llx) old (%p,0x%llx,0x%llx)",
9095 	    new_pmap->nested_pmap,
9096 	    new_pmap->nested_region_addr,
9097 	    new_pmap->nested_region_size,
9098 	    old_pmap->nested_pmap,
9099 	    old_pmap->nested_region_addr,
9100 	    old_pmap->nested_region_size);
9101 	*nesting_start = old_pmap->nested_region_addr;
9102 	*nesting_end = *nesting_start + old_pmap->nested_region_size;
9103 	return KERN_SUCCESS;
9104 }
9105 #endif /* PMAP_FORK_NEST */
9106 
9107 /*
9108  * disable no-execute capability on
9109  * the specified pmap
9110  */
9111 #if DEVELOPMENT || DEBUG
9112 void
9113 pmap_disable_NX(
9114 	pmap_t pmap)
9115 {
9116 	pmap->nx_enabled = FALSE;
9117 }
9118 #else
9119 void
9120 pmap_disable_NX(
9121 	__unused pmap_t pmap)
9122 {
9123 }
9124 #endif
9125 
9126 /*
9127  * flush a range of hardware TLB entries.
9128  * NOTE: assumes the smallest TLB entry in use will be for
9129  * an ARM small page (4K).
9130  */
9131 
9132 #if __ARM_RANGE_TLBI__
9133 #define ARM64_RANGE_TLB_FLUSH_THRESHOLD 1
9134 #define ARM64_FULL_TLB_FLUSH_THRESHOLD  ARM64_TLB_RANGE_MAX_PAGES
9135 #else
9136 #define ARM64_FULL_TLB_FLUSH_THRESHOLD  256
9137 #endif // __ARM_RANGE_TLBI__
9138 
9139 static void
9140 flush_mmu_tlb_region_asid_async(
9141 	vm_offset_t va,
9142 	size_t length,
9143 	pmap_t pmap,
9144 	bool last_level_only __unused)
9145 {
9146 	unsigned long pmap_page_shift = pt_attr_leaf_shift(pmap_get_pt_attr(pmap));
9147 	const uint64_t pmap_page_size = 1ULL << pmap_page_shift;
9148 	ppnum_t npages = (ppnum_t)(length >> pmap_page_shift);
9149 	const uint16_t asid = PMAP_HWASID(pmap);
9150 
9151 	if (npages > ARM64_FULL_TLB_FLUSH_THRESHOLD) {
9152 		boolean_t       flush_all = FALSE;
9153 
9154 		if ((asid == 0) || (pmap->type == PMAP_TYPE_NESTED)) {
9155 			flush_all = TRUE;
9156 		}
9157 		if (flush_all) {
9158 			flush_mmu_tlb_async();
9159 		} else {
9160 			flush_mmu_tlb_asid_async((uint64_t)asid << TLBI_ASID_SHIFT, false);
9161 		}
9162 		return;
9163 	}
9164 #if __ARM_RANGE_TLBI__
9165 	if (npages > ARM64_RANGE_TLB_FLUSH_THRESHOLD) {
9166 		va = generate_rtlbi_param(npages, asid, va, pmap_page_shift);
9167 		if (pmap->type == PMAP_TYPE_NESTED) {
9168 			flush_mmu_tlb_allrange_async(va, last_level_only, false);
9169 		} else {
9170 			flush_mmu_tlb_range_async(va, last_level_only, false);
9171 		}
9172 		return;
9173 	}
9174 #endif
9175 	vm_offset_t end = tlbi_asid(asid) | tlbi_addr(va + length);
9176 	va = tlbi_asid(asid) | tlbi_addr(va);
9177 
9178 	if (pmap->type == PMAP_TYPE_NESTED) {
9179 		flush_mmu_tlb_allentries_async(va, end, pmap_page_size, last_level_only, false);
9180 	} else {
9181 		flush_mmu_tlb_entries_async(va, end, pmap_page_size, last_level_only, false);
9182 	}
9183 }
9184 
9185 void
9186 flush_mmu_tlb_region(
9187 	vm_offset_t va,
9188 	unsigned length)
9189 {
9190 	flush_mmu_tlb_region_asid_async(va, length, kernel_pmap, true);
9191 	sync_tlb_flush();
9192 }
9193 
9194 unsigned int
9195 pmap_cache_attributes(
9196 	ppnum_t pn)
9197 {
9198 	pmap_paddr_t    paddr;
9199 	unsigned int    pai;
9200 	unsigned int    result;
9201 	pp_attr_t       pp_attr_current;
9202 
9203 	paddr = ptoa(pn);
9204 
9205 	assert(vm_last_phys > vm_first_phys); // Check that pmap has been bootstrapped
9206 
9207 	if (!pa_valid(paddr)) {
9208 		pmap_io_range_t *io_rgn = pmap_find_io_attr(paddr);
9209 		return (io_rgn == NULL) ? VM_WIMG_IO : io_rgn->wimg;
9210 	}
9211 
9212 	result = VM_WIMG_DEFAULT;
9213 
9214 	pai = pa_index(paddr);
9215 
9216 	pp_attr_current = pp_attr_table[pai];
9217 	if (pp_attr_current & PP_ATTR_WIMG_MASK) {
9218 		result = pp_attr_current & PP_ATTR_WIMG_MASK;
9219 	}
9220 	return result;
9221 }
9222 
9223 MARK_AS_PMAP_TEXT static void
9224 pmap_sync_wimg(ppnum_t pn, unsigned int wimg_bits_prev, unsigned int wimg_bits_new)
9225 {
9226 	if ((wimg_bits_prev != wimg_bits_new)
9227 	    && ((wimg_bits_prev == VM_WIMG_COPYBACK)
9228 	    || ((wimg_bits_prev == VM_WIMG_INNERWBACK)
9229 	    && (wimg_bits_new != VM_WIMG_COPYBACK))
9230 	    || ((wimg_bits_prev == VM_WIMG_WTHRU)
9231 	    && ((wimg_bits_new != VM_WIMG_COPYBACK) || (wimg_bits_new != VM_WIMG_INNERWBACK))))) {
9232 		pmap_sync_page_attributes_phys(pn);
9233 	}
9234 
9235 	if ((wimg_bits_new == VM_WIMG_RT) && (wimg_bits_prev != VM_WIMG_RT)) {
9236 		pmap_force_dcache_clean(phystokv(ptoa(pn)), PAGE_SIZE);
9237 	}
9238 }
9239 
9240 MARK_AS_PMAP_TEXT __unused void
9241 pmap_update_compressor_page_internal(ppnum_t pn, unsigned int prev_cacheattr, unsigned int new_cacheattr)
9242 {
9243 	pmap_paddr_t paddr = ptoa(pn);
9244 
9245 	if (__improbable(!pa_valid(paddr))) {
9246 		panic("%s called on non-managed page 0x%08x", __func__, pn);
9247 	}
9248 
9249 	pmap_set_cache_attributes_internal(pn, new_cacheattr, false);
9250 
9251 	pmap_sync_wimg(pn, prev_cacheattr & VM_WIMG_MASK, new_cacheattr & VM_WIMG_MASK);
9252 }
9253 
9254 void *
9255 pmap_map_compressor_page(ppnum_t pn)
9256 {
9257 	unsigned int cacheattr = pmap_cache_attributes(pn) & VM_WIMG_MASK;
9258 	if (cacheattr != VM_WIMG_DEFAULT) {
9259 		pmap_update_compressor_page_internal(pn, cacheattr, VM_WIMG_DEFAULT);
9260 	}
9261 
9262 	return (void*)phystokv(ptoa(pn));
9263 }
9264 
9265 void
9266 pmap_unmap_compressor_page(ppnum_t pn __unused, void *kva __unused)
9267 {
9268 	unsigned int cacheattr = pmap_cache_attributes(pn) & VM_WIMG_MASK;
9269 	if (cacheattr != VM_WIMG_DEFAULT) {
9270 		pmap_update_compressor_page_internal(pn, VM_WIMG_DEFAULT, cacheattr);
9271 	}
9272 }
9273 
9274 /**
9275  * Flushes TLB entries associated with the page specified by paddr, but do not
9276  * issue barriers yet.
9277  *
9278  * @param paddr The physical address to be flushed from TLB. Must be a managed address.
9279  */
9280 static void
9281 pmap_flush_tlb_for_paddr_async(pmap_paddr_t paddr)
9282 {
9283 	/* Flush the physical aperture mappings. */
9284 	const vm_offset_t kva = phystokv(paddr);
9285 	flush_mmu_tlb_region_asid_async(kva, PAGE_SIZE, kernel_pmap, true);
9286 
9287 	/* Flush the mappings tracked in the ptes. */
9288 	const unsigned int pai = pa_index(paddr);
9289 	locked_pvh_t locked_pvh = pvh_lock(pai);
9290 
9291 	pt_entry_t *pte_p = PT_ENTRY_NULL;
9292 	pv_entry_t *pve_p = PV_ENTRY_NULL;
9293 
9294 	if (pvh_test_type(locked_pvh.pvh, PVH_TYPE_PTEP)) {
9295 		pte_p = pvh_ptep(locked_pvh.pvh);
9296 	} else if (pvh_test_type(locked_pvh.pvh, PVH_TYPE_PVEP)) {
9297 		pve_p = pvh_pve_list(locked_pvh.pvh);
9298 		pte_p = PT_ENTRY_NULL;
9299 	}
9300 
9301 	unsigned int nptes = 0;
9302 	int pve_ptep_idx = 0;
9303 	while ((pve_p != PV_ENTRY_NULL) || (pte_p != PT_ENTRY_NULL)) {
9304 		if (pve_p != PV_ENTRY_NULL) {
9305 			pte_p = pve_get_ptep(pve_p, pve_ptep_idx);
9306 			if (pte_p == PT_ENTRY_NULL) {
9307 				goto flush_tlb_skip_pte;
9308 			}
9309 		}
9310 
9311 		if (__improbable(nptes == SPTM_MAPPING_LIMIT)) {
9312 			pvh_lock_enter_sleep_mode(&locked_pvh);
9313 		}
9314 		++nptes;
9315 #ifdef PVH_FLAG_IOMMU
9316 		if (pvh_ptep_is_iommu(pte_p)) {
9317 			goto flush_tlb_skip_pte;
9318 		}
9319 #endif /* PVH_FLAG_IOMMU */
9320 		const pmap_t pmap = ptep_get_pmap(pte_p);
9321 		const vm_map_address_t va = ptep_get_va(pte_p);
9322 
9323 		pmap_get_pt_ops(pmap)->flush_tlb_region_async(va, pt_attr_page_size(pmap_get_pt_attr(pmap)) * PAGE_RATIO, pmap, true);
9324 
9325 flush_tlb_skip_pte:
9326 		pte_p = PT_ENTRY_NULL;
9327 		if ((pve_p != PV_ENTRY_NULL) && (++pve_ptep_idx == PTE_PER_PVE)) {
9328 			pve_ptep_idx = 0;
9329 			pve_p = pve_next(pve_p);
9330 		}
9331 	}
9332 	pvh_unlock(&locked_pvh);
9333 }
9334 
9335 /**
9336  * Updates the pp_attr_table entry indexed by pai with cacheattr atomically.
9337  *
9338  * @param pai The Physical Address Index of the entry.
9339  * @param cacheattr The new cache attribute.
9340  */
9341 MARK_AS_PMAP_TEXT static void
9342 pmap_update_pp_attr_wimg_bits_locked(unsigned int pai, unsigned int cacheattr)
9343 {
9344 	pvh_assert_locked(pai);
9345 
9346 	pp_attr_t pp_attr_current, pp_attr_template;
9347 	do {
9348 		pp_attr_current = pp_attr_table[pai];
9349 		pp_attr_template = (pp_attr_current & ~PP_ATTR_WIMG_MASK) | PP_ATTR_WIMG(cacheattr);
9350 
9351 		/**
9352 		 * WIMG bits should only be updated under the PVH lock, but we should do
9353 		 * this in a CAS loop to avoid losing simultaneous updates to other bits like refmod.
9354 		 */
9355 	} while (!OSCompareAndSwap16(pp_attr_current, pp_attr_template, &pp_attr_table[pai]));
9356 }
9357 
9358 /**
9359  * Structure for tracking where we are during the collection of mappings for batch
9360  * cache attribute updates.
9361  *
9362  * @note We need to track where in the per-cpu ops table we are filling the next mappings into,
9363  *       because the collection routine can return with a not completely filled ops table when
9364  *       it exhausts the PV list for a page. In such case, the remaining slots in the ops table
9365  *       will be used for mappings of the next page.
9366  *
9367  * @note We also need to record where we are in the PV list, because the collection routine can
9368  *       also return when the ops table is filled but it's still in the middle of the PV list.
9369  *       Those remaining items in the PV list need to be handled by the next batch operation in
9370  *       a new ops table.
9371  */
9372 typedef struct {
9373 	/* Where we are in the sptm ops table. */
9374 	unsigned int sptm_ops_index;
9375 
9376 	/**
9377 	 * The last collected physical address from the previous full ops array (and in turn, SPTM
9378 	 * call). This is used to know whether the SPTM call for the latest full ops table should
9379 	 * skip updating the PAPT mapping (seeing as the last call would have handled updating it).
9380 	 */
9381 	pmap_paddr_t last_table_last_papt_pa;
9382 
9383 	/**
9384 	 * Where we are in the pv list.
9385 	 *
9386 	 * When ptep is non-null, there's only one mapping to the page and the ptep is the address
9387 	 * of it.
9388 	 *
9389 	 * When pvep is non-null, there's more than one mapping and the mappings are tracked by the
9390 	 * PV list.
9391 	 *
9392 	 * When they are both null, it indicates we are collecting for a new page and the collection
9393 	 * function will initialize them to be one of the two states above.
9394 	 *
9395 	 * It is undefined when they are both non-null.
9396 	 */
9397 	pt_entry_t *ptep;
9398 	pv_entry_t *pvep;
9399 	unsigned int pve_ptep_idx;
9400 } pmap_sptm_update_cache_attr_ops_collect_state_t;
9401 
9402 /**
9403  * Reports whether there is any pending ops in an sptm cache attr ops table.
9404  *
9405  * @param state A pmap_sptm_update_cache_attr_ops_collect_state_t structure.
9406  *
9407  * @return True if there's any outstanding cache attr op.
9408  *         False otherwise.
9409  */
9410 static inline bool
9411 pmap_is_sptm_update_cache_attr_ops_pending(pmap_sptm_update_cache_attr_ops_collect_state_t state)
9412 {
9413 	return state.sptm_ops_index > 0;
9414 }
9415 
9416 /**
9417  * Struct for encoding the collection status into pmap_sptm_update_cache_attr_ops_collect()'s
9418  * return value indicating what kind of attention it needs.
9419  */
9420 typedef enum {
9421 	OPS_COLLECT_NOTHING = 0x0,
9422 
9423 	/* The ops table is full, and the caller should commit the table to SPTM. */
9424 	OPS_COLLECT_RETURN_FULL_TABLE = 0x1,
9425 
9426 	/**
9427 	 * The page has its mappings completely collected, and the caller should
9428 	 * pass in a new page next time.
9429 	 */
9430 	OPS_COLLECT_RETURN_COMPLETED_PAGE = 0x2,
9431 } pmap_sptm_update_cache_attr_ops_collect_return_t;
9432 
9433 /**
9434  * Collects mappings of a physical page into an SPTM ops table for cache attribute updates.
9435  *
9436  * @note This routine returns either when the ops table is full or the page represented by
9437  *       pa has no more mapping to collect. The caller should call this routine again with
9438  *       a fresh ops table, or a new page, or both, depending on the return code.
9439  *
9440  * @note The PVH lock needs to be held for pa.
9441  *
9442  * @param state Tracks the state of PV list traversal and SPTM ops table filling. It is used
9443  *              by this routine to save the progress of the collection.
9444  * @param sptm_ops Pointer to the SPTM ops table.
9445  * @param pa The physical address whose mappings are to be collected.
9446  * @param attributes The new cache attributes.
9447  *
9448  * @return A pmap_sptm_update_cache_attr_ops_collect_return_t that encodes what the caller
9449  *         should do before calling this routine again. See the inline comments around
9450  *         pmap_sptm_update_cache_attr_ops_collect_return_t for details.
9451  */
9452 static pmap_sptm_update_cache_attr_ops_collect_return_t
9453 pmap_sptm_update_cache_attr_ops_collect(
9454 	pmap_sptm_update_cache_attr_ops_collect_state_t *state,
9455 	sptm_update_disjoint_multipage_op_t *sptm_ops,
9456 	pmap_paddr_t pa,
9457 	unsigned int attributes)
9458 {
9459 	if (state == NULL || sptm_ops == NULL) {
9460 		panic("%s: unexpected null arguments - state: %p, sptm_ops: %p", __func__, state, sptm_ops);
9461 	}
9462 
9463 	PMAP_TRACE(2, PMAP_CODE(PMAP__COLLECT_CACHE_OPS) | DBG_FUNC_START, pa, attributes, state->sptm_ops_index);
9464 
9465 	/* Copy the states into local variables. */
9466 	unsigned int sptm_ops_index = state->sptm_ops_index;
9467 	pmap_paddr_t last_table_last_papt_pa = state->last_table_last_papt_pa;
9468 	pv_entry_t *pvep = state->pvep;
9469 	pt_entry_t *ptep = state->ptep;
9470 	unsigned int pve_ptep_idx = state->pve_ptep_idx;
9471 
9472 	unsigned int pai = pa_index(pa);
9473 
9474 	/* We should at least have one free slot in the ops table. */
9475 	assert(sptm_ops_index < SPTM_MAPPING_LIMIT);
9476 
9477 	/* The PVH lock for pa has to be locked. */
9478 	pvh_assert_locked(pai);
9479 
9480 	/* If pvep and ptep are both null in the state, it's a new page. Initialize the states. */
9481 	if (pvep == PV_ENTRY_NULL && ptep == PT_ENTRY_NULL) {
9482 		const uintptr_t pvh = pai_to_pvh(pai);
9483 		if (pvh_test_type(pvh, PVH_TYPE_PVEP)) {
9484 			ptep = PT_ENTRY_NULL;
9485 			pvep = pvh_pve_list(pvh);
9486 			pve_ptep_idx = 0;
9487 		} else if (pvh_test_type(pvh, PVH_TYPE_PTEP)) {
9488 			ptep  = pvh_ptep(pvh);
9489 			pvep = PV_ENTRY_NULL;
9490 			pve_ptep_idx = 0;
9491 		}
9492 	}
9493 
9494 	/**
9495 	 * The first entry filled in is always the PAPT header entry:
9496 	 *
9497 	 * 1) In the case of a fresh ops table, the first entry has to be a PAPT header.
9498 	 * 2) In the case of a fresh page, we need to insert a new PAPT header to request
9499 	 *    SPTM to operate on a new page.
9500 	 *
9501 	 * Remember the index of the PAPT header here so that we can update the number
9502 	 * of mappings field later when we finish collecting.
9503 	 */
9504 	const unsigned int papt_sptm_ops_index = sptm_ops_index;
9505 	unsigned int num_mappings = 0;
9506 
9507 	/* Assemble the PTE template for the PAPT mapping. */
9508 	const vm_address_t kva = phystokv(pa);
9509 	const pt_entry_t *papt_ptep = pmap_pte(kernel_pmap, kva);
9510 
9511 	pt_entry_t template = os_atomic_load(papt_ptep, relaxed);
9512 	template &= ~(ARM_PTE_ATTRINDXMASK | ARM_PTE_SHMASK);
9513 	template |= wimg_to_pte(attributes, pa);
9514 
9515 	/* Fill in the PAPT header entry. */
9516 	sptm_ops[papt_sptm_ops_index].per_paddr_header.paddr = pa;
9517 	sptm_ops[papt_sptm_ops_index].per_paddr_header.papt_pte_template = template;
9518 	sptm_ops[papt_sptm_ops_index].per_paddr_header.options = SPTM_UPDATE_SH | SPTM_UPDATE_MAIR | SPTM_UPDATE_DEFER_TLBI;
9519 
9520 	if ((papt_sptm_ops_index == 0) && (pa == last_table_last_papt_pa)) {
9521 		/**
9522 		 * If the previous SPTM call was made with an ops table that already included
9523 		 * updating the PA of the page that this table starts with, then we can assume
9524 		 * that call already updated the PAPT and we can safely skip it in this
9525 		 * upcoming one.
9526 		 */
9527 		sptm_ops[0].per_paddr_header.options |= SPTM_UPDATE_SKIP_PAPT;
9528 	}
9529 
9530 	sptm_ops_index++;
9531 
9532 	/**
9533 	 * Main loop for collecting the mappings into the ops table. It terminates either
9534 	 * when the ops table is full or the PV list is exhausted.
9535 	 */
9536 	while ((sptm_ops_index < SPTM_MAPPING_LIMIT) && (pvep != PV_ENTRY_NULL || ptep != PT_ENTRY_NULL)) {
9537 		/**
9538 		 * Update ptep. There are really two cases here:
9539 		 *
9540 		 * 1) pvep is PV_ENTRY_NULL. In this case, ptep holds the pointer to
9541 		 *    the only mapping to the page.
9542 		 * 2) pvep is not PV_ENTRY_NULL. In such case, ptep is updated accroding to
9543 		 *    pvep and pve_ptep_idx.
9544 		 */
9545 		if (pvep != PV_ENTRY_NULL) {
9546 			ptep = pve_get_ptep(pvep, pve_ptep_idx);
9547 
9548 			/* This pve is empty, so skip to next one. */
9549 			if (ptep == PT_ENTRY_NULL) {
9550 				goto sucaoc_skip_pte;
9551 			}
9552 		}
9553 
9554 #ifdef PVH_FLAG_IOMMU
9555 		/* Skip IOMMU pteps. */
9556 		if (pvh_ptep_is_iommu(ptep)) {
9557 			goto sucaoc_skip_pte;
9558 		}
9559 #endif
9560 		/* Assemble the PTE template for the mapping. */
9561 		const vm_address_t va = ptep_get_va(ptep);
9562 		const pmap_t pmap = ptep_get_pmap(ptep);
9563 
9564 		template = os_atomic_load(ptep, relaxed);
9565 		template &= ~(ARM_PTE_ATTRINDXMASK | ARM_PTE_SHMASK);
9566 		template |= pmap_get_pt_ops(pmap)->wimg_to_pte(attributes, pa);
9567 
9568 		/* Fill into the ops table. */
9569 		sptm_ops[sptm_ops_index].disjoint_op.root_pt_paddr = pmap->ttep;
9570 		sptm_ops[sptm_ops_index].disjoint_op.vaddr = va;
9571 		sptm_ops[sptm_ops_index].disjoint_op.pte_template = template;
9572 
9573 		/* Move the sptm ops table cursor. */
9574 		sptm_ops_index++;
9575 
9576 		/* Increment the mappings counter. */
9577 		num_mappings++;
9578 
9579 sucaoc_skip_pte:
9580 		/**
9581 		 * Reset ptep to PT_ENTRY_NULL to keep the loop precondition of either ptep
9582 		 * or pvep is nonnull (not both, not neither) true.
9583 		 */
9584 		ptep = PT_ENTRY_NULL;
9585 
9586 		/* Advance to next pvep if we have exhausted the pteps in it. */
9587 		if ((pvep != PV_ENTRY_NULL) && (++pve_ptep_idx == PTE_PER_PVE)) {
9588 			pve_ptep_idx = 0;
9589 			pvep = pve_next(pvep);
9590 		}
9591 	}
9592 
9593 	/* Update the PAPT header for the number of mappings. */
9594 	sptm_ops[papt_sptm_ops_index].per_paddr_header.num_mappings = num_mappings;
9595 
9596 	const bool full_table = (sptm_ops_index >= SPTM_MAPPING_LIMIT);
9597 	const bool collection_done_for_page = (pvep == PV_ENTRY_NULL && ptep == PT_ENTRY_NULL);
9598 
9599 	/**
9600 	 * The ops table is full, so the caller should now invoke the SPTM before calling
9601 	 * into this function again.
9602 	 */
9603 	if (full_table) {
9604 		/* Update last_table_last_papt_pa to be the pa collected in this call. */
9605 		last_table_last_papt_pa = pa;
9606 
9607 		/* Reset sptm_ops_index. */
9608 		sptm_ops_index = 0;
9609 	}
9610 
9611 	/* Copy the updated collection states back to the parameter structure. */
9612 	state->sptm_ops_index = sptm_ops_index;
9613 	state->last_table_last_papt_pa = last_table_last_papt_pa;
9614 	state->pvep = pvep;
9615 	state->ptep = ptep;
9616 	state->pve_ptep_idx = pve_ptep_idx;
9617 
9618 	/* Assemble the return value. */
9619 	pmap_sptm_update_cache_attr_ops_collect_return_t retval = OPS_COLLECT_NOTHING;
9620 
9621 	if (full_table) {
9622 		retval |= OPS_COLLECT_RETURN_FULL_TABLE;
9623 	}
9624 
9625 	if (collection_done_for_page) {
9626 		retval |= OPS_COLLECT_RETURN_COMPLETED_PAGE;
9627 	}
9628 
9629 	PMAP_TRACE(2, PMAP_CODE(PMAP__COLLECT_CACHE_OPS) | DBG_FUNC_END, pa, attributes, sptm_ops_index);
9630 
9631 	return retval;
9632 }
9633 
9634 /* At least one PAPT header plus one mapping. */
9635 static_assert(SPTM_MAPPING_LIMIT >= 2);
9636 
9637 /**
9638  * Returns if a cache attribute is allowed (on managed pages).
9639  *
9640  * @param attributes A 32-bit value whose VM_WIMG_MASK bits represent the
9641  *                   cache attribute.
9642  *
9643  * @return True if the cache attribute is allowed on managed pages.
9644  *         False otherwise.
9645  */
9646 static bool
9647 pmap_is_cache_attribute_allowed(unsigned int attributes)
9648 {
9649 	if (pmap_panic_dev_wimg_on_managed) {
9650 		switch (attributes & VM_WIMG_MASK) {
9651 		/* supported on DRAM, but slow, so we disallow */
9652 		case VM_WIMG_IO:                        // nGnRnE
9653 		case VM_WIMG_POSTED:                    // nGnRE
9654 
9655 		/* unsupported on DRAM */
9656 		case VM_WIMG_POSTED_REORDERED:          // nGRE
9657 		case VM_WIMG_POSTED_COMBINED_REORDERED: // GRE
9658 			return false;
9659 
9660 		default:
9661 			return true;
9662 		}
9663 	}
9664 
9665 	return true;
9666 }
9667 
9668 /**
9669  * Batch updates the cache attributes of a list of pages in three passes.
9670  *
9671  * In pass one, the pp_attr_table and the pte are updated (by SPTM) for the pages in the list.
9672  * In pass two, TLB entries are flushed for each page in the list if necessary.
9673  * In pass three, caches are cleaned for each page in the list if necessary.
9674  *
9675  * @param page_list List of pages to be updated.
9676  * @param cacheattr The new cache attributes.
9677  * @param update_attr_table Whether the pp_attr_table should be updated. This is useful for compressor
9678  *                          pages where it's desired to keep the old WIMG bits.
9679  */
9680 void
9681 pmap_batch_set_cache_attributes_internal(
9682 	const unified_page_list_t *page_list,
9683 	unsigned int cacheattr,
9684 	bool update_attr_table)
9685 {
9686 	bool tlb_flush_pass_needed = false;
9687 	bool rt_cache_flush_pass_needed = false;
9688 	bool preemption_disabled = false;
9689 
9690 	PMAP_TRACE(2, PMAP_CODE(PMAP__BATCH_UPDATE_CACHING), page_list, cacheattr, 0xCECC0DE1);
9691 
9692 	pmap_sptm_percpu_data_t *sptm_pcpu = NULL;
9693 	sptm_update_disjoint_multipage_op_t *sptm_ops = NULL;
9694 
9695 	pmap_sptm_update_cache_attr_ops_collect_state_t state = {0};
9696 
9697 	unified_page_list_iterator_t iter;
9698 
9699 	for (unified_page_list_iterator_init(page_list, &iter);
9700 	    !unified_page_list_iterator_end(&iter);
9701 	    unified_page_list_iterator_next(&iter)) {
9702 		bool is_fictitious = false;
9703 		const ppnum_t pn = unified_page_list_iterator_page(&iter, &is_fictitious);
9704 		const pmap_paddr_t paddr = ptoa(pn);
9705 
9706 		/**
9707 		 * Skip if the page is not managed.
9708 		 *
9709 		 * We don't panic here because sometimes the user just blindly pass in
9710 		 * pages that are not managed. We need to handle that gracefully.
9711 		 */
9712 		if (__improbable(!pa_valid(paddr) || is_fictitious)) {
9713 			continue;
9714 		}
9715 
9716 		const unsigned int pai = pa_index(paddr);
9717 		locked_pvh_t locked_pvh = {.pvh = 0};
9718 
9719 		if (pmap_is_sptm_update_cache_attr_ops_pending(state)) {
9720 			/**
9721 			 * If we're partway through processing a multi-page batched call,
9722 			 * preemption will already be disabled so we can't simply call
9723 			 * pvh_lock() which may block.  Instead, we first try to acquire
9724 			 * the lock without waiting, which in most cases should succeed.
9725 			 * If it fails, we submit the pending batched operations to re-
9726 			 * enable preemption and then acquire the lock normally.
9727 			 */
9728 			locked_pvh = pvh_try_lock(pai);
9729 			if (__improbable(!pvh_try_lock_success(&locked_pvh))) {
9730 				assert(preemption_disabled);
9731 				const sptm_return_t sptm_ret = sptm_update_disjoint_multipage(sptm_pcpu->sptm_ops_pa, state.sptm_ops_index);
9732 				pmap_retype_epoch_exit();
9733 				enable_preemption();
9734 				preemption_disabled = false;
9735 				if (sptm_ret == SPTM_UPDATE_DELAYED_TLBI) {
9736 					tlb_flush_pass_needed = true;
9737 				}
9738 				state.sptm_ops_index = 0;
9739 				locked_pvh = pvh_lock(pai);
9740 			}
9741 		} else {
9742 			locked_pvh = pvh_lock(pai);
9743 		}
9744 		assert(locked_pvh.pvh != 0);
9745 
9746 		const pp_attr_t pp_attr_current = pp_attr_table[pai];
9747 
9748 		unsigned int wimg_bits_prev = VM_WIMG_DEFAULT;
9749 		if (pp_attr_current & PP_ATTR_WIMG_MASK) {
9750 			wimg_bits_prev = pp_attr_current & PP_ATTR_WIMG_MASK;
9751 		}
9752 
9753 		const pp_attr_t pp_attr_template = (pp_attr_current & ~PP_ATTR_WIMG_MASK) | PP_ATTR_WIMG(cacheattr);
9754 
9755 		unsigned int wimg_bits_new = VM_WIMG_DEFAULT;
9756 		if (pp_attr_template & PP_ATTR_WIMG_MASK) {
9757 			wimg_bits_new = pp_attr_template & PP_ATTR_WIMG_MASK;
9758 		}
9759 
9760 		/**
9761 		 * When update_attr_table is false, we know that wimg_bits_prev read from pp_attr_table is not to be trusted,
9762 		 * and we should force update the cache attribute.
9763 		 */
9764 		const bool force_update = !update_attr_table;
9765 		/* Update the cache attributes in PTE and PP_ATTR table. */
9766 		if ((wimg_bits_new != wimg_bits_prev) || force_update) {
9767 			if (!pmap_is_cache_attribute_allowed(cacheattr)) {
9768 				panic("%s: trying to use unsupported VM_WIMG type for managed page, VM_WIMG=%x, pn=%#x",
9769 				    __func__, cacheattr & VM_WIMG_MASK, pn);
9770 			}
9771 
9772 			/* Update PP_ATTR_TABLE */
9773 			if (update_attr_table) {
9774 				pmap_update_pp_attr_wimg_bits_locked(pai, cacheattr);
9775 			}
9776 
9777 			bool mapping_collection_done = false;
9778 			bool pvh_lock_sleep_mode_needed = false;
9779 			do {
9780 				if (__improbable(pvh_lock_sleep_mode_needed)) {
9781 					assert(!preemption_disabled);
9782 					pvh_lock_enter_sleep_mode(&locked_pvh);
9783 					pvh_lock_sleep_mode_needed = false;
9784 				}
9785 
9786 				/* Disable preemption to use the per-CPU structure safely. */
9787 				if (!preemption_disabled) {
9788 					preemption_disabled = true;
9789 					disable_preemption();
9790 					/**
9791 					 * Enter the retype epoch while we gather the disjoint update arguments
9792 					 * and issue the SPTM call.  Since this operation may cover multiple physical
9793 					 * pages, we may construct the argument array and invoke the SPTM without holding
9794 					 * all relevant PVH locks, we need to record that we are collecting and modifying
9795 					 * mapping state so that e.g. pmap_page_protect() does not attempt to retype the
9796 					 * underlying pages and pmap_remove() does not attempt to free the page tables
9797 					 * used for these mappings without first draining our epoch.
9798 					 */
9799 					pmap_retype_epoch_enter();
9800 
9801 					sptm_pcpu = PERCPU_GET(pmap_sptm_percpu);
9802 					sptm_ops = (sptm_update_disjoint_multipage_op_t *) sptm_pcpu->sptm_ops;
9803 				}
9804 
9805 				/* The return value indicates if we should call into SPTM in this iteration. */
9806 				pmap_sptm_update_cache_attr_ops_collect_return_t retval =
9807 				    pmap_sptm_update_cache_attr_ops_collect(&state, sptm_ops, paddr, cacheattr);
9808 
9809 				/* The collection routine should only return if it needs attention. */
9810 				assert(retval != OPS_COLLECT_NOTHING);
9811 
9812 				/* Gather information for next step from the return value. */
9813 				mapping_collection_done = retval & OPS_COLLECT_RETURN_COMPLETED_PAGE;
9814 				const bool call_sptm = retval & OPS_COLLECT_RETURN_FULL_TABLE;
9815 
9816 				if (call_sptm) {
9817 					/* Call into SPTM with this SPTM ops table. */
9818 					sptm_return_t sptm_ret = sptm_update_disjoint_multipage(sptm_pcpu->sptm_ops_pa, SPTM_MAPPING_LIMIT);
9819 					/**
9820 					 * We may be submitting the batch and exiting the epoch partway through
9821 					 * processing the PV list for a page.  That's fine, because in that case we'll
9822 					 * hold the PV lock for that page, which will prevent mappings of that page from
9823 					 * being disconnected and will prevent the completion of pmap_remove() against
9824 					 * any of those mappings, thus also guaranteeing the relevant page table pages
9825 					 * can't be freed.  The epoch still protects mappings for any prior page in
9826 					 * the batch, whose PV locks are no longer held.
9827 					 */
9828 					pmap_retype_epoch_exit();
9829 					/**
9830 					 * Balance out the explicit disable_preemption() made either at the beginning of
9831 					 * the function or on a prior iteration of the loop that placed the PVH lock in
9832 					 * sleep mode.  Note that enable_preemption() decrements a per-thread counter,
9833 					 * so if we still happen to hold the PVH lock in spin mode preemption won't
9834 					 * actually be re-enabled until we switch the lock over to sleep mode on
9835 					 * the next iteration.
9836 					 */
9837 					enable_preemption();
9838 					preemption_disabled = false;
9839 					pvh_lock_sleep_mode_needed = true;
9840 
9841 					if (sptm_ret == SPTM_UPDATE_DELAYED_TLBI) {
9842 						tlb_flush_pass_needed = true;
9843 					}
9844 				}
9845 
9846 				/* We cannot be in a situation where we didn't call into SPTM while also having not finished walking the pv list. */
9847 				assert(call_sptm || mapping_collection_done);
9848 			} while (!mapping_collection_done);
9849 
9850 			/**
9851 			 * We could technically force the cache flush pass here when force_update is true, but
9852 			 * since the compressor mapping/unmapping path handles cache flushing itself, it's fine
9853 			 * leaving this as is.
9854 			 */
9855 			if (wimg_bits_new == VM_WIMG_RT && wimg_bits_prev != VM_WIMG_RT) {
9856 				rt_cache_flush_pass_needed = true;
9857 			}
9858 		}
9859 
9860 		pvh_unlock(&locked_pvh);
9861 	}
9862 
9863 	if (pmap_is_sptm_update_cache_attr_ops_pending(state)) {
9864 		assert(preemption_disabled);
9865 		sptm_return_t sptm_ret = sptm_update_disjoint_multipage(sptm_pcpu->sptm_ops_pa, state.sptm_ops_index);
9866 		pmap_retype_epoch_exit();
9867 		if (sptm_ret == SPTM_UPDATE_DELAYED_TLBI) {
9868 			tlb_flush_pass_needed = true;
9869 		}
9870 
9871 		/**
9872 		 * This is the last sptm_update_cache_attr() call whatsoever, so it's
9873 		 * okay not to update the state variables.
9874 		 */
9875 
9876 		enable_preemption();
9877 	} else if (preemption_disabled) {
9878 		pmap_retype_epoch_exit();
9879 		enable_preemption();
9880 	}
9881 
9882 	if (tlb_flush_pass_needed) {
9883 		/* Sync the PTE writes before potential TLB/Cache flushes. */
9884 		FLUSH_PTE_STRONG();
9885 
9886 		/**
9887 		 * Pass 2: for each physical page and for each mapping, we need to flush
9888 		 * the TLB for it.
9889 		 */
9890 		PMAP_TRACE(2, PMAP_CODE(PMAP__BATCH_UPDATE_CACHING), page_list, cacheattr, 0xCECC0DE2);
9891 		for (unified_page_list_iterator_init(page_list, &iter);
9892 		    !unified_page_list_iterator_end(&iter);
9893 		    unified_page_list_iterator_next(&iter)) {
9894 			bool is_fictitious = false;
9895 			const ppnum_t pn = unified_page_list_iterator_page(&iter, &is_fictitious);
9896 			const pmap_paddr_t paddr = ptoa(pn);
9897 
9898 			if (__improbable(!pa_valid(paddr) || is_fictitious)) {
9899 				continue;
9900 			}
9901 
9902 			pmap_flush_tlb_for_paddr_async(paddr);
9903 		}
9904 
9905 #if HAS_FEAT_XS
9906 		/* With FEAT_XS, ordinary DSBs drain the prefetcher. */
9907 		arm64_sync_tlb(false);
9908 #else
9909 		/**
9910 		 * For targets that distinguish between mild and strong DSB, mild DSB
9911 		 * will not drain the prefetcher.  This can lead to prefetch-driven
9912 		 * cache fills that defeat the uncacheable requirement of the RT memory type.
9913 		 * In those cases, strong DSB must instead be employed to drain the prefetcher.
9914 		 */
9915 		arm64_sync_tlb((cacheattr & VM_WIMG_MASK) == VM_WIMG_RT);
9916 #endif
9917 	}
9918 
9919 	if (rt_cache_flush_pass_needed) {
9920 		/* Pass 3: Flush the cache if the page is recently set to RT */
9921 		PMAP_TRACE(2, PMAP_CODE(PMAP__BATCH_UPDATE_CACHING), page_list, cacheattr, 0xCECC0DE3);
9922 		/**
9923 		 * We disable preemption to ensure we are not preempted
9924 		 * in the state where DC by VA instructions remain enabled.
9925 		 */
9926 		disable_preemption();
9927 
9928 		assert(get_preemption_level() > 0);
9929 
9930 #if defined(APPLE_ARM64_ARCH_FAMILY) && !APPLEVIRTUALPLATFORM
9931 		/**
9932 		 * On APPLEVIRTUALPLATFORM, HID register accesses cause a synchronous exception
9933 		 * and the host will handle cache maintenance for it. So we don't need to
9934 		 * worry about enabling the ops here for AVP.
9935 		 */
9936 		enable_dc_mva_ops();
9937 #endif /* defined(APPLE_ARM64_ARCH_FAMILY) && !APPLEVIRTUALPLATFORM */
9938 		/**
9939 		 * DMB should be sufficient to ensure prior accesses to the memory in question are
9940 		 * correctly ordered relative to the upcoming cache maintenance operations.
9941 		 */
9942 		__builtin_arm_dmb(DMB_SY);
9943 
9944 		for (unified_page_list_iterator_init(page_list, &iter);
9945 		    !unified_page_list_iterator_end(&iter);) {
9946 			bool is_fictitious = false;
9947 			const ppnum_t pn = unified_page_list_iterator_page(&iter, &is_fictitious);
9948 			const pmap_paddr_t paddr = ptoa(pn);
9949 
9950 			if (__improbable(!pa_valid(paddr) || is_fictitious)) {
9951 				unified_page_list_iterator_next(&iter);
9952 				continue;
9953 			}
9954 
9955 			CleanPoC_DcacheRegion_Force_nopreempt_nohid_nobarrier(phystokv(paddr), PAGE_SIZE);
9956 
9957 			unified_page_list_iterator_next(&iter);
9958 			if (__improbable(pmap_pending_preemption() && !unified_page_list_iterator_end(&iter))) {
9959 				__builtin_arm_dsb(DSB_SY);
9960 #if defined(APPLE_ARM64_ARCH_FAMILY) && !APPLEVIRTUALPLATFORM
9961 				disable_dc_mva_ops();
9962 #endif /* defined(APPLE_ARM64_ARCH_FAMILY) && !APPLEVIRTUALPLATFORM */
9963 				enable_preemption();
9964 				assert(preemption_enabled());
9965 				disable_preemption();
9966 #if defined(APPLE_ARM64_ARCH_FAMILY) && !APPLEVIRTUALPLATFORM
9967 				enable_dc_mva_ops();
9968 #endif /* defined(APPLE_ARM64_ARCH_FAMILY) && !APPLEVIRTUALPLATFORM */
9969 			}
9970 		}
9971 
9972 		/* Issue DSB to ensure cache maintenance is fully complete before subsequent accesses. */
9973 		__builtin_arm_dsb(DSB_SY);
9974 #if defined(APPLE_ARM64_ARCH_FAMILY) && !APPLEVIRTUALPLATFORM
9975 		disable_dc_mva_ops();
9976 #endif /* defined(APPLE_ARM64_ARCH_FAMILY) && !APPLEVIRTUALPLATFORM */
9977 
9978 		enable_preemption();
9979 	}
9980 
9981 	PMAP_TRACE(2, PMAP_CODE(PMAP__BATCH_UPDATE_CACHING), page_list, cacheattr, 0xCECC0DE4);
9982 }
9983 
9984 /**
9985  * Batch updates the cache attributes of a list of pages. This is a wrapper for
9986  * the ppl call on PPL-enabled platforms or the _internal helper on other platforms.
9987  *
9988  * @param page_list List of pages to be updated.
9989  * @param cacheattr The new cache attribute.
9990  */
9991 void
9992 pmap_batch_set_cache_attributes(
9993 	const unified_page_list_t *page_list,
9994 	unsigned int cacheattr)
9995 {
9996 	PMAP_TRACE(2, PMAP_CODE(PMAP__BATCH_UPDATE_CACHING) | DBG_FUNC_START, page_list, cacheattr, 0xCECC0DE0);
9997 
9998 	/* Verify we are being called from a preemptible context. */
9999 	pmap_verify_preemptible();
10000 
10001 	pmap_batch_set_cache_attributes_internal(page_list, cacheattr, true);
10002 
10003 	PMAP_TRACE(2, PMAP_CODE(PMAP__BATCH_UPDATE_CACHING) | DBG_FUNC_END, page_list, cacheattr, 0xCECC0DEF);
10004 }
10005 
10006 MARK_AS_PMAP_TEXT void
10007 pmap_set_cache_attributes_internal(
10008 	ppnum_t pn,
10009 	unsigned int cacheattr,
10010 	bool update_attr_table)
10011 {
10012 	upl_page_info_t single_page_upl = { .phys_addr = pn };
10013 	const unified_page_list_t page_list = {
10014 		.upl = {.upl_info = &single_page_upl, .upl_size = 1},
10015 		.type = UNIFIED_PAGE_LIST_TYPE_UPL_ARRAY,
10016 	};
10017 
10018 	pmap_batch_set_cache_attributes_internal(&page_list, cacheattr, update_attr_table);
10019 }
10020 
10021 void
10022 pmap_set_cache_attributes(
10023 	ppnum_t pn,
10024 	unsigned int cacheattr)
10025 {
10026 	pmap_set_cache_attributes_internal(pn, cacheattr, true);
10027 }
10028 
10029 void
10030 pmap_create_commpages(vm_map_address_t *kernel_data_addr, vm_map_address_t *kernel_text_addr,
10031     vm_map_address_t *kernel_ro_data_addr, vm_map_address_t *user_text_addr)
10032 {
10033 	pmap_paddr_t data_pa = 0; // data address
10034 	pmap_paddr_t ro_data_pa = 0; // kernel read-only data address
10035 	pmap_paddr_t text_pa = 0; // text address
10036 
10037 	*kernel_data_addr = 0;
10038 	*kernel_text_addr = 0;
10039 	*user_text_addr = 0;
10040 
10041 	kern_return_t kr = pmap_page_alloc(&data_pa, PMAP_PAGE_ALLOCATE_NONE);
10042 	assert(kr == KERN_SUCCESS);
10043 
10044 	kr = pmap_page_alloc(&ro_data_pa, PMAP_PAGE_ALLOCATE_NONE);
10045 	assert(kr == KERN_SUCCESS);
10046 
10047 #if CONFIG_ARM_PFZ
10048 	kr = pmap_page_alloc(&text_pa, PMAP_PAGE_ALLOCATE_NONE);
10049 	assert(kr == KERN_SUCCESS);
10050 
10051 	/**
10052 	 *  User mapping of comm page text section for 64 bit mapping only
10053 	 *
10054 	 * We don't insert it into the 32 bit mapping because we don't want 32 bit
10055 	 * user processes to get this page mapped in, they should never call into
10056 	 * this page.
10057 	 *
10058 	 * The data comm page is in a pre-reserved L3 VA range and the text commpage
10059 	 * is slid in the same L3 as the data commpage.  It is either outside the
10060 	 * max of user VA or is pre-reserved in vm_map_exec(). This means that
10061 	 * it is reserved and unavailable to mach VM for future mappings.
10062 	 */
10063 	const int num_ptes = pt_attr_leaf_size(native_pt_attr) >> PTE_SHIFT;
10064 
10065 	do {
10066 		const int text_leaf_index = random() % num_ptes;
10067 
10068 		/**
10069 		 * Generate a VA for the commpage text with the same root and twig index as data
10070 		 * comm page, but with new leaf index we've just generated.
10071 		 */
10072 		commpage_text_user_va = (_COMM_PAGE64_BASE_ADDRESS & ~pt_attr_leaf_index_mask(native_pt_attr));
10073 		commpage_text_user_va |= (text_leaf_index << pt_attr_leaf_shift(native_pt_attr));
10074 	} while ((commpage_text_user_va == _COMM_PAGE64_BASE_ADDRESS) ||
10075 	    (commpage_text_user_va == _COMM_PAGE64_RO_ADDRESS)); // Try again if we collide (should be unlikely)
10076 
10077 	*user_text_addr = commpage_text_user_va;
10078 	*kernel_text_addr = phystokv(text_pa);
10079 #endif
10080 
10081 	/* For manipulation in kernel, go straight to physical page */
10082 	commpage_data_pa = data_pa;
10083 	*kernel_data_addr = phystokv(data_pa);
10084 	assert(commpage_ro_data_pa == 0);
10085 	commpage_ro_data_pa = ro_data_pa;
10086 	*kernel_ro_data_addr = phystokv(ro_data_pa);
10087 	assert(commpage_text_pa == 0);
10088 	commpage_text_pa = text_pa;
10089 }
10090 
10091 
10092 /*
10093  * Asserts to ensure that the TTEs we nest to map the shared page do not overlap
10094  * with user controlled TTEs for regions that aren't explicitly reserved by the
10095  * VM (e.g., _COMM_PAGE64_NESTING_START/_COMM_PAGE64_BASE_ADDRESS).
10096  */
10097 #if (ARM_PGSHIFT == 14)
10098 /**
10099  * Ensure that 64-bit devices with 32-bit userspace VAs (arm64_32) can nest the
10100  * commpage completely above the maximum 32-bit userspace VA.
10101  */
10102 static_assert((_COMM_PAGE32_BASE_ADDRESS & ~ARM_TT_L2_OFFMASK) >= VM_MAX_ADDRESS);
10103 static_assert(_COMM_PAGE64_NESTING_START == SPTM_ARM64_COMMPAGE_REGION_START);
10104 static_assert(_COMM_PAGE64_NESTING_SIZE == SPTM_ARM64_COMMPAGE_REGION_SIZE);
10105 
10106 /**
10107  * Normally there'd be an assert to check that 64-bit devices with 64-bit
10108  * userspace VAs can nest the commpage completely above the maximum 64-bit
10109  * userpace VA, but that technically isn't true on macOS. On those systems, the
10110  * commpage lives within the userspace VA range, but is protected by the VM as
10111  * a reserved region (see vm_reserved_regions[] definition for more info).
10112  */
10113 
10114 #elif (ARM_PGSHIFT == 12)
10115 /**
10116  * Ensure that 64-bit devices using 4K pages can nest the commpage completely
10117  * above the maximum userspace VA.
10118  */
10119 static_assert((_COMM_PAGE64_BASE_ADDRESS & ~ARM_TT_L1_OFFMASK) >= MACH_VM_MAX_ADDRESS);
10120 #else
10121 #error Nested shared page mapping is unsupported on this config
10122 #endif
10123 
10124 MARK_AS_PMAP_TEXT kern_return_t
10125 pmap_insert_commpage_internal(
10126 	pmap_t pmap)
10127 {
10128 	kern_return_t kr = KERN_SUCCESS;
10129 	vm_offset_t commpage_vaddr;
10130 	pt_entry_t *ttep;
10131 	pmap_paddr_t commpage_table = commpage_default_table;
10132 
10133 	/* Validate the pmap input before accessing its data. */
10134 	validate_pmap_mutable(pmap);
10135 
10136 	const pt_attr_t * const pt_attr = pmap_get_pt_attr(pmap);
10137 	const unsigned int commpage_level = pt_attr_commpage_level(pt_attr);
10138 
10139 #if __ARM_MIXED_PAGE_SIZE__
10140 #if !__ARM_16K_PG__
10141 	/* The following code assumes that commpage_pmap_default is a 16KB pmap. */
10142 	#error "pmap_insert_commpage_internal requires a 16KB default kernel page size when __ARM_MIXED_PAGE_SIZE__ is enabled"
10143 #endif /* !__ARM_16K_PG__ */
10144 
10145 	/* Choose the correct shared page pmap to use. */
10146 	const uint64_t pmap_page_size = pt_attr_page_size(pt_attr);
10147 	if (pmap_page_size == 4096) {
10148 		if (pmap_is_64bit(pmap)) {
10149 			commpage_table = commpage_4k_table;
10150 		} else {
10151 			panic("32-bit commpage not currently supported for SPTM configurations");
10152 			//commpage_table = commpage32_4k_table;
10153 		}
10154 	} else if (pmap_page_size != 16384) {
10155 		panic("No commpage table exists for the wanted page size: %llu", pmap_page_size);
10156 	} else
10157 #endif /* __ARM_MIXED_PAGE_SIZE__ */
10158 	{
10159 		if (pmap_is_64bit(pmap)) {
10160 			commpage_table = commpage_default_table;
10161 		} else {
10162 			panic("32-bit commpage not currently supported for SPTM configurations");
10163 			//commpage_table = commpage32_default_table;
10164 		}
10165 	}
10166 
10167 #if _COMM_PAGE_AREA_LENGTH != PAGE_SIZE
10168 #error We assume a single page.
10169 #endif
10170 
10171 	if (pmap_is_64bit(pmap)) {
10172 		commpage_vaddr = _COMM_PAGE64_BASE_ADDRESS;
10173 	} else {
10174 		commpage_vaddr = _COMM_PAGE32_BASE_ADDRESS;
10175 	}
10176 
10177 
10178 	pmap_lock(pmap, PMAP_LOCK_SHARED);
10179 
10180 	/*
10181 	 * For 4KB pages, we either "nest" at the level one page table (1GB) or level
10182 	 * two (2MB) depending on the address space layout. For 16KB pages, each level
10183 	 * one entry is 64GB, so we must go to the second level entry (32MB) in order
10184 	 * to "nest".
10185 	 *
10186 	 * Note: This is not "nesting" in the shared cache sense. This definition of
10187 	 * nesting just means inserting pointers to pre-allocated tables inside of
10188 	 * the passed in pmap to allow us to share page tables (which map the shared
10189 	 * page) for every task. This saves at least one page of memory per process
10190 	 * compared to creating new page tables in every process for mapping the
10191 	 * shared page.
10192 	 */
10193 
10194 	/**
10195 	 * Allocate the twig page tables if needed, and slam a pointer to the shared
10196 	 * page's tables into place.
10197 	 */
10198 	while ((ttep = pmap_ttne(pmap, commpage_level, commpage_vaddr)) == TT_ENTRY_NULL) {
10199 		pmap_unlock(pmap, PMAP_LOCK_SHARED);
10200 
10201 		kr = pmap_expand(pmap, commpage_vaddr, 0, commpage_level);
10202 
10203 		if (kr != KERN_SUCCESS) {
10204 			panic("Failed to pmap_expand for commpage, pmap=%p", pmap);
10205 		}
10206 
10207 		pmap_lock(pmap, PMAP_LOCK_SHARED);
10208 	}
10209 
10210 	if (*ttep != ARM_PTE_EMPTY) {
10211 		panic("%s: Found something mapped at the commpage address?!", __FUNCTION__);
10212 	}
10213 
10214 	sptm_map_table(pmap->ttep, pt_attr_align_va(pt_attr, commpage_level, commpage_vaddr), (sptm_pt_level_t)commpage_level,
10215 	    (commpage_table & ARM_TTE_TABLE_MASK) | ARM_TTE_TYPE_TABLE | ARM_TTE_VALID);
10216 
10217 	pmap_unlock(pmap, PMAP_LOCK_SHARED);
10218 
10219 	return kr;
10220 }
10221 
10222 static void
10223 pmap_unmap_commpage(
10224 	pmap_t pmap)
10225 {
10226 	pt_entry_t *ptep;
10227 	vm_offset_t commpage_vaddr;
10228 
10229 	const pt_attr_t * const pt_attr = pmap_get_pt_attr(pmap);
10230 	const unsigned int commpage_level = pt_attr_commpage_level(pt_attr);
10231 	__assert_only pmap_paddr_t commpage_pa = commpage_data_pa;
10232 
10233 	if (pmap_is_64bit(pmap)) {
10234 		commpage_vaddr = _COMM_PAGE64_BASE_ADDRESS;
10235 	} else {
10236 		commpage_vaddr = _COMM_PAGE32_BASE_ADDRESS;
10237 	}
10238 
10239 
10240 	ptep = pmap_pte(pmap, commpage_vaddr);
10241 
10242 	if (ptep == NULL) {
10243 		return;
10244 	}
10245 
10246 	/* It had better be mapped to the shared page. */
10247 	if (pte_to_pa(*ptep) != commpage_pa) {
10248 		panic("%s: non-commpage PA 0x%llx mapped at VA 0x%llx in pmap %p; expected 0x%llx",
10249 		    __func__, (unsigned long long)pte_to_pa(*ptep), (unsigned long long)commpage_vaddr,
10250 		    pmap, (unsigned long long)commpage_pa);
10251 	}
10252 
10253 	sptm_unmap_table(pmap->ttep, pt_attr_align_va(pt_attr, commpage_level, commpage_vaddr), (sptm_pt_level_t)commpage_level);
10254 }
10255 
10256 void
10257 pmap_insert_commpage(
10258 	pmap_t pmap)
10259 {
10260 	pmap_insert_commpage_internal(pmap);
10261 }
10262 
10263 static boolean_t
10264 pmap_is_64bit(
10265 	pmap_t pmap)
10266 {
10267 	return pmap->is_64bit;
10268 }
10269 
10270 bool
10271 pmap_is_exotic(
10272 	pmap_t pmap __unused)
10273 {
10274 	return false;
10275 }
10276 
10277 
10278 /* ARMTODO -- an implementation that accounts for
10279  * holes in the physical map, if any.
10280  */
10281 boolean_t
10282 pmap_valid_page(
10283 	ppnum_t pn)
10284 {
10285 	return pa_valid(ptoa(pn));
10286 }
10287 
10288 boolean_t
10289 pmap_bootloader_page(
10290 	ppnum_t pn)
10291 {
10292 	pmap_paddr_t paddr = ptoa(pn);
10293 
10294 	if (pa_valid(paddr)) {
10295 		return FALSE;
10296 	}
10297 	pmap_io_range_t *io_rgn = pmap_find_io_attr(paddr);
10298 	return (io_rgn != NULL) && (io_rgn->wimg & PMAP_IO_RANGE_CARVEOUT);
10299 }
10300 
10301 MARK_AS_PMAP_TEXT boolean_t
10302 pmap_is_empty_internal(
10303 	pmap_t pmap,
10304 	vm_map_offset_t va_start,
10305 	vm_map_offset_t va_end)
10306 {
10307 	vm_map_offset_t block_start, block_end;
10308 	tt_entry_t *tte_p;
10309 
10310 	if (pmap == NULL) {
10311 		return TRUE;
10312 	}
10313 
10314 	validate_pmap(pmap);
10315 
10316 	__unused const pt_attr_t * const pt_attr = pmap_get_pt_attr(pmap);
10317 	unsigned int initial_not_in_kdp = not_in_kdp;
10318 
10319 	if ((pmap != kernel_pmap) && (initial_not_in_kdp)) {
10320 		pmap_lock(pmap, PMAP_LOCK_SHARED);
10321 	}
10322 
10323 
10324 	/* TODO: This will be faster if we increment ttep at each level. */
10325 	block_start = va_start;
10326 
10327 	while (block_start < va_end) {
10328 		pt_entry_t     *bpte_p, *epte_p;
10329 		pt_entry_t     *pte_p;
10330 
10331 		block_end = (block_start + pt_attr_twig_size(pt_attr)) & ~pt_attr_twig_offmask(pt_attr);
10332 		if (block_end > va_end) {
10333 			block_end = va_end;
10334 		}
10335 
10336 		tte_p = pmap_tte(pmap, block_start);
10337 		if ((tte_p != PT_ENTRY_NULL)
10338 		    && ((*tte_p & ARM_TTE_TYPE_MASK) == ARM_TTE_TYPE_TABLE)) {
10339 			pte_p = (pt_entry_t *) ttetokv(*tte_p);
10340 			bpte_p = &pte_p[pte_index(pt_attr, block_start)];
10341 			epte_p = &pte_p[pte_index(pt_attr, block_end)];
10342 
10343 			for (pte_p = bpte_p; pte_p < epte_p; pte_p++) {
10344 				if (*pte_p != ARM_PTE_EMPTY) {
10345 					if ((pmap != kernel_pmap) && (initial_not_in_kdp)) {
10346 						pmap_unlock(pmap, PMAP_LOCK_SHARED);
10347 					}
10348 					return FALSE;
10349 				}
10350 			}
10351 		}
10352 		block_start = block_end;
10353 	}
10354 
10355 	if ((pmap != kernel_pmap) && (initial_not_in_kdp)) {
10356 		pmap_unlock(pmap, PMAP_LOCK_SHARED);
10357 	}
10358 
10359 	return TRUE;
10360 }
10361 
10362 boolean_t
10363 pmap_is_empty(
10364 	pmap_t pmap,
10365 	vm_map_offset_t va_start,
10366 	vm_map_offset_t va_end)
10367 {
10368 	return pmap_is_empty_internal(pmap, va_start, va_end);
10369 }
10370 
10371 vm_map_offset_t
10372 pmap_max_offset(
10373 	boolean_t               is64,
10374 	unsigned int    option)
10375 {
10376 	return (is64) ? pmap_max_64bit_offset(option) : pmap_max_32bit_offset(option);
10377 }
10378 
10379 vm_map_offset_t
10380 pmap_max_64bit_offset(
10381 	__unused unsigned int option)
10382 {
10383 	vm_map_offset_t max_offset_ret = 0;
10384 
10385 	const vm_map_offset_t min_max_offset = ARM64_MIN_MAX_ADDRESS; // end of shared region + 512MB for various purposes
10386 	if (option == ARM_PMAP_MAX_OFFSET_DEFAULT) {
10387 		max_offset_ret = arm64_pmap_max_offset_default;
10388 	} else if (option == ARM_PMAP_MAX_OFFSET_MIN) {
10389 		max_offset_ret = min_max_offset;
10390 	} else if (option == ARM_PMAP_MAX_OFFSET_MAX) {
10391 		max_offset_ret = MACH_VM_MAX_ADDRESS;
10392 	} else if (option == ARM_PMAP_MAX_OFFSET_DEVICE) {
10393 		if (arm64_pmap_max_offset_default) {
10394 			max_offset_ret = arm64_pmap_max_offset_default;
10395 		} else if (max_mem > 0xC0000000) {
10396 			// devices with > 3GB of memory
10397 			max_offset_ret = ARM64_MAX_OFFSET_DEVICE_LARGE;
10398 		} else if (max_mem > 0x40000000) {
10399 			// devices with > 1GB and <= 3GB of memory
10400 			max_offset_ret = ARM64_MAX_OFFSET_DEVICE_SMALL;
10401 		} else {
10402 			// devices with <= 1 GB of memory
10403 			max_offset_ret = min_max_offset;
10404 		}
10405 	} else if (option == ARM_PMAP_MAX_OFFSET_JUMBO) {
10406 		if (arm64_pmap_max_offset_default) {
10407 			// Allow the boot-arg to override jumbo size
10408 			max_offset_ret = arm64_pmap_max_offset_default;
10409 		} else {
10410 			max_offset_ret = MACH_VM_JUMBO_ADDRESS;     // Max offset is 64GB for pmaps with special "jumbo" blessing
10411 		}
10412 #if XNU_TARGET_OS_IOS && EXTENDED_USER_VA_SUPPORT
10413 	} else if (option == ARM_PMAP_MAX_OFFSET_EXTRA_JUMBO) {
10414 		max_offset_ret = MACH_VM_MAX_ADDRESS;
10415 #endif /* XNU_TARGET_OS_IOS && EXTENDED_USER_VA_SUPPORT */
10416 	} else {
10417 		panic("pmap_max_64bit_offset illegal option 0x%x", option);
10418 	}
10419 
10420 	assert(max_offset_ret <= MACH_VM_MAX_ADDRESS);
10421 	if (option != ARM_PMAP_MAX_OFFSET_DEFAULT) {
10422 		assert(max_offset_ret >= min_max_offset);
10423 	}
10424 
10425 	return max_offset_ret;
10426 }
10427 
10428 vm_map_offset_t
10429 pmap_max_32bit_offset(
10430 	unsigned int option)
10431 {
10432 	vm_map_offset_t max_offset_ret = 0;
10433 
10434 	if (option == ARM_PMAP_MAX_OFFSET_DEFAULT) {
10435 		max_offset_ret = arm_pmap_max_offset_default;
10436 	} else if (option == ARM_PMAP_MAX_OFFSET_MIN) {
10437 		max_offset_ret = VM_MAX_ADDRESS;
10438 	} else if (option == ARM_PMAP_MAX_OFFSET_MAX) {
10439 		max_offset_ret = VM_MAX_ADDRESS;
10440 	} else if (option == ARM_PMAP_MAX_OFFSET_DEVICE) {
10441 		if (arm_pmap_max_offset_default) {
10442 			max_offset_ret = arm_pmap_max_offset_default;
10443 		} else if (max_mem > 0x20000000) {
10444 			max_offset_ret = VM_MAX_ADDRESS;
10445 		} else {
10446 			max_offset_ret = VM_MAX_ADDRESS;
10447 		}
10448 	} else if (option == ARM_PMAP_MAX_OFFSET_JUMBO) {
10449 		max_offset_ret = VM_MAX_ADDRESS;
10450 	} else {
10451 		panic("pmap_max_32bit_offset illegal option 0x%x", option);
10452 	}
10453 
10454 	assert(max_offset_ret <= MACH_VM_MAX_ADDRESS);
10455 	return max_offset_ret;
10456 }
10457 
10458 #if CONFIG_DTRACE
10459 /*
10460  * Constrain DTrace copyin/copyout actions
10461  */
10462 extern kern_return_t dtrace_copyio_preflight(addr64_t);
10463 extern kern_return_t dtrace_copyio_postflight(addr64_t);
10464 
10465 kern_return_t
10466 dtrace_copyio_preflight(
10467 	__unused addr64_t va)
10468 {
10469 	if (current_map() == kernel_map) {
10470 		return KERN_FAILURE;
10471 	} else {
10472 		return KERN_SUCCESS;
10473 	}
10474 }
10475 
10476 kern_return_t
10477 dtrace_copyio_postflight(
10478 	__unused addr64_t va)
10479 {
10480 	return KERN_SUCCESS;
10481 }
10482 #endif /* CONFIG_DTRACE */
10483 
10484 
10485 void
10486 pmap_flush_context_init(__unused pmap_flush_context *pfc)
10487 {
10488 }
10489 
10490 
10491 void
10492 pmap_flush(
10493 	__unused pmap_flush_context *cpus_to_flush)
10494 {
10495 	/* not implemented yet */
10496 	return;
10497 }
10498 
10499 /**
10500  * Perform basic validation checks on the destination only and
10501  * corresponding offset/sizes prior to writing to a read only allocation.
10502  *
10503  * @note Should be called before writing to an allocation from the read
10504  * only allocator.
10505  *
10506  * @param zid The ID of the zone the allocation belongs to.
10507  * @param va VA of element being modified (destination).
10508  * @param offset Offset being written to, in the element.
10509  * @param new_data_size Size of modification.
10510  *
10511  */
10512 
10513 MARK_AS_PMAP_TEXT static void
10514 pmap_ro_zone_validate_element_dst(
10515 	zone_id_t           zid,
10516 	vm_offset_t         va,
10517 	vm_offset_t         offset,
10518 	vm_size_t           new_data_size)
10519 {
10520 	if (__improbable((zid < ZONE_ID__FIRST_RO) || (zid > ZONE_ID__LAST_RO))) {
10521 		panic("%s: ZoneID %u outside RO range %u - %u", __func__, zid,
10522 		    ZONE_ID__FIRST_RO, ZONE_ID__LAST_RO);
10523 	}
10524 
10525 	vm_size_t elem_size = zone_ro_size_params[zid].z_elem_size;
10526 
10527 	/* Check element is from correct zone and properly aligned */
10528 	zone_require_ro(zid, elem_size, (void*)va);
10529 
10530 	if (__improbable(new_data_size > (elem_size - offset))) {
10531 		panic("%s: New data size %lu too large for elem size %lu at addr %p",
10532 		    __func__, (uintptr_t)new_data_size, (uintptr_t)elem_size, (void*)va);
10533 	}
10534 	if (__improbable(offset >= elem_size)) {
10535 		panic("%s: Offset %lu too large for elem size %lu at addr %p",
10536 		    __func__, (uintptr_t)offset, (uintptr_t)elem_size, (void*)va);
10537 	}
10538 }
10539 
10540 
10541 /**
10542  * Perform basic validation checks on the source, destination and
10543  * corresponding offset/sizes prior to writing to a read only allocation.
10544  *
10545  * @note Should be called before writing to an allocation from the read
10546  * only allocator.
10547  *
10548  * @param zid The ID of the zone the allocation belongs to.
10549  * @param va VA of element being modified (destination).
10550  * @param offset Offset being written to, in the element.
10551  * @param new_data Pointer to new data (source).
10552  * @param new_data_size Size of modification.
10553  *
10554  */
10555 
10556 MARK_AS_PMAP_TEXT static void
10557 pmap_ro_zone_validate_element(
10558 	zone_id_t           zid,
10559 	vm_offset_t         va,
10560 	vm_offset_t         offset,
10561 	const vm_offset_t   new_data,
10562 	vm_size_t           new_data_size)
10563 {
10564 	vm_offset_t sum = 0;
10565 
10566 	if (__improbable(os_add_overflow(new_data, new_data_size, &sum))) {
10567 		panic("%s: Integer addition overflow %p + %lu = %lu",
10568 		    __func__, (void*)new_data, (uintptr_t)new_data_size, (uintptr_t)sum);
10569 	}
10570 
10571 	pmap_ro_zone_validate_element_dst(zid, va, offset, new_data_size);
10572 }
10573 
10574 /**
10575  * Function to configure RO zone access permissions for a forthcoming write operation.
10576  */
10577 static void
10578 pmap_ro_zone_prepare_write(void)
10579 {
10580 }
10581 
10582 /**
10583  * Function to indicate that a preceding RO zone write operation is complete.
10584  */
10585 static void
10586 pmap_ro_zone_complete_write(void)
10587 {
10588 }
10589 
10590 /**
10591  * Function to align an address or size to the required RO zone mapping alignment.
10592  *
10593  * For the SPTM the RO zone region must be aligned on a twig boundary so that at least
10594  * the last-level kernel pagetable can be of the appropriate SPTM RO zone table type,
10595  * which allows the SPTM to enforce RO zone mapping permission restrictions.
10596  *
10597  * @param value the address or size to be aligned.
10598  *
10599  * @return the aligned value
10600  */
10601 vm_offset_t
10602 pmap_ro_zone_align(vm_offset_t value)
10603 {
10604 	const pt_attr_t * const pt_attr = pmap_get_pt_attr(kernel_pmap);
10605 	return PMAP_ALIGN(value, pt_attr_twig_size(pt_attr));
10606 }
10607 
10608 /**
10609  * Function to copy kauth_cred from new_data to kv.
10610  * Function defined in "kern_prot.c"
10611  *
10612  * @note Will be removed upon completion of
10613  * <rdar://problem/72635194> Compiler PAC support for memcpy.
10614  *
10615  * @param kv Address to copy new data to.
10616  * @param new_data Pointer to new data.
10617  *
10618  */
10619 
10620 extern void
10621 kauth_cred_copy(const uintptr_t kv, const uintptr_t new_data);
10622 
10623 /**
10624  * Zalloc-specific memcpy that writes through the physical aperture
10625  * and ensures the element being modified is from a read-only zone.
10626  *
10627  * @note Designed to work only with the zone allocator's read-only submap.
10628  *
10629  * @param zid The ID of the zone to allocate from.
10630  * @param va VA of element to be modified.
10631  * @param offset Offset from element.
10632  * @param new_data Pointer to new data.
10633  * @param new_data_size	Size of modification.
10634  *
10635  */
10636 
10637 void
10638 pmap_ro_zone_memcpy(
10639 	zone_id_t           zid,
10640 	vm_offset_t         va,
10641 	vm_offset_t         offset,
10642 	const vm_offset_t   new_data,
10643 	vm_size_t           new_data_size)
10644 {
10645 	pmap_ro_zone_memcpy_internal(zid, va, offset, new_data, new_data_size);
10646 }
10647 
10648 MARK_AS_PMAP_TEXT void
10649 pmap_ro_zone_memcpy_internal(
10650 	zone_id_t             zid,
10651 	vm_offset_t           va,
10652 	vm_offset_t           offset,
10653 	const vm_offset_t     new_data,
10654 	vm_size_t             new_data_size)
10655 {
10656 	if (!new_data || new_data_size == 0) {
10657 		return;
10658 	}
10659 
10660 	const pmap_paddr_t pa = kvtophys_nofail(va + offset);
10661 	const bool istate = ml_set_interrupts_enabled(FALSE);
10662 	pmap_ro_zone_validate_element(zid, va, offset, new_data, new_data_size);
10663 	pmap_ro_zone_prepare_write();
10664 	memcpy((void*)phystokv(pa), (void*)new_data, new_data_size);
10665 	pmap_ro_zone_complete_write();
10666 	ml_set_interrupts_enabled(istate);
10667 }
10668 
10669 /**
10670  * Zalloc-specific function to atomically mutate fields of an element that
10671  * belongs to a read-only zone, via the physcial aperture.
10672  *
10673  * @note Designed to work only with the zone allocator's read-only submap.
10674  *
10675  * @param zid The ID of the zone the element belongs to.
10676  * @param va VA of element to be modified.
10677  * @param offset Offset in element.
10678  * @param op Atomic operation to perform.
10679  * @param value	Mutation value.
10680  *
10681  */
10682 
10683 uint64_t
10684 pmap_ro_zone_atomic_op(
10685 	zone_id_t             zid,
10686 	vm_offset_t           va,
10687 	vm_offset_t           offset,
10688 	zro_atomic_op_t       op,
10689 	uint64_t              value)
10690 {
10691 	return pmap_ro_zone_atomic_op_internal(zid, va, offset, op, value);
10692 }
10693 
10694 MARK_AS_PMAP_TEXT uint64_t
10695 pmap_ro_zone_atomic_op_internal(
10696 	zone_id_t             zid,
10697 	vm_offset_t           va,
10698 	vm_offset_t           offset,
10699 	zro_atomic_op_t       op,
10700 	uint64_t              value)
10701 {
10702 	const pmap_paddr_t pa = kvtophys_nofail(va + offset);
10703 	vm_size_t value_size = op & 0xf;
10704 	const boolean_t istate = ml_set_interrupts_enabled(FALSE);
10705 
10706 	pmap_ro_zone_validate_element_dst(zid, va, offset, value_size);
10707 	pmap_ro_zone_prepare_write();
10708 	value = __zalloc_ro_mut_atomic(phystokv(pa), op, value);
10709 	pmap_ro_zone_complete_write();
10710 	ml_set_interrupts_enabled(istate);
10711 
10712 	return value;
10713 }
10714 
10715 /**
10716  * bzero for allocations from read only zones, that writes through the
10717  * physical aperture.
10718  *
10719  * @note This is called by the zfree path of all allocations from read
10720  * only zones.
10721  *
10722  * @param zid The ID of the zone the allocation belongs to.
10723  * @param va VA of element to be zeroed.
10724  * @param offset Offset in the element.
10725  * @param size	Size of allocation.
10726  *
10727  */
10728 
10729 void
10730 pmap_ro_zone_bzero(
10731 	zone_id_t       zid,
10732 	vm_offset_t     va,
10733 	vm_offset_t     offset,
10734 	vm_size_t       size)
10735 {
10736 	pmap_ro_zone_bzero_internal(zid, va, offset, size);
10737 }
10738 
10739 MARK_AS_PMAP_TEXT void
10740 pmap_ro_zone_bzero_internal(
10741 	zone_id_t       zid,
10742 	vm_offset_t     va,
10743 	vm_offset_t     offset,
10744 	vm_size_t       size)
10745 {
10746 	const pmap_paddr_t pa = kvtophys_nofail(va + offset);
10747 	const boolean_t istate = ml_set_interrupts_enabled(FALSE);
10748 	pmap_ro_zone_validate_element(zid, va, offset, 0, size);
10749 	pmap_ro_zone_prepare_write();
10750 	bzero((void*)phystokv(pa), size);
10751 	pmap_ro_zone_complete_write();
10752 	ml_set_interrupts_enabled(istate);
10753 }
10754 
10755 #define PMAP_RESIDENT_INVALID   ((mach_vm_size_t)-1)
10756 
10757 MARK_AS_PMAP_TEXT mach_vm_size_t
10758 pmap_query_resident_internal(
10759 	pmap_t                  pmap,
10760 	vm_map_address_t        start,
10761 	vm_map_address_t        end,
10762 	mach_vm_size_t          *compressed_bytes_p)
10763 {
10764 	mach_vm_size_t  resident_bytes = 0;
10765 	mach_vm_size_t  compressed_bytes = 0;
10766 
10767 	pt_entry_t     *bpte, *epte;
10768 	pt_entry_t     *pte_p;
10769 	tt_entry_t     *tte_p;
10770 
10771 	if (pmap == NULL) {
10772 		return PMAP_RESIDENT_INVALID;
10773 	}
10774 
10775 	validate_pmap(pmap);
10776 
10777 	const pt_attr_t * const pt_attr = pmap_get_pt_attr(pmap);
10778 
10779 	/* Ensure that this request is valid, and addresses exactly one TTE. */
10780 	if (__improbable((start % pt_attr_page_size(pt_attr)) ||
10781 	    (end % pt_attr_page_size(pt_attr)))) {
10782 		panic("%s: address range %p, %p not page-aligned to 0x%llx", __func__, (void*)start, (void*)end, pt_attr_page_size(pt_attr));
10783 	}
10784 
10785 	if (__improbable((end < start) || (end > ((start + pt_attr_twig_size(pt_attr)) & ~pt_attr_twig_offmask(pt_attr))))) {
10786 		panic("%s: invalid address range %p, %p", __func__, (void*)start, (void*)end);
10787 	}
10788 
10789 	pmap_lock(pmap, PMAP_LOCK_SHARED);
10790 	tte_p = pmap_tte(pmap, start);
10791 	if (tte_p == (tt_entry_t *) NULL) {
10792 		pmap_unlock(pmap, PMAP_LOCK_SHARED);
10793 		return PMAP_RESIDENT_INVALID;
10794 	}
10795 	if ((*tte_p & ARM_TTE_TYPE_MASK) == ARM_TTE_TYPE_TABLE) {
10796 		pte_p = (pt_entry_t *) ttetokv(*tte_p);
10797 		bpte = &pte_p[pte_index(pt_attr, start)];
10798 		epte = &pte_p[pte_index(pt_attr, end)];
10799 
10800 		for (; bpte < epte; bpte++) {
10801 			if (pte_is_compressed(*bpte, bpte)) {
10802 				compressed_bytes += pt_attr_page_size(pt_attr);
10803 			} else if (pa_valid(pte_to_pa(*bpte))) {
10804 				resident_bytes += pt_attr_page_size(pt_attr);
10805 			}
10806 		}
10807 	}
10808 	pmap_unlock(pmap, PMAP_LOCK_SHARED);
10809 
10810 	if (compressed_bytes_p) {
10811 		*compressed_bytes_p += compressed_bytes;
10812 	}
10813 
10814 	return resident_bytes;
10815 }
10816 
10817 mach_vm_size_t
10818 pmap_query_resident(
10819 	pmap_t                  pmap,
10820 	vm_map_address_t        start,
10821 	vm_map_address_t        end,
10822 	mach_vm_size_t          *compressed_bytes_p)
10823 {
10824 	mach_vm_size_t          total_resident_bytes;
10825 	mach_vm_size_t          compressed_bytes;
10826 	vm_map_address_t        va;
10827 
10828 
10829 	if (pmap == PMAP_NULL) {
10830 		if (compressed_bytes_p) {
10831 			*compressed_bytes_p = 0;
10832 		}
10833 		return 0;
10834 	}
10835 
10836 	__unused const pt_attr_t * const pt_attr = pmap_get_pt_attr(pmap);
10837 
10838 	total_resident_bytes = 0;
10839 	compressed_bytes = 0;
10840 
10841 	PMAP_TRACE(3, PMAP_CODE(PMAP__QUERY_RESIDENT) | DBG_FUNC_START,
10842 	    VM_KERNEL_ADDRHIDE(pmap), VM_KERNEL_ADDRHIDE(start),
10843 	    VM_KERNEL_ADDRHIDE(end));
10844 
10845 	va = start;
10846 	while (va < end) {
10847 		vm_map_address_t l;
10848 		mach_vm_size_t resident_bytes;
10849 
10850 		l = ((va + pt_attr_twig_size(pt_attr)) & ~pt_attr_twig_offmask(pt_attr));
10851 
10852 		if (l > end) {
10853 			l = end;
10854 		}
10855 		resident_bytes = pmap_query_resident_internal(pmap, va, l, compressed_bytes_p);
10856 		if (resident_bytes == PMAP_RESIDENT_INVALID) {
10857 			break;
10858 		}
10859 
10860 		total_resident_bytes += resident_bytes;
10861 
10862 		va = l;
10863 	}
10864 
10865 	if (compressed_bytes_p) {
10866 		*compressed_bytes_p = compressed_bytes;
10867 	}
10868 
10869 	PMAP_TRACE(3, PMAP_CODE(PMAP__QUERY_RESIDENT) | DBG_FUNC_END,
10870 	    total_resident_bytes);
10871 
10872 	return total_resident_bytes;
10873 }
10874 
10875 #if MACH_ASSERT
10876 static void
10877 pmap_check_ledgers(
10878 	pmap_t pmap)
10879 {
10880 	int     pid;
10881 	char    *procname;
10882 
10883 	if (pmap->pmap_pid == 0 || pmap->pmap_pid == -1) {
10884 		/*
10885 		 * This pmap was not or is no longer fully associated
10886 		 * with a task (e.g. the old pmap after a fork()/exec() or
10887 		 * spawn()).  Its "ledger" still points at a task that is
10888 		 * now using a different (and active) address space, so
10889 		 * we can't check that all the pmap ledgers are balanced here.
10890 		 *
10891 		 * If the "pid" is set, that means that we went through
10892 		 * pmap_set_process() in task_terminate_internal(), so
10893 		 * this task's ledger should not have been re-used and
10894 		 * all the pmap ledgers should be back to 0.
10895 		 */
10896 		return;
10897 	}
10898 
10899 	pid = pmap->pmap_pid;
10900 	procname = pmap->pmap_procname;
10901 
10902 	vm_map_pmap_check_ledgers(pmap, pmap->ledger, pid, procname);
10903 }
10904 #endif /* MACH_ASSERT */
10905 
10906 void
10907 pmap_advise_pagezero_range(__unused pmap_t p, __unused uint64_t a)
10908 {
10909 }
10910 
10911 /**
10912  * The minimum shared region nesting size is used by the VM to determine when to
10913  * break up large mappings to nested regions. The smallest size that these
10914  * mappings can be broken into is determined by what page table level those
10915  * regions are being nested in at and the size of the page tables.
10916  *
10917  * For instance, if a nested region is nesting at L2 for a process utilizing
10918  * 16KB page tables, then the minimum nesting size would be 32MB (size of an L2
10919  * block entry).
10920  *
10921  * @param pmap The target pmap to determine the block size based on whether it's
10922  *             using 16KB or 4KB page tables.
10923  */
10924 uint64_t
10925 pmap_shared_region_size_min(__unused pmap_t pmap)
10926 {
10927 	const pt_attr_t * const pt_attr = pmap_get_pt_attr(pmap);
10928 
10929 	/**
10930 	 * We always nest the shared region at L2 (32MB for 16KB pages, 8MB for
10931 	 * 4KB pages). This means that a target pmap will contain L2 entries that
10932 	 * point to shared L3 page tables in the shared region pmap.
10933 	 */
10934 	const uint64_t page_ratio = PAGE_SIZE / pt_attr_page_size(pt_attr);
10935 	return pt_attr_twig_size(pt_attr) * page_ratio;
10936 }
10937 
10938 boolean_t
10939 pmap_enforces_execute_only(
10940 	pmap_t pmap)
10941 {
10942 	return pmap != kernel_pmap;
10943 }
10944 
10945 MARK_AS_PMAP_TEXT void
10946 pmap_set_vm_map_cs_enforced_internal(
10947 	pmap_t pmap,
10948 	bool new_value)
10949 {
10950 	validate_pmap_mutable(pmap);
10951 	pmap->pmap_vm_map_cs_enforced = new_value;
10952 }
10953 
10954 void
10955 pmap_set_vm_map_cs_enforced(
10956 	pmap_t pmap,
10957 	bool new_value)
10958 {
10959 	pmap_set_vm_map_cs_enforced_internal(pmap, new_value);
10960 }
10961 
10962 extern int cs_process_enforcement_enable;
10963 bool
10964 pmap_get_vm_map_cs_enforced(
10965 	pmap_t pmap)
10966 {
10967 	if (cs_process_enforcement_enable) {
10968 		return true;
10969 	}
10970 	return pmap->pmap_vm_map_cs_enforced;
10971 }
10972 
10973 MARK_AS_PMAP_TEXT void
10974 pmap_set_jit_entitled_internal(
10975 	__unused pmap_t pmap)
10976 {
10977 }
10978 
10979 void
10980 pmap_set_jit_entitled(
10981 	pmap_t pmap)
10982 {
10983 	pmap_set_jit_entitled_internal(pmap);
10984 }
10985 
10986 bool
10987 pmap_get_jit_entitled(
10988 	__unused pmap_t pmap)
10989 {
10990 	return false;
10991 }
10992 
10993 MARK_AS_PMAP_TEXT void
10994 pmap_set_tpro_internal(
10995 	__unused pmap_t pmap)
10996 {
10997 	return;
10998 }
10999 
11000 void
11001 pmap_set_tpro(
11002 	pmap_t pmap)
11003 {
11004 	pmap_set_tpro_internal(pmap);
11005 }
11006 
11007 bool
11008 pmap_get_tpro(
11009 	__unused pmap_t pmap)
11010 {
11011 	return false;
11012 }
11013 
11014 uint64_t pmap_query_page_info_retries MARK_AS_PMAP_DATA;
11015 
11016 MARK_AS_PMAP_TEXT kern_return_t
11017 pmap_query_page_info_internal(
11018 	pmap_t          pmap,
11019 	vm_map_offset_t va,
11020 	int             *disp_p)
11021 {
11022 	pmap_paddr_t    pa;
11023 	int             disp;
11024 	unsigned int    pai;
11025 	pt_entry_t      *pte_p;
11026 	pv_entry_t      *pve_p;
11027 
11028 	if (pmap == PMAP_NULL || pmap == kernel_pmap) {
11029 		*disp_p = 0;
11030 		return KERN_INVALID_ARGUMENT;
11031 	}
11032 
11033 	validate_pmap(pmap);
11034 	pmap_lock(pmap, PMAP_LOCK_SHARED);
11035 
11036 try_again:
11037 	disp = 0;
11038 
11039 	pte_p = pmap_pte(pmap, va);
11040 	if (pte_p == PT_ENTRY_NULL) {
11041 		goto done;
11042 	}
11043 
11044 	const pt_entry_t pte = os_atomic_load(pte_p, relaxed);
11045 	pa = pte_to_pa(pte);
11046 	if (pa == 0) {
11047 		if (pte_is_compressed(pte, pte_p)) {
11048 			disp |= PMAP_QUERY_PAGE_COMPRESSED;
11049 			if (pte & ARM_PTE_COMPRESSED_ALT) {
11050 				disp |= PMAP_QUERY_PAGE_COMPRESSED_ALTACCT;
11051 			}
11052 		}
11053 	} else {
11054 		disp |= PMAP_QUERY_PAGE_PRESENT;
11055 		pai = pa_index(pa);
11056 		if (!pa_valid(pa)) {
11057 			goto done;
11058 		}
11059 		locked_pvh_t locked_pvh = pvh_lock(pai);
11060 		if (__improbable(pte != os_atomic_load(pte_p, relaxed))) {
11061 			/* something changed: try again */
11062 			pvh_unlock(&locked_pvh);
11063 			pmap_query_page_info_retries++;
11064 			goto try_again;
11065 		}
11066 		pve_p = PV_ENTRY_NULL;
11067 		int pve_ptep_idx = 0;
11068 		if (pvh_test_type(locked_pvh.pvh, PVH_TYPE_PVEP)) {
11069 			unsigned int npves = 0;
11070 			pve_p = pvh_pve_list(locked_pvh.pvh);
11071 			while (pve_p != PV_ENTRY_NULL &&
11072 			    (pve_ptep_idx = pve_find_ptep_index(pve_p, pte_p)) == -1) {
11073 				if (__improbable(npves == (SPTM_MAPPING_LIMIT / PTE_PER_PVE))) {
11074 					pvh_lock_enter_sleep_mode(&locked_pvh);
11075 				}
11076 				pve_p = pve_next(pve_p);
11077 				npves++;
11078 			}
11079 		}
11080 
11081 		if (ppattr_pve_is_altacct(pai, pve_p, pve_ptep_idx)) {
11082 			disp |= PMAP_QUERY_PAGE_ALTACCT;
11083 		} else if (ppattr_test_reusable(pai)) {
11084 			disp |= PMAP_QUERY_PAGE_REUSABLE;
11085 		} else if (ppattr_pve_is_internal(pai, pve_p, pve_ptep_idx)) {
11086 			disp |= PMAP_QUERY_PAGE_INTERNAL;
11087 		}
11088 		pvh_unlock(&locked_pvh);
11089 	}
11090 
11091 done:
11092 	pmap_unlock(pmap, PMAP_LOCK_SHARED);
11093 	*disp_p = disp;
11094 	return KERN_SUCCESS;
11095 }
11096 
11097 kern_return_t
11098 pmap_query_page_info(
11099 	pmap_t          pmap,
11100 	vm_map_offset_t va,
11101 	int             *disp_p)
11102 {
11103 	return pmap_query_page_info_internal(pmap, va, disp_p);
11104 }
11105 
11106 
11107 
11108 uint32_t
11109 pmap_user_va_bits(pmap_t pmap __unused)
11110 {
11111 #if __ARM_MIXED_PAGE_SIZE__
11112 	uint64_t tcr_value = pmap_get_pt_attr(pmap)->pta_tcr_value;
11113 	return 64 - ((tcr_value >> TCR_T0SZ_SHIFT) & TCR_TSZ_MASK);
11114 #else
11115 	return 64 - T0SZ_BOOT;
11116 #endif
11117 }
11118 
11119 uint32_t
11120 pmap_kernel_va_bits(void)
11121 {
11122 	return 64 - T1SZ_BOOT;
11123 }
11124 
11125 static vm_map_size_t
11126 pmap_user_va_size(pmap_t pmap)
11127 {
11128 	return 1ULL << pmap_user_va_bits(pmap);
11129 }
11130 
11131 
11132 bool
11133 pmap_in_ppl(void)
11134 {
11135 	return false;
11136 }
11137 
11138 MARK_AS_PMAP_TEXT void
11139 pmap_footprint_suspend_internal(
11140 	vm_map_t        map,
11141 	boolean_t       suspend)
11142 {
11143 #if DEVELOPMENT || DEBUG
11144 	if (suspend) {
11145 		current_thread()->pmap_footprint_suspended = TRUE;
11146 		map->pmap->footprint_was_suspended = TRUE;
11147 	} else {
11148 		current_thread()->pmap_footprint_suspended = FALSE;
11149 	}
11150 #else /* DEVELOPMENT || DEBUG */
11151 	(void) map;
11152 	(void) suspend;
11153 #endif /* DEVELOPMENT || DEBUG */
11154 }
11155 
11156 void
11157 pmap_footprint_suspend(
11158 	vm_map_t map,
11159 	boolean_t suspend)
11160 {
11161 	pmap_footprint_suspend_internal(map, suspend);
11162 }
11163 
11164 void
11165 pmap_nop(pmap_t pmap)
11166 {
11167 	validate_pmap_mutable(pmap);
11168 }
11169 
11170 pmap_t
11171 pmap_txm_kernel_pmap(void)
11172 {
11173 	return kernel_pmap;
11174 }
11175 
11176 TXMAddressSpace_t*
11177 pmap_txm_addr_space(const pmap_t pmap)
11178 {
11179 	if (pmap) {
11180 		return pmap->txm_addr_space;
11181 	}
11182 
11183 	/*
11184 	 * When the passed in PMAP is NULL, it means the caller wishes to operate
11185 	 * on the current_pmap(). We could resolve and return that, but it is actually
11186 	 * safer to return NULL since these TXM interfaces also accept NULL inputs
11187 	 * which causes TXM to resolve to the current_pmap() equivalent internally.
11188 	 */
11189 	return NULL;
11190 }
11191 
11192 void
11193 pmap_txm_set_addr_space(
11194 	pmap_t pmap,
11195 	TXMAddressSpace_t *txm_addr_space)
11196 {
11197 	assert(pmap != NULL);
11198 
11199 	if (pmap->txm_addr_space && txm_addr_space) {
11200 		/* Attempted to overwrite the address space in the PMAP */
11201 		panic("attempted ovewrite of TXM address space: %p | %p | %p",
11202 		    pmap, pmap->txm_addr_space, txm_addr_space);
11203 	} else if (!pmap->txm_addr_space && !txm_addr_space) {
11204 		/* This should never happen */
11205 		panic("attempted NULL overwrite of TXM address space: %p", pmap);
11206 	}
11207 
11208 	pmap->txm_addr_space = txm_addr_space;
11209 }
11210 
11211 void
11212 pmap_txm_set_trust_level(
11213 	pmap_t pmap,
11214 	CSTrust_t trust_level)
11215 {
11216 	assert(pmap != NULL);
11217 
11218 	CSTrust_t current_trust = pmap->txm_trust_level;
11219 	if (current_trust != kCSTrustUntrusted) {
11220 		panic("attempted to overwrite TXM trust on the pmap: %p", pmap);
11221 	}
11222 
11223 	pmap->txm_trust_level = trust_level;
11224 }
11225 
11226 kern_return_t
11227 pmap_txm_get_trust_level_kdp(
11228 	pmap_t pmap,
11229 	CSTrust_t *trust_level)
11230 {
11231 	if (pmap == NULL) {
11232 		return KERN_INVALID_ARGUMENT;
11233 	} else if (ml_validate_nofault((vm_offset_t)pmap, sizeof(*pmap)) == false) {
11234 		return KERN_INVALID_ARGUMENT;
11235 	}
11236 
11237 	if (trust_level != NULL) {
11238 		*trust_level = pmap->txm_trust_level;
11239 	}
11240 	return KERN_SUCCESS;
11241 }
11242 
11243 kern_return_t
11244 pmap_txm_get_jit_address_range_kdp(
11245 	pmap_t pmap,
11246 	uintptr_t *jit_region_start,
11247 	uintptr_t *jit_region_end)
11248 {
11249 	if (ml_validate_nofault((vm_offset_t)pmap, sizeof(*pmap)) == false) {
11250 		return KERN_INVALID_ARGUMENT;
11251 	}
11252 	TXMAddressSpace_t *txm_addr_space = pmap_txm_addr_space(pmap);
11253 	if (NULL == txm_addr_space) {
11254 		return KERN_INVALID_ARGUMENT;
11255 	}
11256 	if (ml_validate_nofault((vm_offset_t)txm_addr_space, sizeof(*txm_addr_space)) == false) {
11257 		return KERN_INVALID_ARGUMENT;
11258 	}
11259 	/**
11260 	 * It's a bit gross that we're dereferencing what is supposed to be an abstract type.
11261 	 * If we were running in the TXM, we would always perform additional checks on txm_addr_space,
11262 	 * but this isn't necessary here, since we are running in the kernel and only using the results for
11263 	 * diagnostic purposes, rather than any policy enforcement.
11264 	 */
11265 	if (txm_addr_space->jitRegion) {
11266 		if (ml_validate_nofault((vm_offset_t)txm_addr_space->jitRegion, sizeof(txm_addr_space->jitRegion)) == false) {
11267 			return KERN_INVALID_ARGUMENT;
11268 		}
11269 		if (txm_addr_space->jitRegion->addr && txm_addr_space->jitRegion->addrEnd) {
11270 			*jit_region_start = txm_addr_space->jitRegion->addr;
11271 			*jit_region_end = txm_addr_space->jitRegion->addrEnd;
11272 			return KERN_SUCCESS;
11273 		}
11274 	}
11275 	return KERN_NOT_FOUND;
11276 }
11277 
11278 static pmap_t
11279 _pmap_txm_resolve_pmap(pmap_t pmap)
11280 {
11281 	if (pmap == NULL) {
11282 		pmap = current_pmap();
11283 		if (pmap == kernel_pmap) {
11284 			return NULL;
11285 		}
11286 	}
11287 
11288 	return pmap;
11289 }
11290 
11291 void
11292 pmap_txm_acquire_shared_lock(pmap_t pmap)
11293 {
11294 	pmap = _pmap_txm_resolve_pmap(pmap);
11295 	if (!pmap) {
11296 		return;
11297 	}
11298 
11299 	lck_rw_lock_shared(&pmap->txm_lck);
11300 }
11301 
11302 void
11303 pmap_txm_release_shared_lock(pmap_t pmap)
11304 {
11305 	pmap = _pmap_txm_resolve_pmap(pmap);
11306 	if (!pmap) {
11307 		return;
11308 	}
11309 
11310 	lck_rw_unlock_shared(&pmap->txm_lck);
11311 }
11312 
11313 void
11314 pmap_txm_acquire_exclusive_lock(pmap_t pmap)
11315 {
11316 	pmap = _pmap_txm_resolve_pmap(pmap);
11317 	if (!pmap) {
11318 		return;
11319 	}
11320 
11321 	lck_rw_lock_exclusive(&pmap->txm_lck);
11322 }
11323 
11324 void
11325 pmap_txm_release_exclusive_lock(pmap_t pmap)
11326 {
11327 	pmap = _pmap_txm_resolve_pmap(pmap);
11328 	if (!pmap) {
11329 		return;
11330 	}
11331 
11332 	lck_rw_unlock_exclusive(&pmap->txm_lck);
11333 }
11334 
11335 static void
11336 _pmap_txm_transfer_page(const pmap_paddr_t addr)
11337 {
11338 	sptm_retype_params_t retype_params = {
11339 		.raw = SPTM_RETYPE_PARAMS_NULL
11340 	};
11341 
11342 	/* Retype through the SPTM */
11343 	sptm_retype(addr, XNU_DEFAULT, TXM_DEFAULT, retype_params);
11344 }
11345 
11346 /**
11347  * Prepare a page for retyping to TXM_DEFAULT by clearing its
11348  * internal flags.
11349  *
11350  * @param pa Physical address of the page.
11351  */
11352 static inline void
11353 _pmap_txm_retype_prepare(const pmap_paddr_t pa)
11354 {
11355 	const sptm_retype_params_t retype_params = {
11356 		.raw = SPTM_RETYPE_PARAMS_NULL
11357 	};
11358 
11359 	/**
11360 	 * SPTM allows XNU_DEFAULT pages to request deferral of TLB flushing
11361 	 * when their PTE is updated, which is an important performance
11362 	 * optimization. However, this also allows an attacker controlled
11363 	 * XNU to exploit a read reference with a stale write-enabled PTE in
11364 	 * TLB. This is fine as long as the page is not retyped and the damage
11365 	 * will be contained within XNU domain. However, when such a page needs
11366 	 * to be retyped, SPTM has to make sure there's no outstanding
11367 	 * reference, or there's no history of deferring TLBIs. Internally,
11368 	 * SPTM maintains a flag tracking past deferred TLBIs that only gets
11369 	 * cleared on retyping with no outstanding reference. Therefore, we
11370 	 * do a dummy retype to XNU_DEFAULT itself to clear the internal flag,
11371 	 * before we actually transfer this page to TXM domain. To make sure
11372 	 * SPTM won't throw a violation, all the mappings to the page have to
11373 	 * be removed before calling this.
11374 	 */
11375 	sptm_retype(pa, XNU_DEFAULT, XNU_DEFAULT, retype_params);
11376 }
11377 
11378 /**
11379  * Transfer an XNU owned page to TXM domain.
11380  *
11381  * @param addr Kernel virtual address of the page. It has to be page size
11382  *             aligned.
11383  */
11384 void
11385 pmap_txm_transfer_page(const vm_address_t addr)
11386 {
11387 	assert((addr & PAGE_MASK) == 0);
11388 
11389 	const pmap_paddr_t pa = kvtophys_nofail(addr);
11390 	const unsigned int pai = pa_index(pa);
11391 
11392 	/* Lock the PVH lock to prevent concurrent updates to the mappings during the self retype below. */
11393 	locked_pvh_t locked_pvh = pvh_lock(pai);
11394 
11395 	/* Disconnect the mapping to assure SPTM of no pending TLBI. */
11396 	pmap_page_protect_options_with_flush_range((ppnum_t)atop(pa), VM_PROT_NONE,
11397 	    PMAP_OPTIONS_PPO_PENDING_RETYPE, &locked_pvh, NULL);
11398 
11399 	/* Self retype to clear the SPTM internal flags tracking delayed TLBIs for revoked writes. */
11400 	_pmap_txm_retype_prepare(pa);
11401 
11402 	pvh_unlock(&locked_pvh);
11403 
11404 	/* XNU needs to hold an RO reference to the page despite the ownership being transferred to TXM. */
11405 	pmap_enter_addr(kernel_pmap, addr, pa, VM_PROT_READ, VM_PROT_NONE, 0, true, PMAP_MAPPING_TYPE_INFER);
11406 
11407 	/* Finally, retype the page to TXM_DEFAULT. */
11408 	_pmap_txm_transfer_page(pa);
11409 }
11410 
11411 struct vm_object txm_vm_object_storage VM_PAGE_PACKED_ALIGNED;
11412 SECURITY_READ_ONLY_LATE(vm_object_t) txm_vm_object = &txm_vm_object_storage;
11413 
11414 _Static_assert(sizeof(vm_map_address_t) == sizeof(pmap_paddr_t),
11415     "sizeof(vm_map_address_t) != sizeof(pmap_paddr_t)");
11416 
11417 vm_map_address_t
11418 pmap_txm_allocate_page(void)
11419 {
11420 	pmap_paddr_t phys_addr = 0;
11421 	vm_page_t page = VM_PAGE_NULL;
11422 	boolean_t thread_vm_privileged = false;
11423 
11424 	/* We are allowed to allocate privileged memory */
11425 	thread_vm_privileged = set_vm_privilege(true);
11426 
11427 	/* Allocate a page from the VM free list */
11428 	while ((page = vm_page_grab()) == VM_PAGE_NULL) {
11429 		VM_PAGE_WAIT();
11430 	}
11431 
11432 	/* Wire all of the pages allocated for TXM */
11433 	vm_page_lock_queues();
11434 	vm_page_wire(page, VM_KERN_MEMORY_SECURITY, TRUE);
11435 	vm_page_unlock_queues();
11436 
11437 	phys_addr = (pmap_paddr_t)ptoa(VM_PAGE_GET_PHYS_PAGE(page));
11438 	if (phys_addr == 0) {
11439 		panic("invalid VM page allocated for TXM: %llu", phys_addr);
11440 	}
11441 
11442 	/* Add the physical page to the TXM VM object */
11443 	vm_object_lock(txm_vm_object);
11444 	vm_page_insert_wired(
11445 		page,
11446 		txm_vm_object,
11447 		phys_addr - gPhysBase,
11448 		VM_KERN_MEMORY_SECURITY);
11449 	vm_object_unlock(txm_vm_object);
11450 
11451 	/* Reset thread privilege */
11452 	set_vm_privilege(thread_vm_privileged);
11453 
11454 	/* Retype the page */
11455 	_pmap_txm_transfer_page(phys_addr);
11456 
11457 	return phys_addr;
11458 }
11459 
11460 int
11461 pmap_cs_configuration(void)
11462 {
11463 	code_signing_config_t config = 0;
11464 
11465 	/* Compute the code signing configuration */
11466 	code_signing_configuration(NULL, &config);
11467 
11468 	return (int)config;
11469 }
11470 
11471 bool
11472 pmap_performs_stage2_translations(
11473 	__unused pmap_t pmap)
11474 {
11475 	return false;
11476 }
11477 
11478 bool
11479 pmap_has_iofilter_protected_write(void)
11480 {
11481 #if HAS_GUARDED_IO_FILTER
11482 	return true;
11483 #else
11484 	return false;
11485 #endif
11486 }
11487 
11488 #if HAS_GUARDED_IO_FILTER
11489 
11490 void
11491 pmap_iofilter_protected_write(__unused vm_address_t addr, __unused uint64_t value, __unused uint64_t width)
11492 {
11493 	/**
11494 	 * Even though this is done from EL1/2 for an address potentially owned by Guarded
11495 	 * Mode, we should be fine as mmu_kvtop uses "at s1e1r" checking for read access
11496 	 * only.
11497 	 */
11498 	const pmap_paddr_t pa = mmu_kvtop(addr);
11499 
11500 	if (!pa) {
11501 		panic("%s: addr 0x%016llx doesn't have a valid kernel mapping", __func__, (uint64_t) addr);
11502 	}
11503 
11504 	const sptm_frame_type_t frame_type = sptm_get_frame_type(pa);
11505 	if (frame_type == XNU_PROTECTED_IO) {
11506 		sptm_iofilter_protected_write(pa, value, width);
11507 	} else {
11508 		/* Mappings is valid but not specified by I/O filter. However, we still try
11509 		 * accessing the address from kernel mode. This allows addresses that are not
11510 		 * owned by SPTM to be accessed by this interface.
11511 		 */
11512 		switch (width) {
11513 		case 1:
11514 			*(volatile uint8_t *)addr = (uint8_t) value;
11515 			break;
11516 		case 2:
11517 			*(volatile uint16_t *)addr = (uint16_t) value;
11518 			break;
11519 		case 4:
11520 			*(volatile uint32_t *)addr = (uint32_t) value;
11521 			break;
11522 		case 8:
11523 			*(volatile uint64_t *)addr = (uint64_t) value;
11524 			break;
11525 		default:
11526 			panic("%s: width %llu not supported", __func__, width);
11527 		}
11528 	}
11529 }
11530 
11531 #else /* HAS_GUARDED_IO_FILTER */
11532 
11533 __attribute__((__noreturn__))
11534 void
11535 pmap_iofilter_protected_write(__unused vm_address_t addr, __unused uint64_t value, __unused uint64_t width)
11536 {
11537 	panic("%s called on an unsupported platform.", __FUNCTION__);
11538 }
11539 
11540 #endif /* HAS_GUARDED_IO_FILTER */
11541 
11542 void * __attribute__((noreturn))
11543 pmap_claim_reserved_ppl_page(void)
11544 {
11545 	panic("%s: function not supported in this environment", __FUNCTION__);
11546 }
11547 
11548 void __attribute__((noreturn))
11549 pmap_free_reserved_ppl_page(void __unused *kva)
11550 {
11551 	panic("%s: function not supported in this environment", __FUNCTION__);
11552 }
11553 
11554 bool
11555 pmap_lookup_in_loaded_trust_caches(__unused const uint8_t cdhash[CS_CDHASH_LEN])
11556 {
11557 	kern_return_t kr = query_trust_cache(
11558 		kTCQueryTypeLoadable,
11559 		cdhash,
11560 		NULL);
11561 
11562 	if (kr == KERN_SUCCESS) {
11563 		return true;
11564 	}
11565 	return false;
11566 }
11567 
11568 uint32_t
11569 pmap_lookup_in_static_trust_cache(__unused const uint8_t cdhash[CS_CDHASH_LEN])
11570 {
11571 	TrustCacheQueryToken_t query_token = {0};
11572 	kern_return_t kr = KERN_NOT_FOUND;
11573 	uint64_t flags = 0;
11574 	uint8_t hash_type = 0;
11575 
11576 	kr = query_trust_cache(
11577 		kTCQueryTypeStatic,
11578 		cdhash,
11579 		&query_token);
11580 
11581 	if (kr == KERN_SUCCESS) {
11582 		amfi->TrustCache.queryGetFlags(&query_token, &flags);
11583 		amfi->TrustCache.queryGetHashType(&query_token, &hash_type);
11584 
11585 		return (TC_LOOKUP_FOUND << TC_LOOKUP_RESULT_SHIFT) |
11586 		       (hash_type << TC_LOOKUP_HASH_TYPE_SHIFT) |
11587 		       ((uint8_t)flags << TC_LOOKUP_FLAGS_SHIFT);
11588 	}
11589 
11590 	return 0;
11591 }
11592 
11593 #if DEVELOPMENT || DEBUG
11594 
11595 struct page_table_dump_header {
11596 	uint64_t pa;
11597 	uint64_t num_entries;
11598 	uint64_t start_va;
11599 	uint64_t end_va;
11600 };
11601 
11602 static kern_return_t
11603 pmap_dump_page_tables_recurse(pmap_t pmap,
11604     const tt_entry_t *ttp,
11605     unsigned int cur_level,
11606     unsigned int level_mask,
11607     uint64_t start_va,
11608     void *buf_start,
11609     void *buf_end,
11610     size_t *bytes_copied)
11611 {
11612 	const pt_attr_t * const pt_attr = pmap_get_pt_attr(pmap);
11613 	uint64_t num_entries = pt_attr_page_size(pt_attr) / sizeof(*ttp);
11614 
11615 	uint64_t size = pt_attr->pta_level_info[cur_level].size;
11616 	uint64_t valid_mask = pt_attr->pta_level_info[cur_level].valid_mask;
11617 	uint64_t type_mask = pt_attr->pta_level_info[cur_level].type_mask;
11618 	uint64_t type_block = pt_attr->pta_level_info[cur_level].type_block;
11619 
11620 	void *bufp = (uint8_t*)buf_start + *bytes_copied;
11621 
11622 	if (cur_level == pt_attr_root_level(pt_attr)) {
11623 		start_va &= ~(pt_attr->pta_level_info[cur_level].offmask);
11624 		num_entries = pmap_root_alloc_size(pmap) / sizeof(tt_entry_t);
11625 	}
11626 
11627 	uint64_t tt_size = num_entries * sizeof(tt_entry_t);
11628 	const tt_entry_t *tt_end = &ttp[num_entries];
11629 
11630 	if (((vm_offset_t)buf_end - (vm_offset_t)bufp) < (tt_size + sizeof(struct page_table_dump_header))) {
11631 		return KERN_INSUFFICIENT_BUFFER_SIZE;
11632 	}
11633 
11634 	if (level_mask & (1U << cur_level)) {
11635 		struct page_table_dump_header *header = (struct page_table_dump_header*)bufp;
11636 		header->pa = kvtophys_nofail((vm_offset_t)ttp);
11637 		header->num_entries = num_entries;
11638 		header->start_va = start_va;
11639 		header->end_va = start_va + (num_entries * size);
11640 
11641 		bcopy(ttp, (uint8_t*)bufp + sizeof(*header), tt_size);
11642 		*bytes_copied = *bytes_copied + sizeof(*header) + tt_size;
11643 	}
11644 	uint64_t current_va = start_va;
11645 
11646 	for (const tt_entry_t *ttep = ttp; ttep < tt_end; ttep++, current_va += size) {
11647 		tt_entry_t tte = *ttep;
11648 
11649 		if (!(tte & valid_mask)) {
11650 			continue;
11651 		}
11652 
11653 		if ((tte & type_mask) == type_block) {
11654 			continue;
11655 		} else {
11656 			if (cur_level >= pt_attr_leaf_level(pt_attr)) {
11657 				panic("%s: corrupt entry %#llx at %p, "
11658 				    "ttp=%p, cur_level=%u, bufp=%p, buf_end=%p",
11659 				    __FUNCTION__, tte, ttep,
11660 				    ttp, cur_level, bufp, buf_end);
11661 			}
11662 
11663 			const tt_entry_t *next_tt = (const tt_entry_t*)phystokv(tte & ARM_TTE_TABLE_MASK);
11664 
11665 			kern_return_t recurse_result = pmap_dump_page_tables_recurse(pmap, next_tt, cur_level + 1,
11666 			    level_mask, current_va, buf_start, buf_end, bytes_copied);
11667 
11668 			if (recurse_result != KERN_SUCCESS) {
11669 				return recurse_result;
11670 			}
11671 		}
11672 	}
11673 
11674 	return KERN_SUCCESS;
11675 }
11676 
11677 kern_return_t
11678 pmap_dump_page_tables(pmap_t pmap, void *bufp, void *buf_end, unsigned int level_mask, size_t *bytes_copied)
11679 {
11680 	if (not_in_kdp) {
11681 		panic("pmap_dump_page_tables must only be called from kernel debugger context");
11682 	}
11683 	return pmap_dump_page_tables_recurse(pmap, pmap->tte, pt_attr_root_level(pmap_get_pt_attr(pmap)),
11684 	           level_mask, pmap->min, bufp, buf_end, bytes_copied);
11685 }
11686 
11687 #else /* DEVELOPMENT || DEBUG */
11688 
11689 kern_return_t
11690 pmap_dump_page_tables(pmap_t pmap __unused, void *bufp __unused, void *buf_end __unused,
11691     unsigned int level_mask __unused, size_t *bytes_copied __unused)
11692 {
11693 	return KERN_NOT_SUPPORTED;
11694 }
11695 #endif /* !(DEVELOPMENT || DEBUG) */
11696 
11697 
11698 #ifdef CONFIG_XNUPOST
11699 static volatile bool pmap_test_took_fault = false;
11700 
11701 static bool
11702 pmap_test_fault_handler(arm_saved_state_t * state)
11703 {
11704 	bool retval                 = false;
11705 	uint64_t esr                = get_saved_state_esr(state);
11706 	esr_exception_class_t class = ESR_EC(esr);
11707 	fault_status_t fsc          = ISS_IA_FSC(ESR_ISS(esr));
11708 
11709 	if ((class == ESR_EC_DABORT_EL1) &&
11710 	    ((fsc == FSC_PERMISSION_FAULT_L3) || (fsc == FSC_ACCESS_FLAG_FAULT_L3))) {
11711 		pmap_test_took_fault = true;
11712 		/* return to the instruction immediately after the call to NX page */
11713 		set_saved_state_pc(state, get_saved_state_pc(state) + 4);
11714 		retval = true;
11715 	}
11716 
11717 	return retval;
11718 }
11719 
11720 // Disable KASAN instrumentation, as the test pmap's TTBR0 space will not be in the shadow map
11721 static NOKASAN bool
11722 pmap_test_access(pmap_t pmap, vm_map_address_t va, bool should_fault, bool is_write)
11723 {
11724 	pmap_t old_pmap = NULL;
11725 
11726 	pmap_test_took_fault = false;
11727 
11728 	/*
11729 	 * We're potentially switching pmaps without using the normal thread
11730 	 * mechanism; disable interrupts and preemption to avoid any unexpected
11731 	 * memory accesses.
11732 	 */
11733 	const boolean_t old_int_state = ml_set_interrupts_enabled(FALSE);
11734 	mp_disable_preemption();
11735 
11736 	if (pmap != NULL) {
11737 		old_pmap = current_pmap();
11738 		pmap_switch(pmap);
11739 
11740 		/* Disable PAN; pmap shouldn't be the kernel pmap. */
11741 #if __ARM_PAN_AVAILABLE__
11742 		__builtin_arm_wsr("pan", 0);
11743 #endif /* __ARM_PAN_AVAILABLE__ */
11744 	}
11745 
11746 	ml_expect_fault_begin(pmap_test_fault_handler, va);
11747 
11748 	if (is_write) {
11749 		*((volatile uint64_t*)(va)) = 0xdec0de;
11750 	} else {
11751 		volatile uint64_t tmp = *((volatile uint64_t*)(va));
11752 		(void)tmp;
11753 	}
11754 
11755 	/* Save the fault bool, and undo the gross stuff we did. */
11756 	bool took_fault = pmap_test_took_fault;
11757 	ml_expect_fault_end();
11758 
11759 	if (pmap != NULL) {
11760 #if __ARM_PAN_AVAILABLE__
11761 		__builtin_arm_wsr("pan", 1);
11762 #endif /* __ARM_PAN_AVAILABLE__ */
11763 
11764 		pmap_switch(old_pmap);
11765 	}
11766 
11767 	mp_enable_preemption();
11768 	ml_set_interrupts_enabled(old_int_state);
11769 	bool retval = (took_fault == should_fault);
11770 	return retval;
11771 }
11772 
11773 static bool
11774 pmap_test_read(pmap_t pmap, vm_map_address_t va, bool should_fault)
11775 {
11776 	bool retval = pmap_test_access(pmap, va, should_fault, false);
11777 
11778 	if (!retval) {
11779 		T_FAIL("%s: %s, "
11780 		    "pmap=%p, va=%p, should_fault=%u",
11781 		    __func__, should_fault ? "did not fault" : "faulted",
11782 		    pmap, (void*)va, (unsigned)should_fault);
11783 	}
11784 
11785 	return retval;
11786 }
11787 
11788 static bool
11789 pmap_test_write(pmap_t pmap, vm_map_address_t va, bool should_fault)
11790 {
11791 	bool retval = pmap_test_access(pmap, va, should_fault, true);
11792 
11793 	if (!retval) {
11794 		T_FAIL("%s: %s, "
11795 		    "pmap=%p, va=%p, should_fault=%u",
11796 		    __func__, should_fault ? "did not fault" : "faulted",
11797 		    pmap, (void*)va, (unsigned)should_fault);
11798 	}
11799 
11800 	return retval;
11801 }
11802 
11803 static bool
11804 pmap_test_check_refmod(pmap_paddr_t pa, unsigned int should_be_set)
11805 {
11806 	unsigned int should_be_clear = (~should_be_set) & (VM_MEM_REFERENCED | VM_MEM_MODIFIED);
11807 	unsigned int bits = pmap_get_refmod((ppnum_t)atop(pa));
11808 
11809 	bool retval = (((bits & should_be_set) == should_be_set) && ((bits & should_be_clear) == 0));
11810 
11811 	if (!retval) {
11812 		T_FAIL("%s: bits=%u, "
11813 		    "pa=%p, should_be_set=%u",
11814 		    __func__, bits,
11815 		    (void*)pa, should_be_set);
11816 	}
11817 
11818 	return retval;
11819 }
11820 
11821 static __attribute__((noinline)) bool
11822 pmap_test_read_write(pmap_t pmap, vm_map_address_t va, bool allow_read, bool allow_write)
11823 {
11824 	bool retval = (pmap_test_read(pmap, va, !allow_read) | pmap_test_write(pmap, va, !allow_write));
11825 	return retval;
11826 }
11827 
11828 static int
11829 pmap_test_test_config(unsigned int flags)
11830 {
11831 	T_LOG("running pmap_test_test_config flags=0x%X", flags);
11832 	unsigned int map_count = 0;
11833 	unsigned long page_ratio = 0;
11834 	pmap_t pmap = pmap_create_options(NULL, 0, flags);
11835 
11836 	if (!pmap) {
11837 		panic("Failed to allocate pmap");
11838 	}
11839 
11840 	__unused const pt_attr_t * const pt_attr = pmap_get_pt_attr(pmap);
11841 	uintptr_t native_page_size = pt_attr_page_size(native_pt_attr);
11842 	uintptr_t pmap_page_size = pt_attr_page_size(pt_attr);
11843 	uintptr_t pmap_twig_size = pt_attr_twig_size(pt_attr);
11844 
11845 	if (pmap_page_size <= native_page_size) {
11846 		page_ratio = native_page_size / pmap_page_size;
11847 	} else {
11848 		/*
11849 		 * We claim to support a page_ratio of less than 1, which is
11850 		 * not currently supported by the pmap layer; panic.
11851 		 */
11852 		panic("%s: page_ratio < 1, native_page_size=%lu, pmap_page_size=%lu"
11853 		    "flags=%u",
11854 		    __func__, native_page_size, pmap_page_size,
11855 		    flags);
11856 	}
11857 
11858 	if (PAGE_RATIO > 1) {
11859 		/*
11860 		 * The kernel is deliberately pretending to have 16KB pages.
11861 		 * The pmap layer has code that supports this, so pretend the
11862 		 * page size is larger than it is.
11863 		 */
11864 		pmap_page_size = PAGE_SIZE;
11865 		native_page_size = PAGE_SIZE;
11866 	}
11867 
11868 	/*
11869 	 * Get two pages from the VM; one to be mapped wired, and one to be
11870 	 * mapped nonwired.
11871 	 */
11872 	vm_page_t unwired_vm_page = vm_page_grab();
11873 	vm_page_t wired_vm_page = vm_page_grab();
11874 
11875 	if ((unwired_vm_page == VM_PAGE_NULL) || (wired_vm_page == VM_PAGE_NULL)) {
11876 		panic("Failed to grab VM pages");
11877 	}
11878 
11879 	ppnum_t pn = VM_PAGE_GET_PHYS_PAGE(unwired_vm_page);
11880 	ppnum_t wired_pn = VM_PAGE_GET_PHYS_PAGE(wired_vm_page);
11881 
11882 	pmap_paddr_t pa = ptoa(pn);
11883 	pmap_paddr_t wired_pa = ptoa(wired_pn);
11884 
11885 	/*
11886 	 * We'll start mappings at the second twig TT.  This keeps us from only
11887 	 * using the first entry in each TT, which would trivially be address
11888 	 * 0; one of the things we will need to test is retrieving the VA for
11889 	 * a given PTE.
11890 	 */
11891 	vm_map_address_t va_base = pmap_twig_size;
11892 	vm_map_address_t wired_va_base = ((2 * pmap_twig_size) - pmap_page_size);
11893 
11894 	if (wired_va_base < (va_base + (page_ratio * pmap_page_size))) {
11895 		/*
11896 		 * Not exactly a functional failure, but this test relies on
11897 		 * there being a spare PTE slot we can use to pin the TT.
11898 		 */
11899 		panic("Cannot pin translation table");
11900 	}
11901 
11902 	/*
11903 	 * Create the wired mapping; this will prevent the pmap layer from
11904 	 * reclaiming our test TTs, which would interfere with this test
11905 	 * ("interfere" -> "make it panic").
11906 	 */
11907 	pmap_enter_addr(pmap, wired_va_base, wired_pa, VM_PROT_READ, VM_PROT_READ, 0, true, PMAP_MAPPING_TYPE_INFER);
11908 
11909 	T_LOG("Validate that kernel cannot write to SPTM memory.");
11910 	pt_entry_t * ptep = pmap_pte(pmap, va_base);
11911 	pmap_test_write(NULL, (vm_map_address_t)ptep, true);
11912 
11913 	/*
11914 	 * Create read-only mappings of the nonwired page; if the pmap does
11915 	 * not use the same page size as the kernel, create multiple mappings
11916 	 * so that the kernel page is fully mapped.
11917 	 */
11918 	for (map_count = 0; map_count < page_ratio; map_count++) {
11919 		pmap_enter_addr(pmap, va_base + (pmap_page_size * map_count), pa + (pmap_page_size * (map_count)),
11920 		    VM_PROT_READ, VM_PROT_READ, 0, false, PMAP_MAPPING_TYPE_INFER);
11921 	}
11922 
11923 	/* Validate that all the PTEs have the expected PA and VA. */
11924 	for (map_count = 0; map_count < page_ratio; map_count++) {
11925 		ptep = pmap_pte(pmap, va_base + (pmap_page_size * map_count));
11926 
11927 		if (pte_to_pa(*ptep) != (pa + (pmap_page_size * map_count))) {
11928 			T_FAIL("Unexpected pa=%p, expected %p, map_count=%u",
11929 			    (void*)pte_to_pa(*ptep), (void*)(pa + (pmap_page_size * map_count)), map_count);
11930 		}
11931 
11932 		if (ptep_get_va(ptep) != (va_base + (pmap_page_size * map_count))) {
11933 			T_FAIL("Unexpected va=%p, expected %p, map_count=%u",
11934 			    (void*)ptep_get_va(ptep), (void*)(va_base + (pmap_page_size * map_count)), map_count);
11935 		}
11936 	}
11937 
11938 	T_LOG("Validate that reads to our mapping do not fault.");
11939 	pmap_test_read(pmap, va_base, false);
11940 
11941 	T_LOG("Validate that writes to our mapping fault.");
11942 	pmap_test_write(pmap, va_base, true);
11943 
11944 	T_LOG("Make the first mapping writable.");
11945 	pmap_enter_addr(pmap, va_base, pa, VM_PROT_READ | VM_PROT_WRITE, VM_PROT_READ | VM_PROT_WRITE, 0, false, PMAP_MAPPING_TYPE_INFER);
11946 
11947 	T_LOG("Validate that writes to our mapping do not fault.");
11948 	pmap_test_write(pmap, va_base, false);
11949 
11950 	/*
11951 	 * For page ratios of greater than 1: validate that writes to the other
11952 	 * mappings still fault.  Remove the mappings afterwards (we're done
11953 	 * with page ratio testing).
11954 	 */
11955 	for (map_count = 1; map_count < page_ratio; map_count++) {
11956 		pmap_test_write(pmap, va_base + (pmap_page_size * map_count), true);
11957 		pmap_remove(pmap, va_base + (pmap_page_size * map_count), va_base + (pmap_page_size * map_count) + pmap_page_size);
11958 	}
11959 
11960 	/* Remove remaining mapping */
11961 	pmap_remove(pmap, va_base, va_base + pmap_page_size);
11962 
11963 	T_LOG("Make the first mapping execute-only");
11964 	pmap_enter_addr(pmap, va_base, pa, VM_PROT_EXECUTE, VM_PROT_EXECUTE, 0, false, PMAP_MAPPING_TYPE_INFER);
11965 
11966 
11967 	T_LOG("Validate that reads to our mapping do not fault.");
11968 	pmap_test_read(pmap, va_base, false);
11969 
11970 	T_LOG("Validate that reads to our mapping do not fault.");
11971 	pmap_test_read(pmap, va_base, false);
11972 
11973 	T_LOG("Validate that writes to our mapping fault.");
11974 	pmap_test_write(pmap, va_base, true);
11975 
11976 	pmap_remove(pmap, va_base, va_base + pmap_page_size);
11977 
11978 	T_LOG("Mark the page unreferenced and unmodified.");
11979 	pmap_clear_refmod(pn, VM_MEM_MODIFIED | VM_MEM_REFERENCED);
11980 	pmap_test_check_refmod(pa, 0);
11981 
11982 	/*
11983 	 * Begin testing the ref/mod state machine.  Re-enter the mapping with
11984 	 * different protection/fault_type settings, and confirm that the
11985 	 * ref/mod state matches our expectations at each step.
11986 	 */
11987 	T_LOG("!ref/!mod: read, no fault.  Expect ref/!mod");
11988 	pmap_enter_addr(pmap, va_base, pa, VM_PROT_READ, VM_PROT_NONE, 0, false, PMAP_MAPPING_TYPE_INFER);
11989 	pmap_test_check_refmod(pa, VM_MEM_REFERENCED);
11990 
11991 	T_LOG("!ref/!mod: read, read fault.  Expect ref/!mod");
11992 	pmap_clear_refmod(pn, VM_MEM_MODIFIED | VM_MEM_REFERENCED);
11993 	pmap_enter_addr(pmap, va_base, pa, VM_PROT_READ, VM_PROT_READ, 0, false, PMAP_MAPPING_TYPE_INFER);
11994 	pmap_test_check_refmod(pa, VM_MEM_REFERENCED);
11995 
11996 	T_LOG("!ref/!mod: rw, read fault.  Expect ref/!mod");
11997 	pmap_clear_refmod(pn, VM_MEM_MODIFIED | VM_MEM_REFERENCED);
11998 	pmap_enter_addr(pmap, va_base, pa, VM_PROT_READ | VM_PROT_WRITE, VM_PROT_NONE, 0, false, PMAP_MAPPING_TYPE_INFER);
11999 	pmap_test_check_refmod(pa, VM_MEM_REFERENCED);
12000 
12001 	T_LOG("ref/!mod: rw, read fault.  Expect ref/!mod");
12002 	pmap_enter_addr(pmap, va_base, pa, VM_PROT_READ | VM_PROT_WRITE, VM_PROT_READ, 0, false, PMAP_MAPPING_TYPE_INFER);
12003 	pmap_test_check_refmod(pa, VM_MEM_REFERENCED);
12004 
12005 	T_LOG("!ref/!mod: rw, rw fault.  Expect ref/mod");
12006 	pmap_clear_refmod(pn, VM_MEM_MODIFIED | VM_MEM_REFERENCED);
12007 	pmap_enter_addr(pmap, va_base, pa, VM_PROT_READ | VM_PROT_WRITE, VM_PROT_READ | VM_PROT_WRITE, 0, false, PMAP_MAPPING_TYPE_INFER);
12008 	pmap_test_check_refmod(pa, VM_MEM_REFERENCED | VM_MEM_MODIFIED);
12009 
12010 	/*
12011 	 * Shared memory testing; we'll have two mappings; one read-only,
12012 	 * one read-write.
12013 	 */
12014 	vm_map_address_t rw_base = va_base;
12015 	vm_map_address_t ro_base = va_base + pmap_page_size;
12016 
12017 	pmap_enter_addr(pmap, rw_base, pa, VM_PROT_READ | VM_PROT_WRITE, VM_PROT_READ | VM_PROT_WRITE, 0, false, PMAP_MAPPING_TYPE_INFER);
12018 	pmap_enter_addr(pmap, ro_base, pa, VM_PROT_READ, VM_PROT_READ, 0, false, PMAP_MAPPING_TYPE_INFER);
12019 
12020 	/*
12021 	 * Test that we take faults as expected for unreferenced/unmodified
12022 	 * pages.  Also test the arm_fast_fault interface, to ensure that
12023 	 * mapping permissions change as expected.
12024 	 */
12025 	T_LOG("!ref/!mod: expect no access");
12026 	pmap_clear_refmod(pn, VM_MEM_MODIFIED | VM_MEM_REFERENCED);
12027 	pmap_test_read_write(pmap, ro_base, false, false);
12028 	pmap_test_read_write(pmap, rw_base, false, false);
12029 
12030 	T_LOG("Read fault; expect !ref/!mod -> ref/!mod, read access");
12031 	arm_fast_fault(pmap, rw_base, VM_PROT_READ, false, false);
12032 	pmap_test_check_refmod(pa, VM_MEM_REFERENCED);
12033 	pmap_test_read_write(pmap, ro_base, true, false);
12034 	pmap_test_read_write(pmap, rw_base, true, false);
12035 
12036 	T_LOG("Write fault; expect ref/!mod -> ref/mod, read and write access");
12037 	arm_fast_fault(pmap, rw_base, VM_PROT_READ | VM_PROT_WRITE, false, false);
12038 	pmap_test_check_refmod(pa, VM_MEM_REFERENCED | VM_MEM_MODIFIED);
12039 	pmap_test_read_write(pmap, ro_base, true, false);
12040 	pmap_test_read_write(pmap, rw_base, true, true);
12041 
12042 	T_LOG("Write fault; expect !ref/!mod -> ref/mod, read and write access");
12043 	pmap_clear_refmod(pn, VM_MEM_MODIFIED | VM_MEM_REFERENCED);
12044 	arm_fast_fault(pmap, rw_base, VM_PROT_READ | VM_PROT_WRITE, false, false);
12045 	pmap_test_check_refmod(pa, VM_MEM_REFERENCED | VM_MEM_MODIFIED);
12046 	pmap_test_read_write(pmap, ro_base, true, false);
12047 	pmap_test_read_write(pmap, rw_base, true, true);
12048 
12049 	T_LOG("RW protect both mappings; should not change protections.");
12050 	pmap_protect(pmap, ro_base, ro_base + pmap_page_size, VM_PROT_READ | VM_PROT_WRITE);
12051 	pmap_protect(pmap, rw_base, rw_base + pmap_page_size, VM_PROT_READ | VM_PROT_WRITE);
12052 	pmap_test_read_write(pmap, ro_base, true, false);
12053 	pmap_test_read_write(pmap, rw_base, true, true);
12054 
12055 	T_LOG("Read protect both mappings; RW mapping should become RO.");
12056 	pmap_protect(pmap, ro_base, ro_base + pmap_page_size, VM_PROT_READ);
12057 	pmap_protect(pmap, rw_base, rw_base + pmap_page_size, VM_PROT_READ);
12058 	pmap_test_read_write(pmap, ro_base, true, false);
12059 	pmap_test_read_write(pmap, rw_base, true, false);
12060 
12061 	T_LOG("RW protect the page; mappings should not change protections.");
12062 	pmap_enter_addr(pmap, rw_base, pa, VM_PROT_READ | VM_PROT_WRITE, VM_PROT_READ | VM_PROT_WRITE, 0, false, PMAP_MAPPING_TYPE_INFER);
12063 	pmap_page_protect(pn, VM_PROT_ALL);
12064 	pmap_test_read_write(pmap, ro_base, true, false);
12065 	pmap_test_read_write(pmap, rw_base, true, true);
12066 
12067 	T_LOG("Read protect the page; RW mapping should become RO.");
12068 	pmap_page_protect(pn, VM_PROT_READ);
12069 	pmap_test_read_write(pmap, ro_base, true, false);
12070 	pmap_test_read_write(pmap, rw_base, true, false);
12071 
12072 	T_LOG("Validate that disconnect removes all known mappings of the page.");
12073 	pmap_disconnect(pn);
12074 	if (!pmap_verify_free(pn)) {
12075 		T_FAIL("Page still has mappings");
12076 	}
12077 
12078 #if defined(ARM_LARGE_MEMORY)
12079 #define PMAP_TEST_LARGE_MEMORY_VA 64 * (1ULL << 40) /* 64 TB */
12080 
12081 	T_LOG("Create new wired mapping in the extended address space enabled by ARM_LARGE_MEMORY.");
12082 	pmap_enter_addr(pmap, PMAP_TEST_LARGE_MEMORY_VA, wired_pa, VM_PROT_READ | VM_PROT_WRITE, VM_PROT_READ | VM_PROT_WRITE, 0, true, PMAP_MAPPING_TYPE_INFER);
12083 	pmap_test_read_write(pmap, PMAP_TEST_LARGE_MEMORY_VA, true, true);
12084 	pmap_remove(pmap, PMAP_TEST_LARGE_MEMORY_VA, PMAP_TEST_LARGE_MEMORY_VA + pmap_page_size);
12085 #endif /* ARM_LARGE_MEMORY */
12086 
12087 	T_LOG("Remove the wired mapping, so we can tear down the test map.");
12088 	pmap_remove(pmap, wired_va_base, wired_va_base + pmap_page_size);
12089 	pmap_destroy(pmap);
12090 
12091 	T_LOG("Release the pages back to the VM.");
12092 	vm_page_lock_queues();
12093 	vm_page_free(unwired_vm_page);
12094 	vm_page_free(wired_vm_page);
12095 	vm_page_unlock_queues();
12096 
12097 	T_LOG("Testing successful!");
12098 	return 0;
12099 }
12100 
12101 kern_return_t
12102 pmap_test(void)
12103 {
12104 	T_LOG("Starting pmap_tests");
12105 	int flags = 0;
12106 	flags |= PMAP_CREATE_64BIT;
12107 
12108 #if __ARM_MIXED_PAGE_SIZE__ && !CONFIG_SPTM
12109 	T_LOG("Testing VM_PAGE_SIZE_4KB");
12110 	pmap_test_test_config(flags | PMAP_CREATE_FORCE_4K_PAGES);
12111 	T_LOG("Testing VM_PAGE_SIZE_16KB");
12112 	pmap_test_test_config(flags);
12113 #else /* __ARM_MIXED_PAGE_SIZE__ */
12114 	pmap_test_test_config(flags);
12115 #endif /* __ARM_MIXED_PAGE_SIZE__ */
12116 
12117 	T_PASS("completed pmap_test successfully");
12118 	return KERN_SUCCESS;
12119 }
12120 #endif /* CONFIG_XNUPOST */
12121 
12122 /*
12123  * The following function should never make it to RELEASE code, since
12124  * it provides a way to get the PPL to modify text pages.
12125  */
12126 #if DEVELOPMENT || DEBUG
12127 
12128 /**
12129  * Forcibly overwrite executable text with an illegal instruction.
12130  *
12131  * @note Only used for xnu unit testing.
12132  *
12133  * @param pa The physical address to corrupt.
12134  *
12135  * @return KERN_SUCCESS on success.
12136  */
12137 kern_return_t
12138 pmap_test_text_corruption(pmap_paddr_t pa __unused)
12139 {
12140 	/*
12141 	 * SPTM TODO: implement an SPTM version of this.
12142 	 * The physical apertue is owned by the SPTM and text
12143 	 * pages have RO physical aperture mappings.
12144 	 */
12145 	return KERN_SUCCESS;
12146 }
12147 
12148 #endif /* DEVELOPMENT || DEBUG */
12149 
12150