sys/arm/arm/pmap-v6.c

   1 /* From: $NetBSD: pmap.c,v 1.148 2004/04/03 04:35:48 bsh Exp $ */
   2 /*-
   3  * Copyright 2011 Semihalf
   4  * Copyright 2004 Olivier Houchard.
   5  * Copyright 2003 Wasabi Systems, Inc.
   6  * All rights reserved.
   7  *
   8  * Written by Steve C. Woodford for Wasabi Systems, Inc.
   9  *
  10  * Redistribution and use in source and binary forms, with or without
  11  * modification, are permitted provided that the following conditions
  12  * are met:
  13  * 1. Redistributions of source code must retain the above copyright
  14  *    notice, this list of conditions and the following disclaimer.
  15  * 2. Redistributions in binary form must reproduce the above copyright
  16  *    notice, this list of conditions and the following disclaimer in the
  17  *    documentation and/or other materials provided with the distribution.
  18  * 3. All advertising materials mentioning features or use of this software
  19  *    must display the following acknowledgement:
  20  *      This product includes software developed for the NetBSD Project by
  21  *      Wasabi Systems, Inc.
  22  * 4. The name of Wasabi Systems, Inc. may not be used to endorse
  23  *    or promote products derived from this software without specific prior
  24  *    written permission.
  25  *
  26  * THIS SOFTWARE IS PROVIDED BY WASABI SYSTEMS, INC. ``AS IS'' AND
  27  * ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT LIMITED
  28  * TO, THE IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS FOR A PARTICULAR
  29  * PURPOSE ARE DISCLAIMED.  IN NO EVENT SHALL WASABI SYSTEMS, INC
  30  * BE LIABLE FOR ANY DIRECT, INDIRECT, INCIDENTAL, SPECIAL, EXEMPLARY, OR
  31  * CONSEQUENTIAL DAMAGES (INCLUDING, BUT NOT LIMITED TO, PROCUREMENT OF
  32  * SUBSTITUTE GOODS OR SERVICES; LOSS OF USE, DATA, OR PROFITS; OR BUSINESS
  33  * INTERRUPTION) HOWEVER CAUSED AND ON ANY THEORY OF LIABILITY, WHETHER IN
  34  * CONTRACT, STRICT LIABILITY, OR TORT (INCLUDING NEGLIGENCE OR OTHERWISE)
  35  * ARISING IN ANY WAY OUT OF THE USE OF THIS SOFTWARE, EVEN IF ADVISED OF THE
  36  * POSSIBILITY OF SUCH DAMAGE.
  37  *
  38  * From: FreeBSD: src/sys/arm/arm/pmap.c,v 1.113 2009/07/24 13:50:29
  39  */
  40
  41 /*-
  42  * Copyright (c) 2002-2003 Wasabi Systems, Inc.
  43  * Copyright (c) 2001 Richard Earnshaw
  44  * Copyright (c) 2001-2002 Christopher Gilbert
  45  * All rights reserved.
  46  *
  47  * 1. Redistributions of source code must retain the above copyright
  48  *    notice, this list of conditions and the following disclaimer.
  49  * 2. Redistributions in binary form must reproduce the above copyright
  50  *    notice, this list of conditions and the following disclaimer in the
  51  *    documentation and/or other materials provided with the distribution.
  52  * 3. The name of the company nor the name of the author may be used to
  53  *    endorse or promote products derived from this software without specific
  54  *    prior written permission.
  55  *
  56  * THIS SOFTWARE IS PROVIDED BY THE AUTHOR ``AS IS'' AND ANY EXPRESS OR IMPLIED
  57  * WARRANTIES, INCLUDING, BUT NOT LIMITED TO, THE IMPLIED WARRANTIES OF
  58  * MERCHANTABILITY AND FITNESS FOR A PARTICULAR PURPOSE ARE DISCLAIMED.
  59  * IN NO EVENT SHALL THE AUTHOR OR CONTRIBUTORS BE LIABLE FOR ANY DIRECT,
  60  * INDIRECT, INCIDENTAL, SPECIAL, EXEMPLARY, OR CONSEQUENTIAL DAMAGES
  61  * (INCLUDING, BUT NOT LIMITED TO, PROCUREMENT OF SUBSTITUTE GOODS OR
  62  * SERVICES; LOSS OF USE, DATA, OR PROFITS; OR BUSINESS INTERRUPTION)
  63  * HOWEVER CAUSED AND ON ANY THEORY OF LIABILITY, WHETHER IN CONTRACT, STRICT
  64  * LIABILITY, OR TORT (INCLUDING NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY
  65  * OUT OF THE USE OF THIS SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF
  66  * SUCH DAMAGE.
  67  */
  68 /*-
  69  * Copyright (c) 1999 The NetBSD Foundation, Inc.
  70  * All rights reserved.
  71  *
  72  * This code is derived from software contributed to The NetBSD Foundation
  73  * by Charles M. Hannum.
  74  *
  75  * Redistribution and use in source and binary forms, with or without
  76  * modification, are permitted provided that the following conditions
  77  * are met:
  78  * 1. Redistributions of source code must retain the above copyright
  79  *    notice, this list of conditions and the following disclaimer.
  80  * 2. Redistributions in binary form must reproduce the above copyright
  81  *    notice, this list of conditions and the following disclaimer in the
  82  *    documentation and/or other materials provided with the distribution.
  83  *
  84  * THIS SOFTWARE IS PROVIDED BY THE NETBSD FOUNDATION, INC. AND CONTRIBUTORS
  85  * ``AS IS'' AND ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT LIMITED
  86  * TO, THE IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS FOR A PARTICULAR
  87  * PURPOSE ARE DISCLAIMED.  IN NO EVENT SHALL THE FOUNDATION OR CONTRIBUTORS
  88  * BE LIABLE FOR ANY DIRECT, INDIRECT, INCIDENTAL, SPECIAL, EXEMPLARY, OR
  89  * CONSEQUENTIAL DAMAGES (INCLUDING, BUT NOT LIMITED TO, PROCUREMENT OF
  90  * SUBSTITUTE GOODS OR SERVICES; LOSS OF USE, DATA, OR PROFITS; OR BUSINESS
  91  * INTERRUPTION) HOWEVER CAUSED AND ON ANY THEORY OF LIABILITY, WHETHER IN
  92  * CONTRACT, STRICT LIABILITY, OR TORT (INCLUDING NEGLIGENCE OR OTHERWISE)
  93  * ARISING IN ANY WAY OUT OF THE USE OF THIS SOFTWARE, EVEN IF ADVISED OF THE
  94  * POSSIBILITY OF SUCH DAMAGE.
  95  */
  96
  97 /*-
  98  * Copyright (c) 1994-1998 Mark Brinicombe.
  99  * Copyright (c) 1994 Brini.
 100  * All rights reserved.
 101  *
 102  * This code is derived from software written for Brini by Mark Brinicombe
 103  *
 104  * Redistribution and use in source and binary forms, with or without
 105  * modification, are permitted provided that the following conditions
 106  * are met:
 107  * 1. Redistributions of source code must retain the above copyright
 108  *    notice, this list of conditions and the following disclaimer.
 109  * 2. Redistributions in binary form must reproduce the above copyright
 110  *    notice, this list of conditions and the following disclaimer in the
 111  *    documentation and/or other materials provided with the distribution.
 112  * 3. All advertising materials mentioning features or use of this software
 113  *    must display the following acknowledgement:
 114  *      This product includes software developed by Mark Brinicombe.
 115  * 4. The name of the author may not be used to endorse or promote products
 116  *    derived from this software without specific prior written permission.
 117  *
 118  * THIS SOFTWARE IS PROVIDED BY THE AUTHOR ``AS IS'' AND ANY EXPRESS OR
 119  * IMPLIED WARRANTIES, INCLUDING, BUT NOT LIMITED TO, THE IMPLIED WARRANTIES
 120  * OF MERCHANTABILITY AND FITNESS FOR A PARTICULAR PURPOSE ARE DISCLAIMED.
 121  * IN NO EVENT SHALL THE AUTHOR BE LIABLE FOR ANY DIRECT, INDIRECT,
 122  * INCIDENTAL, SPECIAL, EXEMPLARY, OR CONSEQUENTIAL DAMAGES (INCLUDING, BUT
 123  * NOT LIMITED TO, PROCUREMENT OF SUBSTITUTE GOODS OR SERVICES; LOSS OF USE,
 124  * DATA, OR PROFITS; OR BUSINESS INTERRUPTION) HOWEVER CAUSED AND ON ANY
 125  * THEORY OF LIABILITY, WHETHER IN CONTRACT, STRICT LIABILITY, OR TORT
 126  * (INCLUDING NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY OUT OF THE USE OF
 127  *
 128  * RiscBSD kernel project
 129  *
 130  * pmap.c
 131  *
 132  * Machine dependant vm stuff
 133  *
 134  * Created      : 20/09/94
 135  */
 136
 137 /*
 138  * Special compilation symbols
 139  * PMAP_DEBUG           - Build in pmap_debug_level code
 140  *
 141  * Note that pmap_mapdev() and pmap_unmapdev() are implemented in arm/devmap.c
 142 */
 143 /* Include header files */
 144
 145 #include "opt_vm.h"
 146 #include "opt_pmap.h"
 147
 148 #include <sys/cdefs.h>
 149 __FBSDID("$FreeBSD$");
 150 #include <sys/param.h>
 151 #include <sys/systm.h>
 152 #include <sys/kernel.h>
 153 #include <sys/ktr.h>
 154 #include <sys/lock.h>
 155 #include <sys/proc.h>
 156 #include <sys/malloc.h>
 157 #include <sys/msgbuf.h>
 158 #include <sys/mutex.h>
 159 #include <sys/vmmeter.h>
 160 #include <sys/mman.h>
 161 #include <sys/rwlock.h>
 162 #include <sys/smp.h>
 163 #include <sys/sched.h>
 164 #include <sys/sysctl.h>
 165
 166 #include <vm/vm.h>
 167 #include <vm/vm_param.h>
 168 #include <vm/uma.h>
 169 #include <vm/pmap.h>
 170 #include <vm/vm_kern.h>
 171 #include <vm/vm_object.h>
 172 #include <vm/vm_map.h>
 173 #include <vm/vm_page.h>
 174 #include <vm/vm_pageout.h>
 175 #include <vm/vm_extern.h>
 176 #include <vm/vm_reserv.h>
 177
 178 #include <machine/md_var.h>
 179 #include <machine/cpu.h>
 180 #include <machine/cpufunc.h>
 181 #include <machine/pcb.h>
 182
 183 #ifdef DEBUG
 184 extern int last_fault_code;
 185 #endif
 186
 187 #ifdef PMAP_DEBUG
 188 #define PDEBUG(_lev_,_stat_) \
 189         if (pmap_debug_level >= (_lev_)) \
 190                 ((_stat_))
 191 #define dprintf printf
 192
 193 int pmap_debug_level = 0;
 194 #define PMAP_INLINE
 195 #else   /* PMAP_DEBUG */
 196 #define PDEBUG(_lev_,_stat_) /* Nothing */
 197 #define dprintf(x, arg...)
 198 #define PMAP_INLINE __inline
 199 #endif  /* PMAP_DEBUG */
 200
 201 #ifdef PV_STATS
 202 #define PV_STAT(x)      do { x ; } while (0)
 203 #else
 204 #define PV_STAT(x)      do { } while (0)
 205 #endif
 206
 207 #define pa_to_pvh(pa)   (&pv_table[pa_index(pa)])
 208
 209 #ifdef ARM_L2_PIPT
 210 #define pmap_l2cache_wbinv_range(va, pa, size) cpu_l2cache_wbinv_range((pa), (size))
 211 #define pmap_l2cache_inv_range(va, pa, size) cpu_l2cache_inv_range((pa), (size))
 212 #else
 213 #define pmap_l2cache_wbinv_range(va, pa, size) cpu_l2cache_wbinv_range((va), (size))
 214 #define pmap_l2cache_inv_range(va, pa, size) cpu_l2cache_inv_range((va), (size))
 215 #endif
 216
 217 extern struct pv_addr systempage;
 218
 219 /*
 220  * Internal function prototypes
 221  */
 222
 223 static PMAP_INLINE
 224 struct pv_entry         *pmap_find_pv(struct md_page *, pmap_t, vm_offset_t);
 225 static void             pmap_free_pv_chunk(struct pv_chunk *pc);
 226 static void             pmap_free_pv_entry(pmap_t pmap, pv_entry_t pv);
 227 static pv_entry_t       pmap_get_pv_entry(pmap_t pmap, boolean_t try);
 228 static vm_page_t        pmap_pv_reclaim(pmap_t locked_pmap);
 229 static boolean_t        pmap_pv_insert_section(pmap_t, vm_offset_t,
 230     vm_paddr_t);
 231 static struct pv_entry  *pmap_remove_pv(struct vm_page *, pmap_t, vm_offset_t);
 232 static int              pmap_pvh_wired_mappings(struct md_page *, int);
 233
 234 static int              pmap_enter_locked(pmap_t, vm_offset_t, vm_page_t,
 235     vm_prot_t, u_int);
 236 static vm_paddr_t       pmap_extract_locked(pmap_t pmap, vm_offset_t va);
 237 static void             pmap_alloc_l1(pmap_t);
 238 static void             pmap_free_l1(pmap_t);
 239
 240 static void             pmap_map_section(pmap_t, vm_offset_t, vm_offset_t,
 241     vm_prot_t, boolean_t);
 242 static void             pmap_promote_section(pmap_t, vm_offset_t);
 243 static boolean_t        pmap_demote_section(pmap_t, vm_offset_t);
 244 static boolean_t        pmap_enter_section(pmap_t, vm_offset_t, vm_page_t,
 245     vm_prot_t);
 246 static void             pmap_remove_section(pmap_t, vm_offset_t);
 247
 248 static int              pmap_clearbit(struct vm_page *, u_int);
 249
 250 static struct l2_bucket *pmap_get_l2_bucket(pmap_t, vm_offset_t);
 251 static struct l2_bucket *pmap_alloc_l2_bucket(pmap_t, vm_offset_t);
 252 static void             pmap_free_l2_bucket(pmap_t, struct l2_bucket *, u_int);
 253 static vm_offset_t      kernel_pt_lookup(vm_paddr_t);
 254
 255 static MALLOC_DEFINE(M_VMPMAP, "pmap", "PMAP L1");
 256
 257 vm_offset_t virtual_avail;      /* VA of first avail page (after kernel bss) */
 258 vm_offset_t virtual_end;        /* VA of last avail page (end of kernel AS) */
 259 vm_offset_t pmap_curmaxkvaddr;
 260 vm_paddr_t kernel_l1pa;
 261
 262 vm_offset_t kernel_vm_end = 0;
 263
 264 vm_offset_t vm_max_kernel_address;
 265
 266 struct pmap kernel_pmap_store;
 267
 268 /*
 269  * Resources for quickly copying and zeroing pages using virtual address space
 270  * and page table entries that are pre-allocated per-CPU by pmap_init().
 271  */
 272 struct czpages {
 273         struct  mtx     lock;
 274         pt_entry_t      *srcptep;
 275         pt_entry_t      *dstptep;
 276         vm_offset_t     srcva;
 277         vm_offset_t     dstva;
 278 };
 279 static struct czpages cpu_czpages[MAXCPU];
 280
 281 static void             pmap_init_l1(struct l1_ttable *, pd_entry_t *);
 282 /*
 283  * These routines are called when the CPU type is identified to set up
 284  * the PTE prototypes, cache modes, etc.
 285  *
 286  * The variables are always here, just in case LKMs need to reference
 287  * them (though, they shouldn't).
 288  */
 289 static void pmap_set_prot(pt_entry_t *pte, vm_prot_t prot, uint8_t user);
 290 pt_entry_t      pte_l1_s_cache_mode;
 291 pt_entry_t      pte_l1_s_cache_mode_pt;
 292
 293 pt_entry_t      pte_l2_l_cache_mode;
 294 pt_entry_t      pte_l2_l_cache_mode_pt;
 295
 296 pt_entry_t      pte_l2_s_cache_mode;
 297 pt_entry_t      pte_l2_s_cache_mode_pt;
 298
 299 struct msgbuf *msgbufp = 0;
 300
 301 /*
 302  * Crashdump maps.
 303  */
 304 static caddr_t crashdumpmap;
 305
 306 extern void bcopy_page(vm_offset_t, vm_offset_t);
 307 extern void bzero_page(vm_offset_t);
 308
 309 char *_tmppt;
 310
 311 /*
 312  * Metadata for L1 translation tables.
 313  */
 314 struct l1_ttable {
 315         /* Entry on the L1 Table list */
 316         SLIST_ENTRY(l1_ttable) l1_link;
 317
 318         /* Entry on the L1 Least Recently Used list */
 319         TAILQ_ENTRY(l1_ttable) l1_lru;
 320
 321         /* Track how many domains are allocated from this L1 */
 322         volatile u_int l1_domain_use_count;
 323
 324         /*
 325          * A free-list of domain numbers for this L1.
 326          * We avoid using ffs() and a bitmap to track domains since ffs()
 327          * is slow on ARM.
 328          */
 329         u_int8_t l1_domain_first;
 330         u_int8_t l1_domain_free[PMAP_DOMAINS];
 331
 332         /* Physical address of this L1 page table */
 333         vm_paddr_t l1_physaddr;
 334
 335         /* KVA of this L1 page table */
 336         pd_entry_t *l1_kva;
 337 };
 338
 339 /*
 340  * Convert a virtual address into its L1 table index. That is, the
 341  * index used to locate the L2 descriptor table pointer in an L1 table.
 342  * This is basically used to index l1->l1_kva[].
 343  *
 344  * Each L2 descriptor table represents 1MB of VA space.
 345  */
 346 #define L1_IDX(va)              (((vm_offset_t)(va)) >> L1_S_SHIFT)
 347
 348 /*
 349  * L1 Page Tables are tracked using a Least Recently Used list.
 350  *  - New L1s are allocated from the HEAD.
 351  *  - Freed L1s are added to the TAIl.
 352  *  - Recently accessed L1s (where an 'access' is some change to one of
 353  *    the userland pmaps which owns this L1) are moved to the TAIL.
 354  */
 355 static TAILQ_HEAD(, l1_ttable) l1_lru_list;
 356 /*
 357  * A list of all L1 tables
 358  */
 359 static SLIST_HEAD(, l1_ttable) l1_list;
 360 static struct mtx l1_lru_lock;
 361
 362 /*
 363  * The l2_dtable tracks L2_BUCKET_SIZE worth of L1 slots.
 364  *
 365  * This is normally 16MB worth L2 page descriptors for any given pmap.
 366  * Reference counts are maintained for L2 descriptors so they can be
 367  * freed when empty.
 368  */
 369 struct l2_dtable {
 370         /* The number of L2 page descriptors allocated to this l2_dtable */
 371         u_int l2_occupancy;
 372
 373         /* List of L2 page descriptors */
 374         struct l2_bucket {
 375                 pt_entry_t *l2b_kva;    /* KVA of L2 Descriptor Table */
 376                 vm_paddr_t l2b_phys;    /* Physical address of same */
 377                 u_short l2b_l1idx;      /* This L2 table's L1 index */
 378                 u_short l2b_occupancy;  /* How many active descriptors */
 379         } l2_bucket[L2_BUCKET_SIZE];
 380 };
 381
 382 /* pmap_kenter_internal flags */
 383 #define KENTER_CACHE    0x1
 384 #define KENTER_DEVICE   0x2
 385 #define KENTER_USER     0x4
 386
 387 /*
 388  * Given an L1 table index, calculate the corresponding l2_dtable index
 389  * and bucket index within the l2_dtable.
 390  */
 391 #define L2_IDX(l1idx)           (((l1idx) >> L2_BUCKET_LOG2) & \
 392                                  (L2_SIZE - 1))
 393 #define L2_BUCKET(l1idx)        ((l1idx) & (L2_BUCKET_SIZE - 1))
 394
 395 /*
 396  * Given a virtual address, this macro returns the
 397  * virtual address required to drop into the next L2 bucket.
 398  */
 399 #define L2_NEXT_BUCKET(va)      (((va) & L1_S_FRAME) + L1_S_SIZE)
 400
 401 /*
 402  * We try to map the page tables write-through, if possible.  However, not
 403  * all CPUs have a write-through cache mode, so on those we have to sync
 404  * the cache when we frob page tables.
 405  *
 406  * We try to evaluate this at compile time, if possible.  However, it's
 407  * not always possible to do that, hence this run-time var.
 408  */
 409 int     pmap_needs_pte_sync;
 410
 411 /*
 412  * Macro to determine if a mapping might be resident in the
 413  * instruction cache and/or TLB
 414  */
 415 #define PTE_BEEN_EXECD(pte)  (L2_S_EXECUTABLE(pte) && L2_S_REFERENCED(pte))
 416
 417 /*
 418  * Macro to determine if a mapping might be resident in the
 419  * data cache and/or TLB
 420  */
 421 #define PTE_BEEN_REFD(pte)   (L2_S_REFERENCED(pte))
 422
 423 #ifndef PMAP_SHPGPERPROC
 424 #define PMAP_SHPGPERPROC 200
 425 #endif
 426
 427 #define pmap_is_current(pm)     ((pm) == pmap_kernel() || \
 428             curproc->p_vmspace->vm_map.pmap == (pm))
 429
 430 /*
 431  * Data for the pv entry allocation mechanism
 432  */
 433 static TAILQ_HEAD(pch, pv_chunk) pv_chunks = TAILQ_HEAD_INITIALIZER(pv_chunks);
 434 static int pv_entry_count, pv_entry_max, pv_entry_high_water;
 435 static struct md_page *pv_table;
 436 static int shpgperproc = PMAP_SHPGPERPROC;
 437
 438 struct pv_chunk *pv_chunkbase;          /* KVA block for pv_chunks */
 439 int pv_maxchunks;                       /* How many chunks we have KVA for */
 440 vm_offset_t pv_vafree;                  /* Freelist stored in the PTE */
 441
 442 static __inline struct pv_chunk *
 443 pv_to_chunk(pv_entry_t pv)
 444 {
 445
 446         return ((struct pv_chunk *)((uintptr_t)pv & ~(uintptr_t)PAGE_MASK));
 447 }
 448
 449 #define PV_PMAP(pv) (pv_to_chunk(pv)->pc_pmap)
 450
 451 CTASSERT(sizeof(struct pv_chunk) == PAGE_SIZE);
 452 CTASSERT(_NPCM == 8);
 453 CTASSERT(_NPCPV == 252);
 454
 455 #define PC_FREE0_6      0xfffffffful    /* Free values for index 0 through 6 */
 456 #define PC_FREE7        0x0ffffffful    /* Free values for index 7 */
 457
 458 static const uint32_t pc_freemask[_NPCM] = {
 459         PC_FREE0_6, PC_FREE0_6, PC_FREE0_6,
 460         PC_FREE0_6, PC_FREE0_6, PC_FREE0_6,
 461         PC_FREE0_6, PC_FREE7
 462 };
 463
 464 static SYSCTL_NODE(_vm, OID_AUTO, pmap, CTLFLAG_RD, 0, "VM/pmap parameters");
 465
 466 /* Superpages utilization enabled = 1 / disabled = 0 */
 467 static int sp_enabled = 1;
 468 SYSCTL_INT(_vm_pmap, OID_AUTO, sp_enabled, CTLFLAG_RDTUN | CTLFLAG_NOFETCH, &sp_enabled, 0,
 469     "Are large page mappings enabled?");
 470
 471 SYSCTL_INT(_vm_pmap, OID_AUTO, pv_entry_count, CTLFLAG_RD, &pv_entry_count, 0,
 472     "Current number of pv entries");
 473
 474 #ifdef PV_STATS
 475 static int pc_chunk_count, pc_chunk_allocs, pc_chunk_frees, pc_chunk_tryfail;
 476
 477 SYSCTL_INT(_vm_pmap, OID_AUTO, pc_chunk_count, CTLFLAG_RD, &pc_chunk_count, 0,
 478     "Current number of pv entry chunks");
 479 SYSCTL_INT(_vm_pmap, OID_AUTO, pc_chunk_allocs, CTLFLAG_RD, &pc_chunk_allocs, 0,
 480     "Current number of pv entry chunks allocated");
 481 SYSCTL_INT(_vm_pmap, OID_AUTO, pc_chunk_frees, CTLFLAG_RD, &pc_chunk_frees, 0,
 482     "Current number of pv entry chunks frees");
 483 SYSCTL_INT(_vm_pmap, OID_AUTO, pc_chunk_tryfail, CTLFLAG_RD, &pc_chunk_tryfail, 0,
 484     "Number of times tried to get a chunk page but failed.");
 485
 486 static long pv_entry_frees, pv_entry_allocs;
 487 static int pv_entry_spare;
 488
 489 SYSCTL_LONG(_vm_pmap, OID_AUTO, pv_entry_frees, CTLFLAG_RD, &pv_entry_frees, 0,
 490     "Current number of pv entry frees");
 491 SYSCTL_LONG(_vm_pmap, OID_AUTO, pv_entry_allocs, CTLFLAG_RD, &pv_entry_allocs, 0,
 492     "Current number of pv entry allocs");
 493 SYSCTL_INT(_vm_pmap, OID_AUTO, pv_entry_spare, CTLFLAG_RD, &pv_entry_spare, 0,
 494     "Current number of spare pv entries");
 495 #endif
 496
 497 uma_zone_t l2zone;
 498 static uma_zone_t l2table_zone;
 499 static vm_offset_t pmap_kernel_l2dtable_kva;
 500 static vm_offset_t pmap_kernel_l2ptp_kva;
 501 static vm_paddr_t pmap_kernel_l2ptp_phys;
 502 static struct rwlock pvh_global_lock;
 503
 504 int l1_mem_types[] = {
 505         ARM_L1S_STRONG_ORD,
 506         ARM_L1S_DEVICE_NOSHARE,
 507         ARM_L1S_DEVICE_SHARE,
 508         ARM_L1S_NRML_NOCACHE,
 509         ARM_L1S_NRML_IWT_OWT,
 510         ARM_L1S_NRML_IWB_OWB,
 511         ARM_L1S_NRML_IWBA_OWBA
 512 };
 513
 514 int l2l_mem_types[] = {
 515         ARM_L2L_STRONG_ORD,
 516         ARM_L2L_DEVICE_NOSHARE,
 517         ARM_L2L_DEVICE_SHARE,
 518         ARM_L2L_NRML_NOCACHE,
 519         ARM_L2L_NRML_IWT_OWT,
 520         ARM_L2L_NRML_IWB_OWB,
 521         ARM_L2L_NRML_IWBA_OWBA
 522 };
 523
 524 int l2s_mem_types[] = {
 525         ARM_L2S_STRONG_ORD,
 526         ARM_L2S_DEVICE_NOSHARE,
 527         ARM_L2S_DEVICE_SHARE,
 528         ARM_L2S_NRML_NOCACHE,
 529         ARM_L2S_NRML_IWT_OWT,
 530         ARM_L2S_NRML_IWB_OWB,
 531         ARM_L2S_NRML_IWBA_OWBA
 532 };
 533
 534 /*
 535  * This list exists for the benefit of pmap_map_chunk().  It keeps track
 536  * of the kernel L2 tables during bootstrap, so that pmap_map_chunk() can
 537  * find them as necessary.
 538  *
 539  * Note that the data on this list MUST remain valid after initarm() returns,
 540  * as pmap_bootstrap() uses it to contruct L2 table metadata.
 541  */
 542 SLIST_HEAD(, pv_addr) kernel_pt_list = SLIST_HEAD_INITIALIZER(kernel_pt_list);
 543
 544 static void
 545 pmap_init_l1(struct l1_ttable *l1, pd_entry_t *l1pt)
 546 {
 547         int i;
 548
 549         l1->l1_kva = l1pt;
 550         l1->l1_domain_use_count = 0;
 551         l1->l1_domain_first = 0;
 552
 553         for (i = 0; i < PMAP_DOMAINS; i++)
 554                 l1->l1_domain_free[i] = i + 1;
 555
 556         /*
 557          * Copy the kernel's L1 entries to each new L1.
 558          */
 559         if (l1pt != pmap_kernel()->pm_l1->l1_kva)
 560                 memcpy(l1pt, pmap_kernel()->pm_l1->l1_kva, L1_TABLE_SIZE);
 561
 562         if ((l1->l1_physaddr = pmap_extract(pmap_kernel(), (vm_offset_t)l1pt)) == 0)
 563                 panic("pmap_init_l1: can't get PA of L1 at %p", l1pt);
 564         SLIST_INSERT_HEAD(&l1_list, l1, l1_link);
 565         TAILQ_INSERT_TAIL(&l1_lru_list, l1, l1_lru);
 566 }
 567
 568 static vm_offset_t
 569 kernel_pt_lookup(vm_paddr_t pa)
 570 {
 571         struct pv_addr *pv;
 572
 573         SLIST_FOREACH(pv, &kernel_pt_list, pv_list) {
 574                 if (pv->pv_pa == pa)
 575                         return (pv->pv_va);
 576         }
 577         return (0);
 578 }
 579
 580 void
 581 pmap_pte_init_mmu_v6(void)
 582 {
 583
 584         if (PTE_PAGETABLE >= 3)
 585                 pmap_needs_pte_sync = 1;
 586         pte_l1_s_cache_mode = l1_mem_types[PTE_CACHE];
 587         pte_l2_l_cache_mode = l2l_mem_types[PTE_CACHE];
 588         pte_l2_s_cache_mode = l2s_mem_types[PTE_CACHE];
 589
 590         pte_l1_s_cache_mode_pt = l1_mem_types[PTE_PAGETABLE];
 591         pte_l2_l_cache_mode_pt = l2l_mem_types[PTE_PAGETABLE];
 592         pte_l2_s_cache_mode_pt = l2s_mem_types[PTE_PAGETABLE];
 593
 594 }
 595
 596 /*
 597  * Allocate an L1 translation table for the specified pmap.
 598  * This is called at pmap creation time.
 599  */
 600 static void
 601 pmap_alloc_l1(pmap_t pmap)
 602 {
 603         struct l1_ttable *l1;
 604         u_int8_t domain;
 605
 606         /*
 607          * Remove the L1 at the head of the LRU list
 608          */
 609         mtx_lock(&l1_lru_lock);
 610         l1 = TAILQ_FIRST(&l1_lru_list);
 611         TAILQ_REMOVE(&l1_lru_list, l1, l1_lru);
 612
 613         /*
 614          * Pick the first available domain number, and update
 615          * the link to the next number.
 616          */
 617         domain = l1->l1_domain_first;
 618         l1->l1_domain_first = l1->l1_domain_free[domain];
 619
 620         /*
 621          * If there are still free domain numbers in this L1,
 622          * put it back on the TAIL of the LRU list.
 623          */
 624         if (++l1->l1_domain_use_count < PMAP_DOMAINS)
 625                 TAILQ_INSERT_TAIL(&l1_lru_list, l1, l1_lru);
 626
 627         mtx_unlock(&l1_lru_lock);
 628
 629         /*
 630          * Fix up the relevant bits in the pmap structure
 631          */
 632         pmap->pm_l1 = l1;
 633         pmap->pm_domain = domain + 1;
 634 }
 635
 636 /*
 637  * Free an L1 translation table.
 638  * This is called at pmap destruction time.
 639  */
 640 static void
 641 pmap_free_l1(pmap_t pmap)
 642 {
 643         struct l1_ttable *l1 = pmap->pm_l1;
 644
 645         mtx_lock(&l1_lru_lock);
 646
 647         /*
 648          * If this L1 is currently on the LRU list, remove it.
 649          */
 650         if (l1->l1_domain_use_count < PMAP_DOMAINS)
 651                 TAILQ_REMOVE(&l1_lru_list, l1, l1_lru);
 652
 653         /*
 654          * Free up the domain number which was allocated to the pmap
 655          */
 656         l1->l1_domain_free[pmap->pm_domain - 1] = l1->l1_domain_first;
 657         l1->l1_domain_first = pmap->pm_domain - 1;
 658         l1->l1_domain_use_count--;
 659
 660         /*
 661          * The L1 now must have at least 1 free domain, so add
 662          * it back to the LRU list. If the use count is zero,
 663          * put it at the head of the list, otherwise it goes
 664          * to the tail.
 665          */
 666         if (l1->l1_domain_use_count == 0) {
 667                 TAILQ_INSERT_HEAD(&l1_lru_list, l1, l1_lru);
 668         }       else
 669                 TAILQ_INSERT_TAIL(&l1_lru_list, l1, l1_lru);
 670
 671         mtx_unlock(&l1_lru_lock);
 672 }
 673
 674 /*
 675  * Returns a pointer to the L2 bucket associated with the specified pmap
 676  * and VA, or NULL if no L2 bucket exists for the address.
 677  */
 678 static PMAP_INLINE struct l2_bucket *
 679 pmap_get_l2_bucket(pmap_t pmap, vm_offset_t va)
 680 {
 681         struct l2_dtable *l2;
 682         struct l2_bucket *l2b;
 683         u_short l1idx;
 684
 685         l1idx = L1_IDX(va);
 686
 687         if ((l2 = pmap->pm_l2[L2_IDX(l1idx)]) == NULL ||
 688             (l2b = &l2->l2_bucket[L2_BUCKET(l1idx)])->l2b_kva == NULL)
 689                 return (NULL);
 690
 691         return (l2b);
 692 }
 693
 694 /*
 695  * Returns a pointer to the L2 bucket associated with the specified pmap
 696  * and VA.
 697  *
 698  * If no L2 bucket exists, perform the necessary allocations to put an L2
 699  * bucket/page table in place.
 700  *
 701  * Note that if a new L2 bucket/page was allocated, the caller *must*
 702  * increment the bucket occupancy counter appropriately *before*
 703  * releasing the pmap's lock to ensure no other thread or cpu deallocates
 704  * the bucket/page in the meantime.
 705  */
 706 static struct l2_bucket *
 707 pmap_alloc_l2_bucket(pmap_t pmap, vm_offset_t va)
 708 {
 709         struct l2_dtable *l2;
 710         struct l2_bucket *l2b;
 711         u_short l1idx;
 712
 713         l1idx = L1_IDX(va);
 714
 715         PMAP_ASSERT_LOCKED(pmap);
 716         rw_assert(&pvh_global_lock, RA_WLOCKED);
 717         if ((l2 = pmap->pm_l2[L2_IDX(l1idx)]) == NULL) {
 718                 /*
 719                  * No mapping at this address, as there is
 720                  * no entry in the L1 table.
 721                  * Need to allocate a new l2_dtable.
 722                  */
 723                 PMAP_UNLOCK(pmap);
 724                 rw_wunlock(&pvh_global_lock);
 725                 if ((l2 = uma_zalloc(l2table_zone, M_NOWAIT)) == NULL) {
 726                         rw_wlock(&pvh_global_lock);
 727                         PMAP_LOCK(pmap);
 728                         return (NULL);
 729                 }
 730                 rw_wlock(&pvh_global_lock);
 731                 PMAP_LOCK(pmap);
 732                 if (pmap->pm_l2[L2_IDX(l1idx)] != NULL) {
 733                         /*
 734                          * Someone already allocated the l2_dtable while
 735                          * we were doing the same.
 736                          */
 737                         uma_zfree(l2table_zone, l2);
 738                         l2 = pmap->pm_l2[L2_IDX(l1idx)];
 739                 } else {
 740                         bzero(l2, sizeof(*l2));
 741                         /*
 742                          * Link it into the parent pmap
 743                          */
 744                         pmap->pm_l2[L2_IDX(l1idx)] = l2;
 745                 }
 746         }
 747
 748         l2b = &l2->l2_bucket[L2_BUCKET(l1idx)];
 749
 750         /*
 751          * Fetch pointer to the L2 page table associated with the address.
 752          */
 753         if (l2b->l2b_kva == NULL) {
 754                 pt_entry_t *ptep;
 755
 756                 /*
 757                  * No L2 page table has been allocated. Chances are, this
 758                  * is because we just allocated the l2_dtable, above.
 759                  */
 760                 PMAP_UNLOCK(pmap);
 761                 rw_wunlock(&pvh_global_lock);
 762                 ptep = uma_zalloc(l2zone, M_NOWAIT);
 763                 rw_wlock(&pvh_global_lock);
 764                 PMAP_LOCK(pmap);
 765                 if (l2b->l2b_kva != 0) {
 766                         /* We lost the race. */
 767                         uma_zfree(l2zone, ptep);
 768                         return (l2b);
 769                 }
 770                 l2b->l2b_phys = vtophys(ptep);
 771                 if (ptep == NULL) {
 772                         /*
 773                          * Oops, no more L2 page tables available at this
 774                          * time. We may need to deallocate the l2_dtable
 775                          * if we allocated a new one above.
 776                          */
 777                         if (l2->l2_occupancy == 0) {
 778                                 pmap->pm_l2[L2_IDX(l1idx)] = NULL;
 779                                 uma_zfree(l2table_zone, l2);
 780                         }
 781                         return (NULL);
 782                 }
 783
 784                 l2->l2_occupancy++;
 785                 l2b->l2b_kva = ptep;
 786                 l2b->l2b_l1idx = l1idx;
 787         }
 788
 789         return (l2b);
 790 }
 791
 792 static PMAP_INLINE void
 793 pmap_free_l2_ptp(pt_entry_t *l2)
 794 {
 795         uma_zfree(l2zone, l2);
 796 }
 797 /*
 798  * One or more mappings in the specified L2 descriptor table have just been
 799  * invalidated.
 800  *
 801  * Garbage collect the metadata and descriptor table itself if necessary.
 802  *
 803  * The pmap lock must be acquired when this is called (not necessary
 804  * for the kernel pmap).
 805  */
 806 static void
 807 pmap_free_l2_bucket(pmap_t pmap, struct l2_bucket *l2b, u_int count)
 808 {
 809         struct l2_dtable *l2;
 810         pd_entry_t *pl1pd, l1pd;
 811         pt_entry_t *ptep;
 812         u_short l1idx;
 813
 814
 815         /*
 816          * Update the bucket's reference count according to how many
 817          * PTEs the caller has just invalidated.
 818          */
 819         l2b->l2b_occupancy -= count;
 820
 821         /*
 822          * Note:
 823          *
 824          * Level 2 page tables allocated to the kernel pmap are never freed
 825          * as that would require checking all Level 1 page tables and
 826          * removing any references to the Level 2 page table. See also the
 827          * comment elsewhere about never freeing bootstrap L2 descriptors.
 828          *
 829          * We make do with just invalidating the mapping in the L2 table.
 830          *
 831          * This isn't really a big deal in practice and, in fact, leads
 832          * to a performance win over time as we don't need to continually
 833          * alloc/free.
 834          */
 835         if (l2b->l2b_occupancy > 0 || pmap == pmap_kernel())
 836                 return;
 837
 838         /*
 839          * There are no more valid mappings in this level 2 page table.
 840          * Go ahead and NULL-out the pointer in the bucket, then
 841          * free the page table.
 842          */
 843         l1idx = l2b->l2b_l1idx;
 844         ptep = l2b->l2b_kva;
 845         l2b->l2b_kva = NULL;
 846
 847         pl1pd = &pmap->pm_l1->l1_kva[l1idx];
 848
 849         /*
 850          * If the L1 slot matches the pmap's domain
 851          * number, then invalidate it.
 852          */
 853         l1pd = *pl1pd & (L1_TYPE_MASK | L1_C_DOM_MASK);
 854         if (l1pd == (L1_C_DOM(pmap->pm_domain) | L1_TYPE_C)) {
 855                 *pl1pd = 0;
 856                 PTE_SYNC(pl1pd);
 857                 cpu_tlb_flushD_SE((vm_offset_t)ptep);
 858                 cpu_cpwait();
 859         }
 860
 861         /*
 862          * Release the L2 descriptor table back to the pool cache.
 863          */
 864         pmap_free_l2_ptp(ptep);
 865
 866         /*
 867          * Update the reference count in the associated l2_dtable
 868          */
 869         l2 = pmap->pm_l2[L2_IDX(l1idx)];
 870         if (--l2->l2_occupancy > 0)
 871                 return;
 872
 873         /*
 874          * There are no more valid mappings in any of the Level 1
 875          * slots managed by this l2_dtable. Go ahead and NULL-out
 876          * the pointer in the parent pmap and free the l2_dtable.
 877          */
 878         pmap->pm_l2[L2_IDX(l1idx)] = NULL;
 879         uma_zfree(l2table_zone, l2);
 880 }
 881
 882 /*
 883  * Pool cache constructors for L2 descriptor tables, metadata and pmap
 884  * structures.
 885  */
 886 static int
 887 pmap_l2ptp_ctor(void *mem, int size, void *arg, int flags)
 888 {
 889         struct l2_bucket *l2b;
 890         pt_entry_t *ptep, pte;
 891         vm_offset_t va = (vm_offset_t)mem & ~PAGE_MASK;
 892
 893         /*
 894          * The mappings for these page tables were initially made using
 895          * pmap_kenter() by the pool subsystem. Therefore, the cache-
 896          * mode will not be right for page table mappings. To avoid
 897          * polluting the pmap_kenter() code with a special case for
 898          * page tables, we simply fix up the cache-mode here if it's not
 899          * correct.
 900          */
 901         l2b = pmap_get_l2_bucket(pmap_kernel(), va);
 902         ptep = &l2b->l2b_kva[l2pte_index(va)];
 903         pte = *ptep;
 904
 905         cpu_idcache_wbinv_range(va, PAGE_SIZE);
 906         pmap_l2cache_wbinv_range(va, pte & L2_S_FRAME, PAGE_SIZE);
 907         if ((pte & L2_S_CACHE_MASK) != pte_l2_s_cache_mode_pt) {
 908                 /*
 909                  * Page tables must have the cache-mode set to
 910                  * Write-Thru.
 911                  */
 912                 *ptep = (pte & ~L2_S_CACHE_MASK) | pte_l2_s_cache_mode_pt;
 913                 PTE_SYNC(ptep);
 914                 cpu_tlb_flushD_SE(va);
 915                 cpu_cpwait();
 916         }
 917
 918         memset(mem, 0, L2_TABLE_SIZE_REAL);
 919         return (0);
 920 }
 921
 922 /*
 923  * Modify pte bits for all ptes corresponding to the given physical address.
 924  * We use `maskbits' rather than `clearbits' because we're always passing
 925  * constants and the latter would require an extra inversion at run-time.
 926  */
 927 static int
 928 pmap_clearbit(struct vm_page *m, u_int maskbits)
 929 {
 930         struct l2_bucket *l2b;
 931         struct pv_entry *pv, *pve, *next_pv;
 932         struct md_page *pvh;
 933         pd_entry_t *pl1pd;
 934         pt_entry_t *ptep, npte, opte;
 935         pmap_t pmap;
 936         vm_offset_t va;
 937         u_int oflags;
 938         int count = 0;
 939
 940         rw_wlock(&pvh_global_lock);
 941         if ((m->flags & PG_FICTITIOUS) != 0)
 942                 goto small_mappings;
 943
 944         pvh = pa_to_pvh(VM_PAGE_TO_PHYS(m));
 945         TAILQ_FOREACH_SAFE(pv, &pvh->pv_list, pv_list, next_pv) {
 946                 va = pv->pv_va;
 947                 pmap = PV_PMAP(pv);
 948                 PMAP_LOCK(pmap);
 949                 pl1pd = &pmap->pm_l1->l1_kva[L1_IDX(va)];
 950                 KASSERT((*pl1pd & L1_TYPE_MASK) == L1_S_PROTO,
 951                     ("pmap_clearbit: valid section mapping expected"));
 952                 if ((maskbits & PVF_WRITE) && (pv->pv_flags & PVF_WRITE))
 953                         (void)pmap_demote_section(pmap, va);
 954                 else if ((maskbits & PVF_REF) && L1_S_REFERENCED(*pl1pd)) {
 955                         if (pmap_demote_section(pmap, va)) {
 956                                 if ((pv->pv_flags & PVF_WIRED) == 0) {
 957                                         /*
 958                                          * Remove the mapping to a single page
 959                                          * so that a subsequent access may
 960                                          * repromote. Since the underlying
 961                                          * l2_bucket is fully populated, this
 962                                          * removal never frees an entire
 963                                          * l2_bucket.
 964                                          */
 965                                         va += (VM_PAGE_TO_PHYS(m) &
 966                                             L1_S_OFFSET);
 967                                         l2b = pmap_get_l2_bucket(pmap, va);
 968                                         KASSERT(l2b != NULL,
 969                                             ("pmap_clearbit: no l2 bucket for "
 970                                              "va 0x%#x, pmap 0x%p", va, pmap));
 971                                         ptep = &l2b->l2b_kva[l2pte_index(va)];
 972                                         *ptep = 0;
 973                                         PTE_SYNC(ptep);
 974                                         pmap_free_l2_bucket(pmap, l2b, 1);
 975                                         pve = pmap_remove_pv(m, pmap, va);
 976                                         KASSERT(pve != NULL, ("pmap_clearbit: "
 977                                             "no PV entry for managed mapping"));
 978                                         pmap_free_pv_entry(pmap, pve);
 979
 980                                 }
 981                         }
 982                 } else if ((maskbits & PVF_MOD) && L1_S_WRITABLE(*pl1pd)) {
 983                         if (pmap_demote_section(pmap, va)) {
 984                                 if ((pv->pv_flags & PVF_WIRED) == 0) {
 985                                         /*
 986                                          * Write protect the mapping to a
 987                                          * single page so that a subsequent
 988                                          * write access may repromote.
 989                                          */
 990                                         va += (VM_PAGE_TO_PHYS(m) &
 991                                             L1_S_OFFSET);
 992                                         l2b = pmap_get_l2_bucket(pmap, va);
 993                                         KASSERT(l2b != NULL,
 994                                             ("pmap_clearbit: no l2 bucket for "
 995                                              "va 0x%#x, pmap 0x%p", va, pmap));
 996                                         ptep = &l2b->l2b_kva[l2pte_index(va)];
 997                                         if ((*ptep & L2_S_PROTO) != 0) {
 998                                                 pve = pmap_find_pv(&m->md,
 999                                                     pmap, va);
1000                                                 KASSERT(pve != NULL,
1001                                                     ("pmap_clearbit: no PV "
1002                                                     "entry for managed mapping"));
1003                                                 pve->pv_flags &= ~PVF_WRITE;
1004                                                 *ptep |= L2_APX;
1005                                                 PTE_SYNC(ptep);
1006                                         }
1007                                 }
1008                         }
1009                 }
1010                 PMAP_UNLOCK(pmap);
1011         }
1012
1013 small_mappings:
1014         if (TAILQ_EMPTY(&m->md.pv_list)) {
1015                 rw_wunlock(&pvh_global_lock);
1016                 return (0);
1017         }
1018
1019         /*
1020          * Loop over all current mappings setting/clearing as appropos
1021          */
1022         TAILQ_FOREACH(pv, &m->md.pv_list, pv_list) {
1023                 va = pv->pv_va;
1024                 pmap = PV_PMAP(pv);
1025                 oflags = pv->pv_flags;
1026                 pv->pv_flags &= ~maskbits;
1027
1028                 PMAP_LOCK(pmap);
1029
1030                 l2b = pmap_get_l2_bucket(pmap, va);
1031                 KASSERT(l2b != NULL, ("pmap_clearbit: no l2 bucket for "
1032                     "va 0x%#x, pmap 0x%p", va, pmap));
1033
1034                 ptep = &l2b->l2b_kva[l2pte_index(va)];
1035                 npte = opte = *ptep;
1036
1037                 if (maskbits & (PVF_WRITE | PVF_MOD)) {
1038                         /* make the pte read only */
1039                         npte |= L2_APX;
1040                 }
1041
1042                 if (maskbits & PVF_REF) {
1043                         /*
1044                          * Clear referenced flag in PTE so that we
1045                          * will take a flag fault the next time the mapping
1046                          * is referenced.
1047                          */
1048                         npte &= ~L2_S_REF;
1049                 }
1050
1051                 CTR4(KTR_PMAP,"clearbit: pmap:%p bits:%x pte:%x->%x",
1052                     pmap, maskbits, opte, npte);
1053                 if (npte != opte) {
1054                         count++;
1055                         *ptep = npte;
1056                         PTE_SYNC(ptep);
1057                         /* Flush the TLB entry if a current pmap. */
1058                         if (PTE_BEEN_EXECD(opte))
1059                                 cpu_tlb_flushID_SE(pv->pv_va);
1060                         else if (PTE_BEEN_REFD(opte))
1061                                 cpu_tlb_flushD_SE(pv->pv_va);
1062                         cpu_cpwait();
1063                 }
1064
1065                 PMAP_UNLOCK(pmap);
1066
1067         }
1068
1069         if (maskbits & PVF_WRITE)
1070                 vm_page_aflag_clear(m, PGA_WRITEABLE);
1071         rw_wunlock(&pvh_global_lock);
1072         return (count);
1073 }
1074
1075 /*
1076  * main pv_entry manipulation functions:
1077  *   pmap_enter_pv: enter a mapping onto a vm_page list
1078  *   pmap_remove_pv: remove a mappiing from a vm_page list
1079  *
1080  * NOTE: pmap_enter_pv expects to lock the pvh itself
1081  *       pmap_remove_pv expects the caller to lock the pvh before calling
1082  */
1083
1084 /*
1085  * pmap_enter_pv: enter a mapping onto a vm_page's PV list
1086  *
1087  * => caller should hold the proper lock on pvh_global_lock
1088  * => caller should have pmap locked
1089  * => we will (someday) gain the lock on the vm_page's PV list
1090  * => caller should adjust ptp's wire_count before calling
1091  * => caller should not adjust pmap's wire_count
1092  */
1093 static void
1094 pmap_enter_pv(struct vm_page *m, struct pv_entry *pve, pmap_t pmap,
1095     vm_offset_t va, u_int flags)
1096 {
1097
1098         rw_assert(&pvh_global_lock, RA_WLOCKED);
1099
1100         PMAP_ASSERT_LOCKED(pmap);
1101         pve->pv_va = va;
1102         pve->pv_flags = flags;
1103
1104         TAILQ_INSERT_HEAD(&m->md.pv_list, pve, pv_list);
1105         if (pve->pv_flags & PVF_WIRED)
1106                 ++pmap->pm_stats.wired_count;
1107 }
1108
1109 /*
1110  *
1111  * pmap_find_pv: Find a pv entry
1112  *
1113  * => caller should hold lock on vm_page
1114  */
1115 static PMAP_INLINE struct pv_entry *
1116 pmap_find_pv(struct md_page *md, pmap_t pmap, vm_offset_t va)
1117 {
1118         struct pv_entry *pv;
1119
1120         rw_assert(&pvh_global_lock, RA_WLOCKED);
1121         TAILQ_FOREACH(pv, &md->pv_list, pv_list)
1122                 if (pmap == PV_PMAP(pv) && va == pv->pv_va)
1123                         break;
1124
1125         return (pv);
1126 }
1127
1128 /*
1129  * vector_page_setprot:
1130  *
1131  *      Manipulate the protection of the vector page.
1132  */
1133 void
1134 vector_page_setprot(int prot)
1135 {
1136         struct l2_bucket *l2b;
1137         pt_entry_t *ptep;
1138
1139         l2b = pmap_get_l2_bucket(pmap_kernel(), vector_page);
1140
1141         ptep = &l2b->l2b_kva[l2pte_index(vector_page)];
1142         /*
1143          * Set referenced flag.
1144          * Vectors' page is always desired
1145          * to be allowed to reside in TLB.
1146          */
1147         *ptep |= L2_S_REF;
1148
1149         pmap_set_prot(ptep, prot|VM_PROT_EXECUTE, 0);
1150         PTE_SYNC(ptep);
1151         cpu_tlb_flushID_SE(vector_page);
1152         cpu_cpwait();
1153 }
1154
1155 static void
1156 pmap_set_prot(pt_entry_t *ptep, vm_prot_t prot, uint8_t user)
1157 {
1158
1159         *ptep &= ~(L2_S_PROT_MASK | L2_XN);
1160
1161         if (!(prot & VM_PROT_EXECUTE))
1162                 *ptep |= L2_XN;
1163
1164         /* Set defaults first - kernel read access */
1165         *ptep |= L2_APX;
1166         *ptep |= L2_S_PROT_R;
1167         /* Now tune APs as desired */
1168         if (user)
1169                 *ptep |= L2_S_PROT_U;
1170
1171         if (prot & VM_PROT_WRITE)
1172                 *ptep &= ~(L2_APX);
1173 }
1174
1175 /*
1176  * pmap_remove_pv: try to remove a mapping from a pv_list
1177  *
1178  * => caller should hold proper lock on pmap_main_lock
1179  * => pmap should be locked
1180  * => caller should hold lock on vm_page [so that attrs can be adjusted]
1181  * => caller should adjust ptp's wire_count and free PTP if needed
1182  * => caller should NOT adjust pmap's wire_count
1183  * => we return the removed pve
1184  */
1185 static struct pv_entry *
1186 pmap_remove_pv(struct vm_page *m, pmap_t pmap, vm_offset_t va)
1187 {
1188         struct pv_entry *pve;
1189
1190         rw_assert(&pvh_global_lock, RA_WLOCKED);
1191         PMAP_ASSERT_LOCKED(pmap);
1192
1193         pve = pmap_find_pv(&m->md, pmap, va);   /* find corresponding pve */
1194         if (pve != NULL) {
1195                 TAILQ_REMOVE(&m->md.pv_list, pve, pv_list);
1196                 if (pve->pv_flags & PVF_WIRED)
1197                         --pmap->pm_stats.wired_count;
1198         }
1199         if (TAILQ_EMPTY(&m->md.pv_list))
1200                 vm_page_aflag_clear(m, PGA_WRITEABLE);
1201
1202         return(pve);                            /* return removed pve */
1203 }
1204
1205 /*
1206  *
1207  * pmap_modify_pv: Update pv flags
1208  *
1209  * => caller should hold lock on vm_page [so that attrs can be adjusted]
1210  * => caller should NOT adjust pmap's wire_count
1211  * => we return the old flags
1212  *
1213  * Modify a physical-virtual mapping in the pv table
1214  */
1215 static u_int
1216 pmap_modify_pv(struct vm_page *m, pmap_t pmap, vm_offset_t va,
1217     u_int clr_mask, u_int set_mask)
1218 {
1219         struct pv_entry *npv;
1220         u_int flags, oflags;
1221
1222         PMAP_ASSERT_LOCKED(pmap);
1223         rw_assert(&pvh_global_lock, RA_WLOCKED);
1224         if ((npv = pmap_find_pv(&m->md, pmap, va)) == NULL)
1225                 return (0);
1226
1227         /*
1228          * There is at least one VA mapping this page.
1229          */
1230         oflags = npv->pv_flags;
1231         npv->pv_flags = flags = (oflags & ~clr_mask) | set_mask;
1232
1233         if ((flags ^ oflags) & PVF_WIRED) {
1234                 if (flags & PVF_WIRED)
1235                         ++pmap->pm_stats.wired_count;
1236                 else
1237                         --pmap->pm_stats.wired_count;
1238         }
1239
1240         return (oflags);
1241 }
1242
1243 /* Function to set the debug level of the pmap code */
1244 #ifdef PMAP_DEBUG
1245 void
1246 pmap_debug(int level)
1247 {
1248         pmap_debug_level = level;
1249         dprintf("pmap_debug: level=%d\n", pmap_debug_level);
1250 }
1251 #endif  /* PMAP_DEBUG */
1252
1253 void
1254 pmap_pinit0(struct pmap *pmap)
1255 {
1256         PDEBUG(1, printf("pmap_pinit0: pmap = %08x\n", (u_int32_t) pmap));
1257
1258         bcopy(kernel_pmap, pmap, sizeof(*pmap));
1259         bzero(&pmap->pm_mtx, sizeof(pmap->pm_mtx));
1260         PMAP_LOCK_INIT(pmap);
1261         TAILQ_INIT(&pmap->pm_pvchunk);
1262 }
1263
1264 /*
1265  *      Initialize a vm_page's machine-dependent fields.
1266  */
1267 void
1268 pmap_page_init(vm_page_t m)
1269 {
1270
1271         TAILQ_INIT(&m->md.pv_list);
1272         m->md.pv_memattr = VM_MEMATTR_DEFAULT;
1273 }
1274
1275 static vm_offset_t
1276 pmap_ptelist_alloc(vm_offset_t *head)
1277 {
1278         pt_entry_t *pte;
1279         vm_offset_t va;
1280
1281         va = *head;
1282         if (va == 0)
1283                 return (va);    /* Out of memory */
1284         pte = vtopte(va);
1285         *head = *pte;
1286         if ((*head & L2_TYPE_MASK) != L2_TYPE_INV)
1287                 panic("%s: va is not L2_TYPE_INV!", __func__);
1288         *pte = 0;
1289         return (va);
1290 }
1291
1292 static void
1293 pmap_ptelist_free(vm_offset_t *head, vm_offset_t va)
1294 {
1295         pt_entry_t *pte;
1296
1297         if ((va & L2_TYPE_MASK) != L2_TYPE_INV)
1298                 panic("%s: freeing va that is not L2_TYPE INV!", __func__);
1299         pte = vtopte(va);
1300         *pte = *head;           /* virtual! L2_TYPE is L2_TYPE_INV though */
1301         *head = va;
1302 }
1303
1304 static void
1305 pmap_ptelist_init(vm_offset_t *head, void *base, int npages)
1306 {
1307         int i;
1308         vm_offset_t va;
1309
1310         *head = 0;
1311         for (i = npages - 1; i >= 0; i--) {
1312                 va = (vm_offset_t)base + i * PAGE_SIZE;
1313                 pmap_ptelist_free(head, va);
1314         }
1315 }
1316
1317 /*
1318  *      Initialize the pmap module.
1319  *      Called by vm_init, to initialize any structures that the pmap
1320  *      system needs to map virtual memory.
1321  */
1322 void
1323 pmap_init(void)
1324 {
1325         vm_size_t s;
1326         int i, pv_npg;
1327
1328         l2zone = uma_zcreate("L2 Table", L2_TABLE_SIZE_REAL, pmap_l2ptp_ctor,
1329             NULL, NULL, NULL, UMA_ALIGN_PTR, UMA_ZONE_VM | UMA_ZONE_NOFREE);
1330         l2table_zone = uma_zcreate("L2 Table", sizeof(struct l2_dtable), NULL,
1331             NULL, NULL, NULL, UMA_ALIGN_PTR, UMA_ZONE_VM | UMA_ZONE_NOFREE);
1332
1333         /*
1334          * Are large page mappings supported and enabled?
1335          */
1336         TUNABLE_INT_FETCH("vm.pmap.sp_enabled", &sp_enabled);
1337         if (sp_enabled) {
1338                 KASSERT(MAXPAGESIZES > 1 && pagesizes[1] == 0,
1339                     ("pmap_init: can't assign to pagesizes[1]"));
1340                 pagesizes[1] = NBPDR;
1341         }
1342
1343         /*
1344          * Calculate the size of the pv head table for superpages.
1345          */
1346         for (i = 0; phys_avail[i + 1]; i += 2);
1347         pv_npg = round_1mpage(phys_avail[(i - 2) + 1]) / NBPDR;
1348
1349         /*
1350          * Allocate memory for the pv head table for superpages.
1351          */
1352         s = (vm_size_t)(pv_npg * sizeof(struct md_page));
1353         s = round_page(s);
1354         pv_table = (struct md_page *)kmem_malloc(kernel_arena, s,
1355             M_WAITOK | M_ZERO);
1356         for (i = 0; i < pv_npg; i++)
1357                 TAILQ_INIT(&pv_table[i].pv_list);
1358
1359         /*
1360          * Initialize the address space for the pv chunks.
1361          */
1362
1363         TUNABLE_INT_FETCH("vm.pmap.shpgperproc", &shpgperproc);
1364         pv_entry_max = shpgperproc * maxproc + vm_cnt.v_page_count;
1365         TUNABLE_INT_FETCH("vm.pmap.pv_entries", &pv_entry_max);
1366         pv_entry_max = roundup(pv_entry_max, _NPCPV);
1367         pv_entry_high_water = 9 * (pv_entry_max / 10);
1368
1369         pv_maxchunks = MAX(pv_entry_max / _NPCPV, maxproc);
1370         pv_chunkbase = (struct pv_chunk *)kva_alloc(PAGE_SIZE * pv_maxchunks);
1371
1372         if (pv_chunkbase == NULL)
1373                 panic("pmap_init: not enough kvm for pv chunks");
1374
1375         pmap_ptelist_init(&pv_vafree, pv_chunkbase, pv_maxchunks);
1376
1377         /*
1378          * Now it is safe to enable pv_table recording.
1379          */
1380         PDEBUG(1, printf("pmap_init: done!\n"));
1381 }
1382
1383 SYSCTL_INT(_vm_pmap, OID_AUTO, pv_entry_max, CTLFLAG_RD, &pv_entry_max, 0,
1384         "Max number of PV entries");
1385 SYSCTL_INT(_vm_pmap, OID_AUTO, shpgperproc, CTLFLAG_RD, &shpgperproc, 0,
1386         "Page share factor per proc");
1387
1388 static SYSCTL_NODE(_vm_pmap, OID_AUTO, section, CTLFLAG_RD, 0,
1389     "1MB page mapping counters");
1390
1391 static u_long pmap_section_demotions;
1392 SYSCTL_ULONG(_vm_pmap_section, OID_AUTO, demotions, CTLFLAG_RD,
1393     &pmap_section_demotions, 0, "1MB page demotions");
1394
1395 static u_long pmap_section_mappings;
1396 SYSCTL_ULONG(_vm_pmap_section, OID_AUTO, mappings, CTLFLAG_RD,
1397     &pmap_section_mappings, 0, "1MB page mappings");
1398
1399 static u_long pmap_section_p_failures;
1400 SYSCTL_ULONG(_vm_pmap_section, OID_AUTO, p_failures, CTLFLAG_RD,
1401     &pmap_section_p_failures, 0, "1MB page promotion failures");
1402
1403 static u_long pmap_section_promotions;
1404 SYSCTL_ULONG(_vm_pmap_section, OID_AUTO, promotions, CTLFLAG_RD,
1405     &pmap_section_promotions, 0, "1MB page promotions");
1406
1407 int
1408 pmap_fault_fixup(pmap_t pmap, vm_offset_t va, vm_prot_t ftype, int user)
1409 {
1410         struct l2_dtable *l2;
1411         struct l2_bucket *l2b;
1412         pd_entry_t *pl1pd, l1pd;
1413         pt_entry_t *ptep, pte;
1414         vm_paddr_t pa;
1415         u_int l1idx;
1416         int rv = 0;
1417
1418         l1idx = L1_IDX(va);
1419         rw_wlock(&pvh_global_lock);
1420         PMAP_LOCK(pmap);
1421         /*
1422          * Check and possibly fix-up L1 section mapping
1423          * only when superpage mappings are enabled to speed up.
1424          */
1425         if (sp_enabled) {
1426                 pl1pd = &pmap->pm_l1->l1_kva[l1idx];
1427                 l1pd = *pl1pd;
1428                 if ((l1pd & L1_TYPE_MASK) == L1_S_PROTO) {
1429                         /* Catch an access to the vectors section */
1430                         if (l1idx == L1_IDX(vector_page))
1431                                 goto out;
1432                         /*
1433                          * Stay away from the kernel mappings.
1434                          * None of them should fault from L1 entry.
1435                          */
1436                         if (pmap == pmap_kernel())
1437                                 goto out;
1438                         /*
1439                          * Catch a forbidden userland access
1440                          */
1441                         if (user && !(l1pd & L1_S_PROT_U))
1442                                 goto out;
1443                         /*
1444                          * Superpage is always either mapped read only
1445                          * or it is modified and permitted to be written
1446                          * by default. Therefore, process only reference
1447                          * flag fault and demote page in case of write fault.
1448                          */
1449                         if ((ftype & VM_PROT_WRITE) && !L1_S_WRITABLE(l1pd) &&
1450                             L1_S_REFERENCED(l1pd)) {
1451                                 (void)pmap_demote_section(pmap, va);
1452                                 goto out;
1453                         } else if (!L1_S_REFERENCED(l1pd)) {
1454                                 /* Mark the page "referenced" */
1455                                 *pl1pd = l1pd | L1_S_REF;
1456                                 PTE_SYNC(pl1pd);
1457                                 goto l1_section_out;
1458                         } else
1459                                 goto out;
1460                 }
1461         }
1462         /*
1463          * If there is no l2_dtable for this address, then the process
1464          * has no business accessing it.
1465          *
1466          * Note: This will catch userland processes trying to access
1467          * kernel addresses.
1468          */
1469         l2 = pmap->pm_l2[L2_IDX(l1idx)];
1470         if (l2 == NULL)
1471                 goto out;
1472
1473         /*
1474          * Likewise if there is no L2 descriptor table
1475          */
1476         l2b = &l2->l2_bucket[L2_BUCKET(l1idx)];
1477         if (l2b->l2b_kva == NULL)
1478                 goto out;
1479
1480         /*
1481          * Check the PTE itself.
1482          */
1483         ptep = &l2b->l2b_kva[l2pte_index(va)];
1484         pte = *ptep;
1485         if (pte == 0)
1486                 goto out;
1487
1488         /*
1489          * Catch a userland access to the vector page mapped at 0x0
1490          */
1491         if (user && !(pte & L2_S_PROT_U))
1492                 goto out;
1493         if (va == vector_page)
1494                 goto out;
1495
1496         pa = l2pte_pa(pte);
1497         CTR5(KTR_PMAP, "pmap_fault_fix: pmap:%p va:%x pte:0x%x ftype:%x user:%x",
1498             pmap, va, pte, ftype, user);
1499         if ((ftype & VM_PROT_WRITE) && !(L2_S_WRITABLE(pte)) &&
1500             L2_S_REFERENCED(pte)) {
1501                 /*
1502                  * This looks like a good candidate for "page modified"
1503                  * emulation...
1504                  */
1505                 struct pv_entry *pv;
1506                 struct vm_page *m;
1507
1508                 /* Extract the physical address of the page */
1509                 if ((m = PHYS_TO_VM_PAGE(pa)) == NULL) {
1510                         goto out;
1511                 }
1512                 /* Get the current flags for this page. */
1513
1514                 pv = pmap_find_pv(&m->md, pmap, va);
1515                 if (pv == NULL) {
1516                         goto out;
1517                 }
1518
1519                 /*
1520                  * Do the flags say this page is writable? If not then it
1521                  * is a genuine write fault. If yes then the write fault is
1522                  * our fault as we did not reflect the write access in the
1523                  * PTE. Now we know a write has occurred we can correct this
1524                  * and also set the modified bit
1525                  */
1526                 if ((pv->pv_flags & PVF_WRITE) == 0) {
1527                         goto out;
1528                 }
1529
1530                 vm_page_dirty(m);
1531
1532                 /* Re-enable write permissions for the page */
1533                 *ptep = (pte & ~L2_APX);
1534                 PTE_SYNC(ptep);
1535                 rv = 1;
1536                 CTR1(KTR_PMAP, "pmap_fault_fix: new pte:0x%x", *ptep);
1537         } else if (!L2_S_REFERENCED(pte)) {
1538                 /*
1539                  * This looks like a good candidate for "page referenced"
1540                  * emulation.
1541                  */
1542                 struct pv_entry *pv;
1543                 struct vm_page *m;
1544
1545                 /* Extract the physical address of the page */
1546                 if ((m = PHYS_TO_VM_PAGE(pa)) == NULL)
1547                         goto out;
1548                 /* Get the current flags for this page. */
1549                 pv = pmap_find_pv(&m->md, pmap, va);
1550                 if (pv == NULL)
1551                         goto out;
1552
1553                 vm_page_aflag_set(m, PGA_REFERENCED);
1554
1555                 /* Mark the page "referenced" */
1556                 *ptep = pte | L2_S_REF;
1557                 PTE_SYNC(ptep);
1558                 rv = 1;
1559                 CTR1(KTR_PMAP, "pmap_fault_fix: new pte:0x%x", *ptep);
1560         }
1561
1562         /*
1563          * We know there is a valid mapping here, so simply
1564          * fix up the L1 if necessary.
1565          */
1566         pl1pd = &pmap->pm_l1->l1_kva[l1idx];
1567         l1pd = l2b->l2b_phys | L1_C_DOM(pmap->pm_domain) | L1_C_PROTO;
1568         if (*pl1pd != l1pd) {
1569                 *pl1pd = l1pd;
1570                 PTE_SYNC(pl1pd);
1571                 rv = 1;
1572         }
1573
1574 #ifdef DEBUG
1575         /*
1576          * If 'rv == 0' at this point, it generally indicates that there is a
1577          * stale TLB entry for the faulting address. This happens when two or
1578          * more processes are sharing an L1. Since we don't flush the TLB on
1579          * a context switch between such processes, we can take domain faults
1580          * for mappings which exist at the same VA in both processes. EVEN IF
1581          * WE'VE RECENTLY FIXED UP THE CORRESPONDING L1 in pmap_enter(), for
1582          * example.
1583          *
1584          * This is extremely likely to happen if pmap_enter() updated the L1
1585          * entry for a recently entered mapping. In this case, the TLB is
1586          * flushed for the new mapping, but there may still be TLB entries for
1587          * other mappings belonging to other processes in the 1MB range
1588          * covered by the L1 entry.
1589          *
1590          * Since 'rv == 0', we know that the L1 already contains the correct
1591          * value, so the fault must be due to a stale TLB entry.
1592          *
1593          * Since we always need to flush the TLB anyway in the case where we
1594          * fixed up the L1, or frobbed the L2 PTE, we effectively deal with
1595          * stale TLB entries dynamically.
1596          *
1597          * However, the above condition can ONLY happen if the current L1 is
1598          * being shared. If it happens when the L1 is unshared, it indicates
1599          * that other parts of the pmap are not doing their job WRT managing
1600          * the TLB.
1601          */
1602         if (rv == 0 && pmap->pm_l1->l1_domain_use_count == 1) {
1603                 printf("fixup: pmap %p, va 0x%08x, ftype %d - nothing to do!\n",
1604                     pmap, va, ftype);
1605                 printf("fixup: l2 %p, l2b %p, ptep %p, pl1pd %p\n",
1606                     l2, l2b, ptep, pl1pd);
1607                 printf("fixup: pte 0x%x, l1pd 0x%x, last code 0x%x\n",
1608                     pte, l1pd, last_fault_code);
1609 #ifdef DDB
1610                 Debugger();
1611 #endif
1612         }
1613 #endif
1614
1615 l1_section_out:
1616         cpu_tlb_flushID_SE(va);
1617         cpu_cpwait();
1618
1619         rv = 1;
1620
1621 out:
1622         rw_wunlock(&pvh_global_lock);
1623         PMAP_UNLOCK(pmap);
1624         return (rv);
1625 }
1626
1627 void
1628 pmap_postinit(void)
1629 {
1630         struct l2_bucket *l2b;
1631         struct l1_ttable *l1;
1632         pd_entry_t *pl1pt;
1633         pt_entry_t *ptep, pte;
1634         vm_offset_t va, eva;
1635         u_int loop, needed;
1636
1637         needed = (maxproc / PMAP_DOMAINS) + ((maxproc % PMAP_DOMAINS) ? 1 : 0);
1638         needed -= 1;
1639         l1 = malloc(sizeof(*l1) * needed, M_VMPMAP, M_WAITOK);
1640
1641         for (loop = 0; loop < needed; loop++, l1++) {
1642                 /* Allocate a L1 page table */
1643                 va = (vm_offset_t)contigmalloc(L1_TABLE_SIZE, M_VMPMAP, 0, 0x0,
1644                     0xffffffff, L1_TABLE_SIZE, 0);
1645
1646                 if (va == 0)
1647                         panic("Cannot allocate L1 KVM");
1648
1649                 eva = va + L1_TABLE_SIZE;
1650                 pl1pt = (pd_entry_t *)va;
1651
1652                 while (va < eva) {
1653                                 l2b = pmap_get_l2_bucket(pmap_kernel(), va);
1654                                 ptep = &l2b->l2b_kva[l2pte_index(va)];
1655                                 pte = *ptep;
1656                                 pte = (pte & ~L2_S_CACHE_MASK) | pte_l2_s_cache_mode_pt;
1657                                 *ptep = pte;
1658                                 PTE_SYNC(ptep);
1659                                 cpu_tlb_flushID_SE(va);
1660                                 cpu_cpwait();
1661                                 va += PAGE_SIZE;
1662                 }
1663                 pmap_init_l1(l1, pl1pt);
1664         }
1665 #ifdef DEBUG
1666         printf("pmap_postinit: Allocated %d static L1 descriptor tables\n",
1667             needed);
1668 #endif
1669 }
1670
1671 /*
1672  * This is used to stuff certain critical values into the PCB where they
1673  * can be accessed quickly from cpu_switch() et al.
1674  */
1675 void
1676 pmap_set_pcb_pagedir(pmap_t pmap, struct pcb *pcb)
1677 {
1678         struct l2_bucket *l2b;
1679
1680         pcb->pcb_pagedir = pmap->pm_l1->l1_physaddr;
1681         pcb->pcb_dacr = (DOMAIN_CLIENT << (PMAP_DOMAIN_KERNEL * 2)) |
1682             (DOMAIN_CLIENT << (pmap->pm_domain * 2));
1683
1684         if (vector_page < KERNBASE) {
1685                 pcb->pcb_pl1vec = &pmap->pm_l1->l1_kva[L1_IDX(vector_page)];
1686                 l2b = pmap_get_l2_bucket(pmap, vector_page);
1687                 pcb->pcb_l1vec = l2b->l2b_phys | L1_C_PROTO |
1688                     L1_C_DOM(pmap->pm_domain) | L1_C_DOM(PMAP_DOMAIN_KERNEL);
1689         } else
1690                 pcb->pcb_pl1vec = NULL;
1691 }
1692
1693 void
1694 pmap_activate(struct thread *td)
1695 {
1696         pmap_t pmap;
1697         struct pcb *pcb;
1698
1699         pmap = vmspace_pmap(td->td_proc->p_vmspace);
1700         pcb = td->td_pcb;
1701
1702         critical_enter();
1703         pmap_set_pcb_pagedir(pmap, pcb);
1704
1705         if (td == curthread) {
1706                 u_int cur_dacr, cur_ttb;
1707
1708                 __asm __volatile("mrc p15, 0, %0, c2, c0, 0" : "=r"(cur_ttb));
1709                 __asm __volatile("mrc p15, 0, %0, c3, c0, 0" : "=r"(cur_dacr));
1710
1711                 cur_ttb &= ~(L1_TABLE_SIZE - 1);
1712
1713                 if (cur_ttb == (u_int)pcb->pcb_pagedir &&
1714                     cur_dacr == pcb->pcb_dacr) {
1715                         /*
1716                          * No need to switch address spaces.
1717                          */
1718                         critical_exit();
1719                         return;
1720                 }
1721
1722
1723                 /*
1724                  * We MUST, I repeat, MUST fix up the L1 entry corresponding
1725                  * to 'vector_page' in the incoming L1 table before switching
1726                  * to it otherwise subsequent interrupts/exceptions (including
1727                  * domain faults!) will jump into hyperspace.
1728                  */
1729                 if (pcb->pcb_pl1vec) {
1730                         *pcb->pcb_pl1vec = pcb->pcb_l1vec;
1731                 }
1732
1733                 cpu_domains(pcb->pcb_dacr);
1734                 cpu_setttb(pcb->pcb_pagedir);
1735         }
1736         critical_exit();
1737 }
1738
1739 static int
1740 pmap_set_pt_cache_mode(pd_entry_t *kl1, vm_offset_t va)
1741 {
1742         pd_entry_t *pdep, pde;
1743         pt_entry_t *ptep, pte;
1744         vm_offset_t pa;
1745         int rv = 0;
1746
1747         /*
1748          * Make sure the descriptor itself has the correct cache mode
1749          */
1750         pdep = &kl1[L1_IDX(va)];
1751         pde = *pdep;
1752
1753         if (l1pte_section_p(pde)) {
1754                 if ((pde & L1_S_CACHE_MASK) != pte_l1_s_cache_mode_pt) {
1755                         *pdep = (pde & ~L1_S_CACHE_MASK) |
1756                             pte_l1_s_cache_mode_pt;
1757                         PTE_SYNC(pdep);
1758                         rv = 1;
1759                 }
1760         } else {
1761                 pa = (vm_paddr_t)(pde & L1_C_ADDR_MASK);
1762                 ptep = (pt_entry_t *)kernel_pt_lookup(pa);
1763                 if (ptep == NULL)
1764                         panic("pmap_bootstrap: No L2 for L2 @ va %p\n", ptep);
1765
1766                 ptep = &ptep[l2pte_index(va)];
1767                 pte = *ptep;
1768                 if ((pte & L2_S_CACHE_MASK) != pte_l2_s_cache_mode_pt) {
1769                         *ptep = (pte & ~L2_S_CACHE_MASK) |
1770                             pte_l2_s_cache_mode_pt;
1771                         PTE_SYNC(ptep);
1772                         rv = 1;
1773                 }
1774         }
1775
1776         return (rv);
1777 }
1778
1779 static void
1780 pmap_alloc_specials(vm_offset_t *availp, int pages, vm_offset_t *vap,
1781     pt_entry_t **ptep)
1782 {
1783         vm_offset_t va = *availp;
1784         struct l2_bucket *l2b;
1785
1786         if (ptep) {
1787                 l2b = pmap_get_l2_bucket(pmap_kernel(), va);
1788                 if (l2b == NULL)
1789                         panic("pmap_alloc_specials: no l2b for 0x%x", va);
1790
1791                 *ptep = &l2b->l2b_kva[l2pte_index(va)];
1792         }
1793
1794         *vap = va;
1795         *availp = va + (PAGE_SIZE * pages);
1796 }
1797
1798 /*
1799  *      Bootstrap the system enough to run with virtual memory.
1800  *
1801  *      On the arm this is called after mapping has already been enabled
1802  *      and just syncs the pmap module with what has already been done.
1803  *      [We can't call it easily with mapping off since the kernel is not
1804  *      mapped with PA == VA, hence we would have to relocate every address
1805  *      from the linked base (virtual) address "KERNBASE" to the actual
1806  *      (physical) address starting relative to 0]
1807  */
1808 #define PMAP_STATIC_L2_SIZE 16
1809
1810 void
1811 pmap_bootstrap(vm_offset_t firstaddr, struct pv_addr *l1pt)
1812 {
1813         static struct l1_ttable static_l1;
1814         static struct l2_dtable static_l2[PMAP_STATIC_L2_SIZE];
1815         struct l1_ttable *l1 = &static_l1;
1816         struct l2_dtable *l2;
1817         struct l2_bucket *l2b;
1818         struct czpages *czp;
1819         pd_entry_t pde;
1820         pd_entry_t *kernel_l1pt = (pd_entry_t *)l1pt->pv_va;
1821         pt_entry_t *ptep;
1822         vm_paddr_t pa;
1823         vm_offset_t va;
1824         vm_size_t size;
1825         int i, l1idx, l2idx, l2next = 0;
1826
1827         PDEBUG(1, printf("firstaddr = %08x, lastaddr = %08x\n",
1828             firstaddr, vm_max_kernel_address));
1829
1830         virtual_avail = firstaddr;
1831         kernel_pmap->pm_l1 = l1;
1832         kernel_l1pa = l1pt->pv_pa;
1833
1834         /*
1835          * Scan the L1 translation table created by initarm() and create
1836          * the required metadata for all valid mappings found in it.
1837          */
1838         for (l1idx = 0; l1idx < (L1_TABLE_SIZE / sizeof(pd_entry_t)); l1idx++) {
1839                 pde = kernel_l1pt[l1idx];
1840
1841                 /*
1842                  * We're only interested in Coarse mappings.
1843                  * pmap_extract() can deal with section mappings without
1844                  * recourse to checking L2 metadata.
1845                  */
1846                 if ((pde & L1_TYPE_MASK) != L1_TYPE_C)
1847                         continue;
1848
1849                 /*
1850                  * Lookup the KVA of this L2 descriptor table
1851                  */
1852                 pa = (vm_paddr_t)(pde & L1_C_ADDR_MASK);
1853                 ptep = (pt_entry_t *)kernel_pt_lookup(pa);
1854
1855                 if (ptep == NULL) {
1856                         panic("pmap_bootstrap: No L2 for va 0x%x, pa 0x%lx",
1857                             (u_int)l1idx << L1_S_SHIFT, (long unsigned int)pa);
1858                 }
1859
1860                 /*
1861                  * Fetch the associated L2 metadata structure.
1862                  * Allocate a new one if necessary.
1863                  */
1864                 if ((l2 = kernel_pmap->pm_l2[L2_IDX(l1idx)]) == NULL) {
1865                         if (l2next == PMAP_STATIC_L2_SIZE)
1866                                 panic("pmap_bootstrap: out of static L2s");
1867                         kernel_pmap->pm_l2[L2_IDX(l1idx)] = l2 =
1868                             &static_l2[l2next++];
1869                 }
1870
1871                 /*
1872                  * One more L1 slot tracked...
1873                  */
1874                 l2->l2_occupancy++;
1875
1876                 /*
1877                  * Fill in the details of the L2 descriptor in the
1878                  * appropriate bucket.
1879                  */
1880                 l2b = &l2->l2_bucket[L2_BUCKET(l1idx)];
1881                 l2b->l2b_kva = ptep;
1882                 l2b->l2b_phys = pa;
1883                 l2b->l2b_l1idx = l1idx;
1884
1885                 /*
1886                  * Establish an initial occupancy count for this descriptor
1887                  */
1888                 for (l2idx = 0;
1889                     l2idx < (L2_TABLE_SIZE_REAL / sizeof(pt_entry_t));
1890                     l2idx++) {
1891                         if ((ptep[l2idx] & L2_TYPE_MASK) != L2_TYPE_INV) {
1892                                 l2b->l2b_occupancy++;
1893                         }
1894                 }
1895
1896                 /*
1897                  * Make sure the descriptor itself has the correct cache mode.
1898                  * If not, fix it, but whine about the problem. Port-meisters
1899                  * should consider this a clue to fix up their initarm()
1900                  * function. :)
1901                  */
1902                 if (pmap_set_pt_cache_mode(kernel_l1pt, (vm_offset_t)ptep)) {
1903                         printf("pmap_bootstrap: WARNING! wrong cache mode for "
1904                             "L2 pte @ %p\n", ptep);
1905                 }
1906         }
1907
1908
1909         /*
1910          * Ensure the primary (kernel) L1 has the correct cache mode for
1911          * a page table. Bitch if it is not correctly set.
1912          */
1913         for (va = (vm_offset_t)kernel_l1pt;
1914             va < ((vm_offset_t)kernel_l1pt + L1_TABLE_SIZE); va += PAGE_SIZE) {
1915                 if (pmap_set_pt_cache_mode(kernel_l1pt, va))
1916                         printf("pmap_bootstrap: WARNING! wrong cache mode for "
1917                             "primary L1 @ 0x%x\n", va);
1918         }
1919
1920         cpu_dcache_wbinv_all();
1921         cpu_l2cache_wbinv_all();
1922         cpu_tlb_flushID();
1923         cpu_cpwait();
1924
1925         PMAP_LOCK_INIT(kernel_pmap);
1926         CPU_FILL(&kernel_pmap->pm_active);
1927         kernel_pmap->pm_domain = PMAP_DOMAIN_KERNEL;
1928         TAILQ_INIT(&kernel_pmap->pm_pvchunk);
1929
1930         /*
1931          * Initialize the global pv list lock.
1932          */
1933         rw_init(&pvh_global_lock, "pmap pv global");
1934
1935         /*
1936          * Reserve some special page table entries/VA space for temporary
1937          * mapping of pages that are being copied or zeroed.
1938          */
1939         for (czp = cpu_czpages, i = 0; i < MAXCPU; ++i, ++czp) {
1940                 mtx_init(&czp->lock, "czpages", NULL, MTX_DEF);
1941                 pmap_alloc_specials(&virtual_avail, 1, &czp->srcva, &czp->srcptep);
1942                 pmap_set_pt_cache_mode(kernel_l1pt, (vm_offset_t)czp->srcptep);
1943                 pmap_alloc_specials(&virtual_avail, 1, &czp->dstva, &czp->dstptep);
1944                 pmap_set_pt_cache_mode(kernel_l1pt, (vm_offset_t)czp->dstptep);
1945         }
1946
1947         size = ((vm_max_kernel_address - pmap_curmaxkvaddr) + L1_S_OFFSET) /
1948             L1_S_SIZE;
1949         pmap_alloc_specials(&virtual_avail,
1950             round_page(size * L2_TABLE_SIZE_REAL) / PAGE_SIZE,
1951             &pmap_kernel_l2ptp_kva, NULL);
1952
1953         size = (size + (L2_BUCKET_SIZE - 1)) / L2_BUCKET_SIZE;
1954         pmap_alloc_specials(&virtual_avail,
1955             round_page(size * sizeof(struct l2_dtable)) / PAGE_SIZE,
1956             &pmap_kernel_l2dtable_kva, NULL);
1957
1958         pmap_alloc_specials(&virtual_avail,
1959             1, (vm_offset_t*)&_tmppt, NULL);
1960         pmap_alloc_specials(&virtual_avail,
1961             MAXDUMPPGS, (vm_offset_t *)&crashdumpmap, NULL);
1962         SLIST_INIT(&l1_list);
1963         TAILQ_INIT(&l1_lru_list);
1964         mtx_init(&l1_lru_lock, "l1 list lock", NULL, MTX_DEF);
1965         pmap_init_l1(l1, kernel_l1pt);
1966         cpu_dcache_wbinv_all();
1967         cpu_l2cache_wbinv_all();
1968         cpu_tlb_flushID();
1969         cpu_cpwait();
1970
1971         virtual_avail = round_page(virtual_avail);
1972         virtual_end = vm_max_kernel_address;
1973         kernel_vm_end = pmap_curmaxkvaddr;
1974
1975         pmap_set_pcb_pagedir(kernel_pmap, thread0.td_pcb);
1976 }
1977
1978 /***************************************************
1979  * Pmap allocation/deallocation routines.
1980  ***************************************************/
1981
1982 /*
1983  * Release any resources held by the given physical map.
1984  * Called when a pmap initialized by pmap_pinit is being released.
1985  * Should only be called if the map contains no valid mappings.
1986  */
1987 void
1988 pmap_release(pmap_t pmap)
1989 {
1990         struct pcb *pcb;
1991
1992         cpu_tlb_flushID();
1993         cpu_cpwait();
1994         if (vector_page < KERNBASE) {
1995                 struct pcb *curpcb = PCPU_GET(curpcb);
1996                 pcb = thread0.td_pcb;
1997                 if (pmap_is_current(pmap)) {
1998                         /*
1999                          * Frob the L1 entry corresponding to the vector
2000                          * page so that it contains the kernel pmap's domain
2001                          * number. This will ensure pmap_remove() does not
2002                          * pull the current vector page out from under us.
2003                          */
2004                         critical_enter();
2005                         *pcb->pcb_pl1vec = pcb->pcb_l1vec;
2006                         cpu_domains(pcb->pcb_dacr);
2007                         cpu_setttb(pcb->pcb_pagedir);
2008                         critical_exit();
2009                 }
2010                 pmap_remove(pmap, vector_page, vector_page + PAGE_SIZE);
2011                 /*
2012                  * Make sure cpu_switch(), et al, DTRT. This is safe to do
2013                  * since this process has no remaining mappings of its own.
2014                  */
2015                 curpcb->pcb_pl1vec = pcb->pcb_pl1vec;
2016                 curpcb->pcb_l1vec = pcb->pcb_l1vec;
2017                 curpcb->pcb_dacr = pcb->pcb_dacr;
2018                 curpcb->pcb_pagedir = pcb->pcb_pagedir;
2019
2020         }
2021         pmap_free_l1(pmap);
2022
2023         dprintf("pmap_release()\n");
2024 }
2025
2026
2027
2028 /*
2029  * Helper function for pmap_grow_l2_bucket()
2030  */
2031 static __inline int
2032 pmap_grow_map(vm_offset_t va, pt_entry_t cache_mode, vm_paddr_t *pap)
2033 {
2034         struct l2_bucket *l2b;
2035         pt_entry_t *ptep;
2036         vm_paddr_t pa;
2037         struct vm_page *m;
2038
2039         m = vm_page_alloc(NULL, 0, VM_ALLOC_NOOBJ | VM_ALLOC_WIRED);
2040         if (m == NULL)
2041                 return (1);
2042         pa = VM_PAGE_TO_PHYS(m);
2043
2044         if (pap)
2045                 *pap = pa;
2046
2047         l2b = pmap_get_l2_bucket(pmap_kernel(), va);
2048
2049         ptep = &l2b->l2b_kva[l2pte_index(va)];
2050         *ptep = L2_S_PROTO | pa | cache_mode | L2_S_REF;
2051         pmap_set_prot(ptep, VM_PROT_READ | VM_PROT_WRITE, 0);
2052         PTE_SYNC(ptep);
2053         cpu_tlb_flushD_SE(va);
2054         cpu_cpwait();
2055
2056         return (0);
2057 }
2058
2059 /*
2060  * This is the same as pmap_alloc_l2_bucket(), except that it is only
2061  * used by pmap_growkernel().
2062  */
2063 static __inline struct l2_bucket *
2064 pmap_grow_l2_bucket(pmap_t pmap, vm_offset_t va)
2065 {
2066         struct l2_dtable *l2;
2067         struct l2_bucket *l2b;
2068         struct l1_ttable *l1;
2069         pd_entry_t *pl1pd;
2070         u_short l1idx;
2071         vm_offset_t nva;
2072
2073         l1idx = L1_IDX(va);
2074
2075         if ((l2 = pmap->pm_l2[L2_IDX(l1idx)]) == NULL) {
2076                 /*
2077                  * No mapping at this address, as there is
2078                  * no entry in the L1 table.
2079                  * Need to allocate a new l2_dtable.
2080                  */
2081                 nva = pmap_kernel_l2dtable_kva;
2082                 if ((nva & PAGE_MASK) == 0) {
2083                         /*
2084                          * Need to allocate a backing page
2085                          */
2086                         if (pmap_grow_map(nva, pte_l2_s_cache_mode, NULL))
2087                                 return (NULL);
2088                 }
2089
2090                 l2 = (struct l2_dtable *)nva;
2091                 nva += sizeof(struct l2_dtable);
2092
2093                 if ((nva & PAGE_MASK) < (pmap_kernel_l2dtable_kva &
2094                     PAGE_MASK)) {
2095                         /*
2096                          * The new l2_dtable straddles a page boundary.
2097                          * Map in another page to cover it.
2098                          */
2099                         if (pmap_grow_map(nva, pte_l2_s_cache_mode, NULL))
2100                                 return (NULL);
2101                 }
2102
2103                 pmap_kernel_l2dtable_kva = nva;
2104
2105                 /*
2106                  * Link it into the parent pmap
2107                  */
2108                 pmap->pm_l2[L2_IDX(l1idx)] = l2;
2109                 memset(l2, 0, sizeof(*l2));
2110         }
2111
2112         l2b = &l2->l2_bucket[L2_BUCKET(l1idx)];
2113
2114         /*
2115          * Fetch pointer to the L2 page table associated with the address.
2116          */
2117         if (l2b->l2b_kva == NULL) {
2118                 pt_entry_t *ptep;
2119
2120                 /*
2121                  * No L2 page table has been allocated. Chances are, this
2122                  * is because we just allocated the l2_dtable, above.
2123                  */
2124                 nva = pmap_kernel_l2ptp_kva;
2125                 ptep = (pt_entry_t *)nva;
2126                 if ((nva & PAGE_MASK) == 0) {
2127                         /*
2128                          * Need to allocate a backing page
2129                          */
2130                         if (pmap_grow_map(nva, pte_l2_s_cache_mode_pt,
2131                             &pmap_kernel_l2ptp_phys))
2132                                 return (NULL);
2133                 }
2134                 memset(ptep, 0, L2_TABLE_SIZE_REAL);
2135                 l2->l2_occupancy++;
2136                 l2b->l2b_kva = ptep;
2137                 l2b->l2b_l1idx = l1idx;
2138                 l2b->l2b_phys = pmap_kernel_l2ptp_phys;
2139
2140                 pmap_kernel_l2ptp_kva += L2_TABLE_SIZE_REAL;
2141                 pmap_kernel_l2ptp_phys += L2_TABLE_SIZE_REAL;
2142         }
2143
2144         /* Distribute new L1 entry to all other L1s */
2145         SLIST_FOREACH(l1, &l1_list, l1_link) {
2146                         pl1pd = &l1->l1_kva[L1_IDX(va)];
2147                         *pl1pd = l2b->l2b_phys | L1_C_DOM(PMAP_DOMAIN_KERNEL) |
2148                             L1_C_PROTO;
2149                         PTE_SYNC(pl1pd);
2150         }
2151         cpu_tlb_flushID_SE(va);
2152         cpu_cpwait();
2153
2154         return (l2b);
2155 }
2156
2157
2158 /*
2159  * grow the number of kernel page table entries, if needed
2160  */
2161 void
2162 pmap_growkernel(vm_offset_t addr)
2163 {
2164         pmap_t kpmap = pmap_kernel();
2165
2166         if (addr <= pmap_curmaxkvaddr)
2167                 return;         /* we are OK */
2168
2169         /*
2170          * whoops!   we need to add kernel PTPs
2171          */
2172
2173         /* Map 1MB at a time */
2174         for (; pmap_curmaxkvaddr < addr; pmap_curmaxkvaddr += L1_S_SIZE)
2175                 pmap_grow_l2_bucket(kpmap, pmap_curmaxkvaddr);
2176
2177         kernel_vm_end = pmap_curmaxkvaddr;
2178 }
2179
2180 /*
2181  * Returns TRUE if the given page is mapped individually or as part of
2182  * a 1MB section.  Otherwise, returns FALSE.
2183  */
2184 boolean_t
2185 pmap_page_is_mapped(vm_page_t m)
2186 {
2187         boolean_t rv;
2188
2189         if ((m->oflags & VPO_UNMANAGED) != 0)
2190                 return (FALSE);
2191         rw_wlock(&pvh_global_lock);
2192         rv = !TAILQ_EMPTY(&m->md.pv_list) ||
2193             ((m->flags & PG_FICTITIOUS) == 0 &&
2194             !TAILQ_EMPTY(&pa_to_pvh(VM_PAGE_TO_PHYS(m))->pv_list));
2195         rw_wunlock(&pvh_global_lock);
2196         return (rv);
2197 }
2198
2199 /*
2200  * Remove all pages from specified address space
2201  * this aids process exit speeds.  Also, this code
2202  * is special cased for current process only, but
2203  * can have the more generic (and slightly slower)
2204  * mode enabled.  This is much faster than pmap_remove
2205  * in the case of running down an entire address space.
2206  */
2207 void
2208 pmap_remove_pages(pmap_t pmap)
2209 {
2210         struct pv_entry *pv;
2211         struct l2_bucket *l2b = NULL;
2212         struct pv_chunk *pc, *npc;
2213         struct md_page *pvh;
2214         pd_entry_t *pl1pd, l1pd;
2215         pt_entry_t *ptep;
2216         vm_page_t m, mt;
2217         vm_offset_t va;
2218         uint32_t inuse, bitmask;
2219         int allfree, bit, field, idx;
2220
2221         rw_wlock(&pvh_global_lock);
2222         PMAP_LOCK(pmap);
2223
2224         TAILQ_FOREACH_SAFE(pc, &pmap->pm_pvchunk, pc_list, npc) {
2225                 allfree = 1;
2226                 for (field = 0; field < _NPCM; field++) {
2227                         inuse = ~pc->pc_map[field] & pc_freemask[field];
2228                         while (inuse != 0) {
2229                                 bit = ffs(inuse) - 1;
2230                                 bitmask = 1ul << bit;
2231                                 idx = field * sizeof(inuse) * NBBY + bit;
2232                                 pv = &pc->pc_pventry[idx];
2233                                 va = pv->pv_va;
2234                                 inuse &= ~bitmask;
2235                                 if (pv->pv_flags & PVF_WIRED) {
2236                                         /* Cannot remove wired pages now. */
2237                                         allfree = 0;
2238                                         continue;
2239                                 }
2240                                 pl1pd = &pmap->pm_l1->l1_kva[L1_IDX(va)];
2241                                 l1pd = *pl1pd;
2242                                 l2b = pmap_get_l2_bucket(pmap, va);
2243                                 if ((l1pd & L1_TYPE_MASK) == L1_S_PROTO) {
2244                                         pvh = pa_to_pvh(l1pd & L1_S_FRAME);
2245                                         TAILQ_REMOVE(&pvh->pv_list, pv, pv_list);
2246                                         if (TAILQ_EMPTY(&pvh->pv_list)) {
2247                                                 m = PHYS_TO_VM_PAGE(l1pd & L1_S_FRAME);
2248                                                 KASSERT((vm_offset_t)m >= KERNBASE,
2249                                                     ("Trying to access non-existent page "
2250                                                      "va %x l1pd %x", trunc_1mpage(va), l1pd));
2251                                                 for (mt = m; mt < &m[L2_PTE_NUM_TOTAL]; mt++) {
2252                                                         if (TAILQ_EMPTY(&mt->md.pv_list))
2253                                                                 vm_page_aflag_clear(mt, PGA_WRITEABLE);
2254                                                 }
2255                                         }
2256                                         if (l2b != NULL) {
2257                                                 KASSERT(l2b->l2b_occupancy == L2_PTE_NUM_TOTAL,
2258                                                     ("pmap_remove_pages: l2_bucket occupancy error"));
2259                                                 pmap_free_l2_bucket(pmap, l2b, L2_PTE_NUM_TOTAL);
2260                                         }
2261                                         pmap->pm_stats.resident_count -= L2_PTE_NUM_TOTAL;
2262                                         *pl1pd = 0;
2263                                         PTE_SYNC(pl1pd);
2264                                 } else {
2265                                         KASSERT(l2b != NULL,
2266                                             ("No L2 bucket in pmap_remove_pages"));
2267                                         ptep = &l2b->l2b_kva[l2pte_index(va)];
2268                                         m = PHYS_TO_VM_PAGE(l2pte_pa(*ptep));
2269                                         KASSERT((vm_offset_t)m >= KERNBASE,
2270                                             ("Trying to access non-existent page "
2271                                              "va %x pte %x", va, *ptep));
2272                                         TAILQ_REMOVE(&m->md.pv_list, pv, pv_list);
2273                                         if (TAILQ_EMPTY(&m->md.pv_list) &&
2274                                             (m->flags & PG_FICTITIOUS) == 0) {
2275                                                 pvh = pa_to_pvh(l2pte_pa(*ptep));
2276                                                 if (TAILQ_EMPTY(&pvh->pv_list))
2277                                                         vm_page_aflag_clear(m, PGA_WRITEABLE);
2278                                         }
2279                                         *ptep = 0;
2280                                         PTE_SYNC(ptep);
2281                                         pmap_free_l2_bucket(pmap, l2b, 1);
2282                                         pmap->pm_stats.resident_count--;
2283                                 }
2284
2285                                 /* Mark free */
2286                                 PV_STAT(pv_entry_frees++);
2287                                 PV_STAT(pv_entry_spare++);
2288                                 pv_entry_count--;
2289                                 pc->pc_map[field] |= bitmask;
2290                         }
2291                 }
2292                 if (allfree) {
2293                         TAILQ_REMOVE(&pmap->pm_pvchunk, pc, pc_list);
2294                         pmap_free_pv_chunk(pc);
2295                 }
2296
2297         }
2298
2299         rw_wunlock(&pvh_global_lock);
2300         cpu_tlb_flushID();
2301         cpu_cpwait();
2302         PMAP_UNLOCK(pmap);
2303 }
2304
2305
2306 /***************************************************
2307  * Low level mapping routines.....
2308  ***************************************************/
2309
2310 #ifdef ARM_HAVE_SUPERSECTIONS
2311 /* Map a super section into the KVA. */
2312
2313 void
2314 pmap_kenter_supersection(vm_offset_t va, uint64_t pa, int flags)
2315 {
2316         pd_entry_t pd = L1_S_PROTO | L1_S_SUPERSEC | (pa & L1_SUP_FRAME) |
2317             (((pa >> 32) & 0xf) << 20) | L1_S_PROT(PTE_KERNEL,
2318             VM_PROT_READ|VM_PROT_WRITE|VM_PROT_EXECUTE) |
2319             L1_S_DOM(PMAP_DOMAIN_KERNEL);
2320         struct l1_ttable *l1;
2321         vm_offset_t va0, va_end;
2322
2323         KASSERT(((va | pa) & L1_SUP_OFFSET) == 0,
2324             ("Not a valid super section mapping"));
2325         if (flags & SECTION_CACHE)
2326                 pd |= pte_l1_s_cache_mode;
2327         else if (flags & SECTION_PT)
2328                 pd |= pte_l1_s_cache_mode_pt;
2329
2330         va0 = va & L1_SUP_FRAME;
2331         va_end = va + L1_SUP_SIZE;
2332         SLIST_FOREACH(l1, &l1_list, l1_link) {
2333                 va = va0;
2334                 for (; va < va_end; va += L1_S_SIZE) {
2335                         l1->l1_kva[L1_IDX(va)] = pd;
2336                         PTE_SYNC(&l1->l1_kva[L1_IDX(va)]);
2337                 }
2338         }
2339 }
2340 #endif
2341
2342 /* Map a section into the KVA. */
2343
2344 void
2345 pmap_kenter_section(vm_offset_t va, vm_offset_t pa, int flags)
2346 {
2347         pd_entry_t pd = L1_S_PROTO | pa | L1_S_PROT(PTE_KERNEL,
2348             VM_PROT_READ|VM_PROT_WRITE|VM_PROT_EXECUTE) | L1_S_REF |
2349             L1_S_DOM(PMAP_DOMAIN_KERNEL);
2350         struct l1_ttable *l1;
2351
2352         KASSERT(((va | pa) & L1_S_OFFSET) == 0,
2353             ("Not a valid section mapping"));
2354         if (flags & SECTION_CACHE)
2355                 pd |= pte_l1_s_cache_mode;
2356         else if (flags & SECTION_PT)
2357                 pd |= pte_l1_s_cache_mode_pt;
2358
2359         SLIST_FOREACH(l1, &l1_list, l1_link) {
2360                 l1->l1_kva[L1_IDX(va)] = pd;
2361                 PTE_SYNC(&l1->l1_kva[L1_IDX(va)]);
2362         }
2363         cpu_tlb_flushID_SE(va);
2364         cpu_cpwait();
2365 }
2366
2367 /*
2368  * Make a temporary mapping for a physical address.  This is only intended
2369  * to be used for panic dumps.
2370  */
2371 void *
2372 pmap_kenter_temporary(vm_paddr_t pa, int i)
2373 {
2374         vm_offset_t va;
2375
2376         va = (vm_offset_t)crashdumpmap + (i * PAGE_SIZE);
2377         pmap_kenter(va, pa);
2378         return ((void *)crashdumpmap);
2379 }
2380
2381 /*
2382  * add a wired page to the kva
2383  * note that in order for the mapping to take effect -- you
2384  * should do a invltlb after doing the pmap_kenter...
2385  */
2386 static PMAP_INLINE void
2387 pmap_kenter_internal(vm_offset_t va, vm_offset_t pa, int flags)
2388 {
2389         struct l2_bucket *l2b;
2390         pt_entry_t *ptep;
2391         pt_entry_t opte;
2392
2393         PDEBUG(1, printf("pmap_kenter: va = %08x, pa = %08x\n",
2394             (uint32_t) va, (uint32_t) pa));
2395
2396
2397         l2b = pmap_get_l2_bucket(pmap_kernel(), va);
2398         if (l2b == NULL)
2399                 l2b = pmap_grow_l2_bucket(pmap_kernel(), va);
2400         KASSERT(l2b != NULL, ("No L2 Bucket"));
2401
2402         ptep = &l2b->l2b_kva[l2pte_index(va)];
2403         opte = *ptep;
2404
2405         if (flags & KENTER_CACHE)
2406                 *ptep = L2_S_PROTO | l2s_mem_types[PTE_CACHE] | pa | L2_S_REF;
2407         else if (flags & KENTER_DEVICE)
2408                 *ptep = L2_S_PROTO | l2s_mem_types[PTE_DEVICE] | pa | L2_S_REF;
2409         else
2410                 *ptep = L2_S_PROTO | l2s_mem_types[PTE_NOCACHE] | pa | L2_S_REF;
2411
2412         if (flags & KENTER_CACHE) {
2413                 pmap_set_prot(ptep, VM_PROT_READ | VM_PROT_WRITE,
2414                     flags & KENTER_USER);
2415         } else {
2416                 pmap_set_prot(ptep, VM_PROT_READ|VM_PROT_WRITE|VM_PROT_EXECUTE,
2417                     0);
2418         }
2419
2420         PTE_SYNC(ptep);
2421         if (l2pte_valid(opte)) {
2422                 if (L2_S_EXECUTABLE(opte) || L2_S_EXECUTABLE(*ptep))
2423                         cpu_tlb_flushID_SE(va);
2424                 else
2425                         cpu_tlb_flushD_SE(va);
2426         } else {
2427                 if (opte == 0)
2428                         l2b->l2b_occupancy++;
2429         }
2430         cpu_cpwait();
2431
2432         PDEBUG(1, printf("pmap_kenter: pte = %08x, opte = %08x, npte = %08x\n",
2433             (uint32_t) ptep, opte, *ptep));
2434 }
2435
2436 void
2437 pmap_kenter(vm_offset_t va, vm_paddr_t pa)
2438 {
2439         pmap_kenter_internal(va, pa, KENTER_CACHE);
2440 }
2441
2442 void
2443 pmap_kenter_nocache(vm_offset_t va, vm_paddr_t pa)
2444 {
2445
2446         pmap_kenter_internal(va, pa, 0);
2447 }
2448
2449 void
2450 pmap_kenter_device(vm_offset_t va, vm_paddr_t pa)
2451 {
2452
2453         pmap_kenter_internal(va, pa, KENTER_DEVICE);
2454 }
2455
2456 void
2457 pmap_kenter_user(vm_offset_t va, vm_paddr_t pa)
2458 {
2459
2460         pmap_kenter_internal(va, pa, KENTER_CACHE|KENTER_USER);
2461         /*
2462          * Call pmap_fault_fixup now, to make sure we'll have no exception
2463          * at the first use of the new address, or bad things will happen,
2464          * as we use one of these addresses in the exception handlers.
2465          */
2466         pmap_fault_fixup(pmap_kernel(), va, VM_PROT_READ|VM_PROT_WRITE, 1);
2467 }
2468
2469 vm_paddr_t
2470 pmap_kextract(vm_offset_t va)
2471 {
2472
2473         if (kernel_vm_end == 0)
2474                 return (0);
2475         return (pmap_extract_locked(kernel_pmap, va));
2476 }
2477
2478 /*
2479  * remove a page from the kernel pagetables
2480  */
2481 void
2482 pmap_kremove(vm_offset_t va)
2483 {
2484         struct l2_bucket *l2b;
2485         pt_entry_t *ptep, opte;
2486
2487         l2b = pmap_get_l2_bucket(pmap_kernel(), va);
2488         if (!l2b)
2489                 return;
2490         KASSERT(l2b != NULL, ("No L2 Bucket"));
2491         ptep = &l2b->l2b_kva[l2pte_index(va)];
2492         opte = *ptep;
2493         if (l2pte_valid(opte)) {
2494                 va = va & ~PAGE_MASK;
2495                 *ptep = 0;
2496                 PTE_SYNC(ptep);
2497                 if (L2_S_EXECUTABLE(opte))
2498                         cpu_tlb_flushID_SE(va);
2499                 else
2500                         cpu_tlb_flushD_SE(va);
2501                 cpu_cpwait();
2502         }
2503 }
2504
2505
2506 /*
2507  *      Used to map a range of physical addresses into kernel
2508  *      virtual address space.
2509  *
2510  *      The value passed in '*virt' is a suggested virtual address for
2511  *      the mapping. Architectures which can support a direct-mapped
2512  *      physical to virtual region can return the appropriate address
2513  *      within that region, leaving '*virt' unchanged. Other
2514  *      architectures should map the pages starting at '*virt' and
2515  *      update '*virt' with the first usable address after the mapped
2516  *      region.
2517  */
2518 vm_offset_t
2519 pmap_map(vm_offset_t *virt, vm_offset_t start, vm_offset_t end, int prot)
2520 {
2521         vm_offset_t sva = *virt;
2522         vm_offset_t va = sva;
2523
2524         PDEBUG(1, printf("pmap_map: virt = %08x, start = %08x, end = %08x, "
2525             "prot = %d\n", (uint32_t) *virt, (uint32_t) start, (uint32_t) end,
2526             prot));
2527
2528         while (start < end) {
2529                 pmap_kenter(va, start);
2530                 va += PAGE_SIZE;
2531                 start += PAGE_SIZE;
2532         }
2533         *virt = va;
2534         return (sva);
2535 }
2536
2537 /*
2538  * Add a list of wired pages to the kva
2539  * this routine is only used for temporary
2540  * kernel mappings that do not need to have
2541  * page modification or references recorded.
2542  * Note that old mappings are simply written
2543  * over.  The page *must* be wired.
2544  */
2545 void
2546 pmap_qenter(vm_offset_t va, vm_page_t *m, int count)
2547 {
2548         int i;
2549
2550         for (i = 0; i < count; i++) {
2551                 pmap_kenter_internal(va, VM_PAGE_TO_PHYS(m[i]),
2552                     KENTER_CACHE);
2553                 va += PAGE_SIZE;
2554         }
2555 }
2556
2557
2558 /*
2559  * this routine jerks page mappings from the
2560  * kernel -- it is meant only for temporary mappings.
2561  */
2562 void
2563 pmap_qremove(vm_offset_t va, int count)
2564 {
2565         int i;
2566
2567         for (i = 0; i < count; i++) {
2568                 if (vtophys(va))
2569                         pmap_kremove(va);
2570
2571                 va += PAGE_SIZE;
2572         }
2573 }
2574
2575
2576 /*
2577  * pmap_object_init_pt preloads the ptes for a given object
2578  * into the specified pmap.  This eliminates the blast of soft
2579  * faults on process startup and immediately after an mmap.
2580  */
2581 void
2582 pmap_object_init_pt(pmap_t pmap, vm_offset_t addr, vm_object_t object,
2583     vm_pindex_t pindex, vm_size_t size)
2584 {
2585
2586         VM_OBJECT_ASSERT_WLOCKED(object);
2587         KASSERT(object->type == OBJT_DEVICE || object->type == OBJT_SG,
2588             ("pmap_object_init_pt: non-device object"));
2589 }
2590
2591
2592 /*
2593  *      pmap_is_prefaultable:
2594  *
2595  *      Return whether or not the specified virtual address is elgible
2596  *      for prefault.
2597  */
2598 boolean_t
2599 pmap_is_prefaultable(pmap_t pmap, vm_offset_t addr)
2600 {
2601         pd_entry_t *pdep;
2602         pt_entry_t *ptep;
2603
2604         if (!pmap_get_pde_pte(pmap, addr, &pdep, &ptep))
2605                 return (FALSE);
2606         KASSERT((pdep != NULL && (l1pte_section_p(*pdep) || ptep != NULL)),
2607             ("Valid mapping but no pte ?"));
2608         if (*pdep != 0 && !l1pte_section_p(*pdep))
2609                 if (*ptep == 0)
2610                         return (TRUE);
2611         return (FALSE);
2612 }
2613
2614 /*
2615  * Fetch pointers to the PDE/PTE for the given pmap/VA pair.
2616  * Returns TRUE if the mapping exists, else FALSE.
2617  *
2618  * NOTE: This function is only used by a couple of arm-specific modules.
2619  * It is not safe to take any pmap locks here, since we could be right
2620  * in the middle of debugging the pmap anyway...
2621  *
2622  * It is possible for this routine to return FALSE even though a valid
2623  * mapping does exist. This is because we don't lock, so the metadata
2624  * state may be inconsistent.
2625  *
2626  * NOTE: We can return a NULL *ptp in the case where the L1 pde is
2627  * a "section" mapping.
2628  */
2629 boolean_t
2630 pmap_get_pde_pte(pmap_t pmap, vm_offset_t va, pd_entry_t **pdp,
2631     pt_entry_t **ptp)
2632 {
2633         struct l2_dtable *l2;
2634         pd_entry_t *pl1pd, l1pd;
2635         pt_entry_t *ptep;
2636         u_short l1idx;
2637
2638         if (pmap->pm_l1 == NULL)
2639                 return (FALSE);
2640
2641         l1idx = L1_IDX(va);
2642         *pdp = pl1pd = &pmap->pm_l1->l1_kva[l1idx];
2643         l1pd = *pl1pd;
2644
2645         if (l1pte_section_p(l1pd)) {
2646                 *ptp = NULL;
2647                 return (TRUE);
2648         }
2649
2650         if (pmap->pm_l2 == NULL)
2651                 return (FALSE);
2652
2653         l2 = pmap->pm_l2[L2_IDX(l1idx)];
2654
2655         if (l2 == NULL ||
2656             (ptep = l2->l2_bucket[L2_BUCKET(l1idx)].l2b_kva) == NULL) {
2657                 return (FALSE);
2658         }
2659
2660         *ptp = &ptep[l2pte_index(va)];
2661         return (TRUE);
2662 }
2663
2664 /*
2665  *      Routine:        pmap_remove_all
2666  *      Function:
2667  *              Removes this physical page from
2668  *              all physical maps in which it resides.
2669  *              Reflects back modify bits to the pager.
2670  *
2671  *      Notes:
2672  *              Original versions of this routine were very
2673  *              inefficient because they iteratively called
2674  *              pmap_remove (slow...)
2675  */
2676 void
2677 pmap_remove_all(vm_page_t m)
2678 {
2679         struct md_page *pvh;
2680         pv_entry_t pv;
2681         pmap_t pmap;
2682         pt_entry_t *ptep;
2683         struct l2_bucket *l2b;
2684         boolean_t flush = FALSE;
2685         pmap_t curpmap;
2686         u_int is_exec = 0;
2687
2688         KASSERT((m->oflags & VPO_UNMANAGED) == 0,
2689             ("pmap_remove_all: page %p is not managed", m));
2690         rw_wlock(&pvh_global_lock);
2691         if ((m->flags & PG_FICTITIOUS) != 0)
2692                 goto small_mappings;
2693         pvh = pa_to_pvh(VM_PAGE_TO_PHYS(m));
2694         while ((pv = TAILQ_FIRST(&pvh->pv_list)) != NULL) {
2695                 pmap = PV_PMAP(pv);
2696                 PMAP_LOCK(pmap);
2697                 pd_entry_t *pl1pd;
2698                 pl1pd = &pmap->pm_l1->l1_kva[L1_IDX(pv->pv_va)];
2699                 KASSERT((*pl1pd & L1_TYPE_MASK) == L1_S_PROTO,
2700                     ("pmap_remove_all: valid section mapping expected"));
2701                 (void)pmap_demote_section(pmap, pv->pv_va);
2702                 PMAP_UNLOCK(pmap);
2703         }
2704 small_mappings:
2705         curpmap = vmspace_pmap(curproc->p_vmspace);
2706         while ((pv = TAILQ_FIRST(&m->md.pv_list)) != NULL) {
2707                 pmap = PV_PMAP(pv);
2708                 if (flush == FALSE && (pmap == curpmap ||
2709                     pmap == pmap_kernel()))
2710                         flush = TRUE;
2711
2712                 PMAP_LOCK(pmap);
2713                 l2b = pmap_get_l2_bucket(pmap, pv->pv_va);
2714                 KASSERT(l2b != NULL, ("No l2 bucket"));
2715                 ptep = &l2b->l2b_kva[l2pte_index(pv->pv_va)];
2716                 is_exec |= PTE_BEEN_EXECD(*ptep);
2717                 *ptep = 0;
2718                 if (pmap_is_current(pmap))
2719                         PTE_SYNC(ptep);
2720                 pmap_free_l2_bucket(pmap, l2b, 1);
2721                 pmap->pm_stats.resident_count--;
2722                 TAILQ_REMOVE(&m->md.pv_list, pv, pv_list);
2723                 if (pv->pv_flags & PVF_WIRED)
2724                         pmap->pm_stats.wired_count--;
2725                 pmap_free_pv_entry(pmap, pv);
2726                 PMAP_UNLOCK(pmap);
2727         }
2728
2729         if (flush) {
2730                 if (is_exec)
2731                         cpu_tlb_flushID();
2732                 else
2733                         cpu_tlb_flushD();
2734                 cpu_cpwait();
2735         }
2736         vm_page_aflag_clear(m, PGA_WRITEABLE);
2737         rw_wunlock(&pvh_global_lock);
2738 }
2739
2740 int
2741 pmap_change_attr(vm_offset_t sva, vm_size_t len, int mode)
2742 {
2743         vm_offset_t base, offset, tmpva;
2744         vm_size_t size;
2745         struct l2_bucket *l2b;
2746         pt_entry_t *ptep, pte;
2747         vm_offset_t next_bucket;
2748
2749         PMAP_LOCK(kernel_pmap);
2750
2751         base = trunc_page(sva);
2752         offset = sva & PAGE_MASK;
2753         size = roundup(offset + len, PAGE_SIZE);
2754
2755         for (tmpva = base; tmpva < base + size; ) {
2756                 next_bucket = L2_NEXT_BUCKET(tmpva);
2757                 if (next_bucket > base + size)
2758                         next_bucket = base + size;
2759
2760                 l2b = pmap_get_l2_bucket(kernel_pmap, tmpva);
2761                 if (l2b == NULL) {
2762                         tmpva = next_bucket;
2763                         continue;
2764                 }
2765
2766                 ptep = &l2b->l2b_kva[l2pte_index(tmpva)];
2767
2768                 if (*ptep == 0) {
2769                         PMAP_UNLOCK(kernel_pmap);
2770                         return(EINVAL);
2771                 }
2772
2773                 pte = *ptep &~ L2_S_CACHE_MASK;
2774                 cpu_idcache_wbinv_range(tmpva, PAGE_SIZE);
2775                 pmap_l2cache_wbinv_range(tmpva, pte & L2_S_FRAME, PAGE_SIZE);
2776                 *ptep = pte;
2777                 cpu_tlb_flushID_SE(tmpva);
2778                 cpu_cpwait();
2779
2780                 dprintf("%s: for va:%x ptep:%x pte:%x\n",
2781                     __func__, tmpva, (uint32_t)ptep, pte);
2782                 tmpva += PAGE_SIZE;
2783         }
2784
2785         PMAP_UNLOCK(kernel_pmap);
2786
2787         return (0);
2788 }
2789
2790 /*
2791  *      Set the physical protection on the
2792  *      specified range of this map as requested.
2793  */
2794 void
2795 pmap_protect(pmap_t pmap, vm_offset_t sva, vm_offset_t eva, vm_prot_t prot)
2796 {
2797         struct l2_bucket *l2b;
2798         struct md_page *pvh;
2799         struct pv_entry *pve;
2800         pd_entry_t *pl1pd, l1pd;
2801         pt_entry_t *ptep, pte;
2802         vm_offset_t next_bucket;
2803         u_int is_exec, is_refd;
2804         int flush;
2805
2806         if ((prot & VM_PROT_READ) == 0) {
2807                 pmap_remove(pmap, sva, eva);
2808                 return;
2809         }
2810
2811         if (prot & VM_PROT_WRITE) {
2812                 /*
2813                  * If this is a read->write transition, just ignore it and let
2814                  * vm_fault() take care of it later.
2815                  */
2816                 return;
2817         }
2818
2819         rw_wlock(&pvh_global_lock);
2820         PMAP_LOCK(pmap);
2821
2822         /*
2823          * OK, at this point, we know we're doing write-protect operation.
2824          * If the pmap is active, write-back the range.
2825          */
2826
2827         flush = ((eva - sva) >= (PAGE_SIZE * 4)) ? 0 : -1;
2828         is_exec = is_refd = 0;
2829
2830         while (sva < eva) {
2831                 next_bucket = L2_NEXT_BUCKET(sva);
2832                 /*
2833                  * Check for large page.
2834                  */
2835                 pl1pd = &pmap->pm_l1->l1_kva[L1_IDX(sva)];
2836                 l1pd = *pl1pd;
2837                 if ((l1pd & L1_TYPE_MASK) == L1_S_PROTO) {
2838                         KASSERT(pmap != pmap_kernel(),
2839                             ("pmap_protect: trying to modify "
2840                             "kernel section protections"));
2841                         /*
2842                          * Are we protecting the entire large page? If not,
2843                          * demote the mapping and fall through.
2844                          */
2845                         if (sva + L1_S_SIZE == next_bucket &&
2846                             eva >= next_bucket) {
2847                                 l1pd &= ~(L1_S_PROT_MASK | L1_S_XN);
2848                                 if (!(prot & VM_PROT_EXECUTE))
2849                                         l1pd |= L1_S_XN;
2850                                 /*
2851                                  * At this point we are always setting
2852                                  * write-protect bit.
2853                                  */
2854                                 l1pd |= L1_S_APX;
2855                                 /* All managed superpages are user pages. */
2856                                 l1pd |= L1_S_PROT_U;
2857                                 *pl1pd = l1pd;
2858                                 PTE_SYNC(pl1pd);
2859                                 pvh = pa_to_pvh(l1pd & L1_S_FRAME);
2860                                 pve = pmap_find_pv(pvh, pmap,
2861                                     trunc_1mpage(sva));
2862                                 pve->pv_flags &= ~PVF_WRITE;
2863                                 sva = next_bucket;
2864                                 continue;
2865                         } else if (!pmap_demote_section(pmap, sva)) {
2866                                 /* The large page mapping was destroyed. */
2867                                 sva = next_bucket;
2868                                 continue;
2869                         }
2870                 }
2871                 if (next_bucket > eva)
2872                         next_bucket = eva;
2873                 l2b = pmap_get_l2_bucket(pmap, sva);
2874                 if (l2b == NULL) {
2875                         sva = next_bucket;
2876                         continue;
2877                 }
2878
2879                 ptep = &l2b->l2b_kva[l2pte_index(sva)];
2880
2881                 while (sva < next_bucket) {
2882                         if ((pte = *ptep) != 0 && L2_S_WRITABLE(pte)) {
2883                                 struct vm_page *m;
2884
2885                                 m = PHYS_TO_VM_PAGE(l2pte_pa(pte));
2886                                 pmap_set_prot(ptep, prot,
2887                                     !(pmap == pmap_kernel()));
2888                                 PTE_SYNC(ptep);
2889
2890                                 pmap_modify_pv(m, pmap, sva, PVF_WRITE, 0);
2891
2892                                 if (flush >= 0) {
2893                                         flush++;
2894                                         is_exec |= PTE_BEEN_EXECD(pte);
2895                                         is_refd |= PTE_BEEN_REFD(pte);
2896                                 } else {
2897                                         if (PTE_BEEN_EXECD(pte))
2898                                                 cpu_tlb_flushID_SE(sva);
2899                                         else if (PTE_BEEN_REFD(pte))
2900                                                 cpu_tlb_flushD_SE(sva);
2901                                 }
2902                         }
2903
2904                         sva += PAGE_SIZE;
2905                         ptep++;
2906                 }
2907         }
2908
2909
2910         if (flush) {
2911                 if (is_exec)
2912                         cpu_tlb_flushID();
2913                 else
2914                 if (is_refd)
2915                         cpu_tlb_flushD();
2916                 cpu_cpwait();
2917         }
2918         rw_wunlock(&pvh_global_lock);
2919
2920         PMAP_UNLOCK(pmap);
2921 }
2922
2923
2924 /*
2925  *      Insert the given physical page (p) at
2926  *      the specified virtual address (v) in the
2927  *      target physical map with the protection requested.
2928  *
2929  *      If specified, the page will be wired down, meaning
2930  *      that the related pte can not be reclaimed.
2931  *
2932  *      NB:  This is the only routine which MAY NOT lazy-evaluate
2933  *      or lose information.  That is, this routine must actually
2934  *      insert this page into the given map NOW.
2935  */
2936
2937 int
2938 pmap_enter(pmap_t pmap, vm_offset_t va, vm_page_t m, vm_prot_t prot,
2939     u_int flags, int8_t psind __unused)
2940 {
2941         struct l2_bucket *l2b;
2942         int rv;
2943
2944         rw_wlock(&pvh_global_lock);
2945         PMAP_LOCK(pmap);
2946         rv = pmap_enter_locked(pmap, va, m, prot, flags);
2947         if (rv == KERN_SUCCESS) {
2948                 /*
2949                  * If both the l2b_occupancy and the reservation are fully
2950                  * populated, then attempt promotion.
2951                  */
2952                 l2b = pmap_get_l2_bucket(pmap, va);
2953                 if (l2b != NULL && l2b->l2b_occupancy == L2_PTE_NUM_TOTAL &&
2954                     sp_enabled && (m->flags & PG_FICTITIOUS) == 0 &&
2955                     vm_reserv_level_iffullpop(m) == 0)
2956                         pmap_promote_section(pmap, va);
2957         }
2958         PMAP_UNLOCK(pmap);
2959         rw_wunlock(&pvh_global_lock);
2960         return (rv);
2961 }
2962
2963 /*
2964  *      The pvh global and pmap locks must be held.
2965  */
2966 static int
2967 pmap_enter_locked(pmap_t pmap, vm_offset_t va, vm_page_t m, vm_prot_t prot,
2968     u_int flags)
2969 {
2970         struct l2_bucket *l2b = NULL;
2971         struct vm_page *om;
2972         struct pv_entry *pve = NULL;
2973         pd_entry_t *pl1pd, l1pd;
2974         pt_entry_t *ptep, npte, opte;
2975         u_int nflags;
2976         u_int is_exec, is_refd;
2977         vm_paddr_t pa;
2978         u_char user;
2979
2980         PMAP_ASSERT_LOCKED(pmap);
2981         rw_assert(&pvh_global_lock, RA_WLOCKED);
2982         if (va == vector_page) {
2983                 pa = systempage.pv_pa;
2984                 m = NULL;
2985         } else {
2986                 if ((m->oflags & VPO_UNMANAGED) == 0 && !vm_page_xbusied(m))
2987                         VM_OBJECT_ASSERT_LOCKED(m->object);
2988                 pa = VM_PAGE_TO_PHYS(m);
2989         }
2990
2991         pl1pd = &pmap->pm_l1->l1_kva[L1_IDX(va)];
2992         if ((va < VM_MAXUSER_ADDRESS) &&
2993             (*pl1pd & L1_TYPE_MASK) == L1_S_PROTO) {
2994                 (void)pmap_demote_section(pmap, va);
2995         }
2996
2997         user = 0;
2998         /*
2999          * Make sure userland mappings get the right permissions
3000          */
3001         if (pmap != pmap_kernel() && va != vector_page)
3002                 user = 1;
3003
3004         nflags = 0;
3005
3006         if (prot & VM_PROT_WRITE)
3007                 nflags |= PVF_WRITE;
3008         if ((flags & PMAP_ENTER_WIRED) != 0)
3009                 nflags |= PVF_WIRED;
3010
3011         PDEBUG(1, printf("pmap_enter: pmap = %08x, va = %08x, m = %08x, "
3012             "prot = %x, flags = %x\n", (uint32_t) pmap, va, (uint32_t) m,
3013             prot, flags));
3014
3015         if (pmap == pmap_kernel()) {
3016                 l2b = pmap_get_l2_bucket(pmap, va);
3017                 if (l2b == NULL)
3018                         l2b = pmap_grow_l2_bucket(pmap, va);
3019         } else {
3020 do_l2b_alloc:
3021                 l2b = pmap_alloc_l2_bucket(pmap, va);
3022                 if (l2b == NULL) {
3023                         if ((flags & PMAP_ENTER_NOSLEEP) == 0) {
3024                                 PMAP_UNLOCK(pmap);
3025                                 rw_wunlock(&pvh_global_lock);
3026                                 VM_WAIT;
3027                                 rw_wlock(&pvh_global_lock);
3028                                 PMAP_LOCK(pmap);
3029                                 goto do_l2b_alloc;
3030                         }
3031                         return (KERN_RESOURCE_SHORTAGE);
3032                 }
3033         }
3034
3035         pl1pd = &pmap->pm_l1->l1_kva[L1_IDX(va)];
3036         if ((*pl1pd & L1_TYPE_MASK) == L1_S_PROTO)
3037                 panic("pmap_enter: attempt to enter on 1MB page, va: %#x", va);
3038
3039         ptep = &l2b->l2b_kva[l2pte_index(va)];
3040
3041         opte = *ptep;
3042         npte = pa;
3043         is_exec = is_refd = 0;
3044
3045         if (opte) {
3046                 if (l2pte_pa(opte) == pa) {
3047                         /*
3048                          * We're changing the attrs of an existing mapping.
3049                          */
3050                         if (m != NULL)
3051                                 pmap_modify_pv(m, pmap, va,
3052                                     PVF_WRITE | PVF_WIRED, nflags);
3053                         is_exec |= PTE_BEEN_EXECD(opte);
3054                         is_refd |= PTE_BEEN_REFD(opte);
3055                         goto validate;
3056                 }
3057                 if ((om = PHYS_TO_VM_PAGE(l2pte_pa(opte)))) {
3058                         /*
3059                          * Replacing an existing mapping with a new one.
3060                          * It is part of our managed memory so we
3061                          * must remove it from the PV list
3062                          */
3063                         if ((pve = pmap_remove_pv(om, pmap, va))) {
3064                                 is_exec |= PTE_BEEN_EXECD(opte);
3065                                 is_refd |= PTE_BEEN_REFD(opte);
3066
3067                                 if (m && ((m->oflags & VPO_UNMANAGED)))
3068                                         pmap_free_pv_entry(pmap, pve);
3069                         }
3070                 }
3071
3072         } else {
3073                 /*
3074                  * Keep the stats up to date
3075                  */
3076                 l2b->l2b_occupancy++;
3077                 pmap->pm_stats.resident_count++;
3078         }
3079
3080         /*
3081          * Enter on the PV list if part of our managed memory.
3082          */
3083         if ((m && !(m->oflags & VPO_UNMANAGED))) {
3084                 if ((!pve) && (pve = pmap_get_pv_entry(pmap, FALSE)) == NULL)
3085                         panic("pmap_enter: no pv entries");
3086
3087                 KASSERT(va < kmi.clean_sva || va >= kmi.clean_eva,
3088                 ("pmap_enter: managed mapping within the clean submap"));
3089                 KASSERT(pve != NULL, ("No pv"));
3090                 pmap_enter_pv(m, pve, pmap, va, nflags);
3091         }
3092
3093 validate:
3094         /* Make the new PTE valid */
3095         npte |= L2_S_PROTO;
3096 #ifdef SMP
3097         npte |= L2_SHARED;
3098 #endif
3099         /* Set defaults first - kernel read access */
3100         npte |= L2_APX;
3101         npte |= L2_S_PROT_R;
3102         /* Set "referenced" flag */
3103         npte |= L2_S_REF;
3104
3105         /* Now tune APs as desired */
3106         if (user)
3107                 npte |= L2_S_PROT_U;
3108         /*
3109          * If this is not a vector_page
3110          * then continue setting mapping parameters
3111          */
3112         if (m != NULL) {
3113                 if ((m->oflags & VPO_UNMANAGED) == 0) {
3114                         if (prot & (VM_PROT_ALL)) {
3115                                 vm_page_aflag_set(m, PGA_REFERENCED);
3116                         } else {
3117                                 /*
3118                                  * Need to do page referenced emulation.
3119                                  */
3120                                 npte &= ~L2_S_REF;
3121                         }
3122                 }
3123
3124                 if (prot & VM_PROT_WRITE) {
3125                         if ((m->oflags & VPO_UNMANAGED) == 0) {
3126                                 vm_page_aflag_set(m, PGA_WRITEABLE);
3127                                 /*
3128                                  * XXX: Skip modified bit emulation for now.
3129                                  *      The emulation reveals problems
3130                                  *      that result in random failures
3131                                  *      during memory allocation on some
3132                                  *      platforms.
3133                                  *      Therefore, the page is marked RW
3134                                  *      immediately.
3135                                  */
3136                                 npte &= ~(L2_APX);
3137                                 vm_page_dirty(m);
3138                         } else
3139                                 npte &= ~(L2_APX);
3140                 }
3141                 if (!(prot & VM_PROT_EXECUTE))
3142                         npte |= L2_XN;
3143
3144                 if (m->md.pv_memattr != VM_MEMATTR_UNCACHEABLE)
3145                         npte |= pte_l2_s_cache_mode;
3146         }
3147
3148         CTR5(KTR_PMAP,"enter: pmap:%p va:%x prot:%x pte:%x->%x",
3149             pmap, va, prot, opte, npte);
3150         /*
3151          * If this is just a wiring change, the two PTEs will be
3152          * identical, so there's no need to update the page table.
3153          */
3154         if (npte != opte) {
3155                 boolean_t is_cached = pmap_is_current(pmap);
3156
3157                 *ptep = npte;
3158                 PTE_SYNC(ptep);
3159                 if (is_cached) {
3160                         /*
3161                          * We only need to frob the cache/tlb if this pmap
3162                          * is current
3163                          */
3164                         if (L1_IDX(va) != L1_IDX(vector_page) &&
3165                             l2pte_valid(npte)) {
3166                                 /*
3167                                  * This mapping is likely to be accessed as
3168                                  * soon as we return to userland. Fix up the
3169                                  * L1 entry to avoid taking another
3170                                  * page/domain fault.
3171                                  */
3172                                 l1pd = l2b->l2b_phys |
3173                                     L1_C_DOM(pmap->pm_domain) | L1_C_PROTO;
3174                                 if (*pl1pd != l1pd) {
3175                                         *pl1pd = l1pd;
3176                                         PTE_SYNC(pl1pd);
3177                                 }
3178                         }
3179                 }
3180
3181                 if (is_exec)
3182                         cpu_tlb_flushID_SE(va);
3183                 else if (is_refd)
3184                         cpu_tlb_flushD_SE(va);
3185                 cpu_cpwait();
3186         }
3187
3188         if ((pmap != pmap_kernel()) && (pmap == &curproc->p_vmspace->vm_pmap))
3189                 cpu_icache_sync_range(va, PAGE_SIZE);
3190         return (KERN_SUCCESS);
3191 }
3192
3193 /*
3194  * Maps a sequence of resident pages belonging to the same object.
3195  * The sequence begins with the given page m_start.  This page is
3196  * mapped at the given virtual address start.  Each subsequent page is
3197  * mapped at a virtual address that is offset from start by the same
3198  * amount as the page is offset from m_start within the object.  The
3199  * last page in the sequence is the page with the largest offset from
3200  * m_start that can be mapped at a virtual address less than the given
3201  * virtual address end.  Not every virtual page between start and end
3202  * is mapped; only those for which a resident page exists with the
3203  * corresponding offset from m_start are mapped.
3204  */
3205 void
3206 pmap_enter_object(pmap_t pmap, vm_offset_t start, vm_offset_t end,
3207     vm_page_t m_start, vm_prot_t prot)
3208 {
3209         vm_offset_t va;
3210         vm_page_t m;
3211         vm_pindex_t diff, psize;
3212
3213         VM_OBJECT_ASSERT_LOCKED(m_start->object);
3214
3215         psize = atop(end - start);
3216         m = m_start;
3217         prot &= VM_PROT_READ | VM_PROT_EXECUTE;
3218         rw_wlock(&pvh_global_lock);
3219         PMAP_LOCK(pmap);
3220         while (m != NULL && (diff = m->pindex - m_start->pindex) < psize) {
3221                 va = start + ptoa(diff);
3222                 if ((va & L1_S_OFFSET) == 0 && L2_NEXT_BUCKET(va) <= end &&
3223                     m->psind == 1 && sp_enabled &&
3224                     pmap_enter_section(pmap, va, m, prot))
3225                         m = &m[L1_S_SIZE / PAGE_SIZE - 1];
3226                 else
3227                         pmap_enter_locked(pmap, va, m, prot,
3228                             PMAP_ENTER_NOSLEEP);
3229                 m = TAILQ_NEXT(m, listq);
3230         }
3231         PMAP_UNLOCK(pmap);
3232         rw_wunlock(&pvh_global_lock);
3233 }
3234
3235 /*
3236  * this code makes some *MAJOR* assumptions:
3237  * 1. Current pmap & pmap exists.
3238  * 2. Not wired.
3239  * 3. Read access.
3240  * 4. No page table pages.
3241  * but is *MUCH* faster than pmap_enter...
3242  */
3243
3244 void
3245 pmap_enter_quick(pmap_t pmap, vm_offset_t va, vm_page_t m, vm_prot_t prot)
3246 {
3247
3248         prot &= VM_PROT_READ | VM_PROT_EXECUTE;
3249         rw_wlock(&pvh_global_lock);
3250         PMAP_LOCK(pmap);
3251         pmap_enter_locked(pmap, va, m, prot, PMAP_ENTER_NOSLEEP);
3252         PMAP_UNLOCK(pmap);
3253         rw_wunlock(&pvh_global_lock);
3254 }
3255
3256 /*
3257  *      Clear the wired attribute from the mappings for the specified range of
3258  *      addresses in the given pmap.  Every valid mapping within that range
3259  *      must have the wired attribute set.  In contrast, invalid mappings
3260  *      cannot have the wired attribute set, so they are ignored.
3261  *
3262  *      XXX Wired mappings of unmanaged pages cannot be counted by this pmap
3263  *      implementation.
3264  */
3265 void
3266 pmap_unwire(pmap_t pmap, vm_offset_t sva, vm_offset_t eva)
3267 {
3268         struct l2_bucket *l2b;
3269         struct md_page *pvh;
3270         pd_entry_t l1pd;
3271         pt_entry_t *ptep, pte;
3272         pv_entry_t pv;
3273         vm_offset_t next_bucket;
3274         vm_paddr_t pa;
3275         vm_page_t m;
3276
3277         rw_wlock(&pvh_global_lock);
3278         PMAP_LOCK(pmap);
3279         while (sva < eva) {
3280                 next_bucket = L2_NEXT_BUCKET(sva);
3281                 l1pd = pmap->pm_l1->l1_kva[L1_IDX(sva)];
3282                 if ((l1pd & L1_TYPE_MASK) == L1_S_PROTO) {
3283                         pa = l1pd & L1_S_FRAME;
3284                         m = PHYS_TO_VM_PAGE(pa);
3285                         KASSERT(m != NULL && (m->oflags & VPO_UNMANAGED) == 0,
3286                             ("pmap_unwire: unmanaged 1mpage %p", m));
3287                         pvh = pa_to_pvh(pa);
3288                         pv = pmap_find_pv(pvh, pmap, trunc_1mpage(sva));
3289                         if ((pv->pv_flags & PVF_WIRED) == 0)
3290                                 panic("pmap_unwire: pv %p isn't wired", pv);
3291
3292                         /*
3293                          * Are we unwiring the entire large page? If not,
3294                          * demote the mapping and fall through.
3295                          */
3296                         if (sva + L1_S_SIZE == next_bucket &&
3297                             eva >= next_bucket) {
3298                                 pv->pv_flags &= ~PVF_WIRED;
3299                                 pmap->pm_stats.wired_count -= L2_PTE_NUM_TOTAL;
3300                                 sva = next_bucket;
3301                                 continue;
3302                         } else if (!pmap_demote_section(pmap, sva))
3303                                 panic("pmap_unwire: demotion failed");
3304                 }
3305                 if (next_bucket > eva)
3306                         next_bucket = eva;
3307                 l2b = pmap_get_l2_bucket(pmap, sva);
3308                 if (l2b == NULL) {
3309                         sva = next_bucket;
3310                         continue;
3311                 }
3312                 for (ptep = &l2b->l2b_kva[l2pte_index(sva)]; sva < next_bucket;
3313                     sva += PAGE_SIZE, ptep++) {
3314                         if ((pte = *ptep) == 0 ||
3315                             (m = PHYS_TO_VM_PAGE(l2pte_pa(pte))) == NULL ||
3316                             (m->oflags & VPO_UNMANAGED) != 0)
3317                                 continue;
3318                         pv = pmap_find_pv(&m->md, pmap, sva);
3319                         if ((pv->pv_flags & PVF_WIRED) == 0)
3320                                 panic("pmap_unwire: pv %p isn't wired", pv);
3321                         pv->pv_flags &= ~PVF_WIRED;
3322                         pmap->pm_stats.wired_count--;
3323                 }
3324         }
3325         rw_wunlock(&pvh_global_lock);
3326         PMAP_UNLOCK(pmap);
3327 }
3328
3329
3330 /*
3331  *      Copy the range specified by src_addr/len
3332  *      from the source map to the range dst_addr/len
3333  *      in the destination map.
3334  *
3335  *      This routine is only advisory and need not do anything.
3336  */
3337 void
3338 pmap_copy(pmap_t dst_pmap, pmap_t src_pmap, vm_offset_t dst_addr,
3339     vm_size_t len, vm_offset_t src_addr)
3340 {
3341 }
3342
3343
3344 /*
3345  *      Routine:        pmap_extract
3346  *      Function:
3347  *              Extract the physical page address associated
3348  *              with the given map/virtual_address pair.
3349  */
3350 vm_paddr_t
3351 pmap_extract(pmap_t pmap, vm_offset_t va)
3352 {
3353         vm_paddr_t pa;
3354
3355         PMAP_LOCK(pmap);
3356         pa = pmap_extract_locked(pmap, va);
3357         PMAP_UNLOCK(pmap);
3358         return (pa);
3359 }
3360
3361 static vm_paddr_t
3362 pmap_extract_locked(pmap_t pmap, vm_offset_t va)
3363 {
3364         struct l2_dtable *l2;
3365         pd_entry_t l1pd;
3366         pt_entry_t *ptep, pte;
3367         vm_paddr_t pa;
3368         u_int l1idx;
3369
3370         if (kernel_vm_end != 0 && pmap != kernel_pmap)
3371                 PMAP_ASSERT_LOCKED(pmap);
3372         l1idx = L1_IDX(va);
3373         l1pd = pmap->pm_l1->l1_kva[l1idx];
3374         if (l1pte_section_p(l1pd)) {
3375                 /* XXX: what to do about the bits > 32 ? */
3376                 if (l1pd & L1_S_SUPERSEC)
3377                         pa = (l1pd & L1_SUP_FRAME) | (va & L1_SUP_OFFSET);
3378                 else
3379                         pa = (l1pd & L1_S_FRAME) | (va & L1_S_OFFSET);
3380         } else {
3381                 /*
3382                  * Note that we can't rely on the validity of the L1
3383                  * descriptor as an indication that a mapping exists.
3384                  * We have to look it up in the L2 dtable.
3385                  */
3386                 l2 = pmap->pm_l2[L2_IDX(l1idx)];
3387                 if (l2 == NULL ||
3388                     (ptep = l2->l2_bucket[L2_BUCKET(l1idx)].l2b_kva) == NULL)
3389                         return (0);
3390                 pte = ptep[l2pte_index(va)];
3391                 if (pte == 0)
3392                         return (0);
3393                 switch (pte & L2_TYPE_MASK) {
3394                 case L2_TYPE_L:
3395                         pa = (pte & L2_L_FRAME) | (va & L2_L_OFFSET);
3396                         break;
3397                 default:
3398                         pa = (pte & L2_S_FRAME) | (va & L2_S_OFFSET);
3399                         break;
3400                 }
3401         }
3402         return (pa);
3403 }
3404
3405 /*
3406  * Atomically extract and hold the physical page with the given
3407  * pmap and virtual address pair if that mapping permits the given
3408  * protection.
3409  *
3410  */
3411 vm_page_t
3412 pmap_extract_and_hold(pmap_t pmap, vm_offset_t va, vm_prot_t prot)
3413 {
3414         struct l2_dtable *l2;
3415         pd_entry_t l1pd;
3416         pt_entry_t *ptep, pte;
3417         vm_paddr_t pa, paddr;
3418         vm_page_t m = NULL;
3419         u_int l1idx;
3420         l1idx = L1_IDX(va);
3421         paddr = 0;
3422
3423         PMAP_LOCK(pmap);
3424 retry:
3425         l1pd = pmap->pm_l1->l1_kva[l1idx];
3426         if (l1pte_section_p(l1pd)) {
3427                 /* XXX: what to do about the bits > 32 ? */
3428                 if (l1pd & L1_S_SUPERSEC)
3429                         pa = (l1pd & L1_SUP_FRAME) | (va & L1_SUP_OFFSET);
3430                 else
3431                         pa = (l1pd & L1_S_FRAME) | (va & L1_S_OFFSET);
3432                 if (vm_page_pa_tryrelock(pmap, pa & PG_FRAME, &paddr))
3433                         goto retry;
3434                 if (L1_S_WRITABLE(l1pd) || (prot & VM_PROT_WRITE) == 0) {
3435                         m = PHYS_TO_VM_PAGE(pa);
3436                         vm_page_hold(m);
3437                 }
3438         } else {
3439                 /*
3440                  * Note that we can't rely on the validity of the L1
3441                  * descriptor as an indication that a mapping exists.
3442                  * We have to look it up in the L2 dtable.
3443                  */
3444                 l2 = pmap->pm_l2[L2_IDX(l1idx)];
3445
3446                 if (l2 == NULL ||
3447                     (ptep = l2->l2_bucket[L2_BUCKET(l1idx)].l2b_kva) == NULL) {
3448                         PMAP_UNLOCK(pmap);
3449                         return (NULL);
3450                 }
3451
3452                 ptep = &ptep[l2pte_index(va)];
3453                 pte = *ptep;
3454
3455                 if (pte == 0) {
3456                         PMAP_UNLOCK(pmap);
3457                         return (NULL);
3458                 } else if ((prot & VM_PROT_WRITE) && (pte & L2_APX)) {
3459                         PMAP_UNLOCK(pmap);
3460                         return (NULL);
3461                 } else {
3462                         switch (pte & L2_TYPE_MASK) {
3463                         case L2_TYPE_L:
3464                                 panic("extract and hold section mapping");
3465                                 break;
3466                         default:
3467                                 pa = (pte & L2_S_FRAME) | (va & L2_S_OFFSET);
3468                                 break;
3469                         }
3470                         if (vm_page_pa_tryrelock(pmap, pa & PG_FRAME, &paddr))
3471                                 goto retry;
3472                         m = PHYS_TO_VM_PAGE(pa);
3473                         vm_page_hold(m);
3474                 }
3475
3476         }
3477
3478         PMAP_UNLOCK(pmap);
3479         PA_UNLOCK_COND(paddr);
3480         return (m);
3481 }
3482
3483 /*
3484  * Initialize a preallocated and zeroed pmap structure,
3485  * such as one in a vmspace structure.
3486  */
3487
3488 int
3489 pmap_pinit(pmap_t pmap)
3490 {
3491         PDEBUG(1, printf("pmap_pinit: pmap = %08x\n", (uint32_t) pmap));
3492
3493         pmap_alloc_l1(pmap);
3494         bzero(pmap->pm_l2, sizeof(pmap->pm_l2));
3495
3496         CPU_ZERO(&pmap->pm_active);
3497
3498         TAILQ_INIT(&pmap->pm_pvchunk);
3499         bzero(&pmap->pm_stats, sizeof pmap->pm_stats);
3500         pmap->pm_stats.resident_count = 1;
3501         if (vector_page < KERNBASE) {
3502                 pmap_enter(pmap, vector_page,
3503                     PHYS_TO_VM_PAGE(systempage.pv_pa), VM_PROT_READ,
3504                     PMAP_ENTER_WIRED, 0);
3505         }
3506         return (1);
3507 }
3508
3509
3510 /***************************************************
3511  * Superpage management routines.
3512  ***************************************************/
3513
3514 static PMAP_INLINE struct pv_entry *
3515 pmap_pvh_remove(struct md_page *pvh, pmap_t pmap, vm_offset_t va)
3516 {
3517         pv_entry_t pv;
3518
3519         rw_assert(&pvh_global_lock, RA_WLOCKED);
3520
3521         pv = pmap_find_pv(pvh, pmap, va);
3522         if (pv != NULL)
3523                 TAILQ_REMOVE(&pvh->pv_list, pv, pv_list);
3524
3525         return (pv);
3526 }
3527
3528 static void
3529 pmap_pvh_free(struct md_page *pvh, pmap_t pmap, vm_offset_t va)
3530 {
3531         pv_entry_t pv;
3532
3533         pv = pmap_pvh_remove(pvh, pmap, va);
3534         KASSERT(pv != NULL, ("pmap_pvh_free: pv not found"));
3535         pmap_free_pv_entry(pmap, pv);
3536 }
3537
3538 static boolean_t
3539 pmap_pv_insert_section(pmap_t pmap, vm_offset_t va, vm_paddr_t pa)
3540 {
3541         struct md_page *pvh;
3542         pv_entry_t pv;
3543
3544         rw_assert(&pvh_global_lock, RA_WLOCKED);
3545         if (pv_entry_count < pv_entry_high_water &&
3546             (pv = pmap_get_pv_entry(pmap, TRUE)) != NULL) {
3547                 pv->pv_va = va;
3548                 pvh = pa_to_pvh(pa);
3549                 TAILQ_INSERT_TAIL(&pvh->pv_list, pv, pv_list);
3550                 return (TRUE);
3551         } else
3552                 return (FALSE);
3553 }
3554
3555 /*
3556  * Create the pv entries for each of the pages within a superpage.
3557  */
3558 static void
3559 pmap_pv_demote_section(pmap_t pmap, vm_offset_t va, vm_paddr_t pa)
3560 {
3561         struct md_page *pvh;
3562         pv_entry_t pve, pv;
3563         vm_offset_t va_last;
3564         vm_page_t m;
3565
3566         rw_assert(&pvh_global_lock, RA_WLOCKED);
3567         KASSERT((pa & L1_S_OFFSET) == 0,
3568             ("pmap_pv_demote_section: pa is not 1mpage aligned"));
3569
3570         /*
3571          * Transfer the 1mpage's pv entry for this mapping to the first
3572          * page's pv list.
3573          */
3574         pvh = pa_to_pvh(pa);
3575         va = trunc_1mpage(va);
3576         pv = pmap_pvh_remove(pvh, pmap, va);
3577         KASSERT(pv != NULL, ("pmap_pv_demote_section: pv not found"));
3578         m = PHYS_TO_VM_PAGE(pa);
3579         TAILQ_INSERT_HEAD(&m->md.pv_list, pv, pv_list);
3580         /* Instantiate the remaining pv entries. */
3581         va_last = L2_NEXT_BUCKET(va) - PAGE_SIZE;
3582         do {
3583                 m++;
3584                 KASSERT((m->oflags & VPO_UNMANAGED) == 0,
3585                     ("pmap_pv_demote_section: page %p is not managed", m));
3586                 va += PAGE_SIZE;
3587                 pve = pmap_get_pv_entry(pmap, FALSE);
3588                 pmap_enter_pv(m, pve, pmap, va, pv->pv_flags);
3589         } while (va < va_last);
3590 }
3591
3592 static void
3593 pmap_pv_promote_section(pmap_t pmap, vm_offset_t va, vm_paddr_t pa)
3594 {
3595         struct md_page *pvh;
3596         pv_entry_t pv;
3597         vm_offset_t va_last;
3598         vm_page_t m;
3599
3600         rw_assert(&pvh_global_lock, RA_WLOCKED);
3601         KASSERT((pa & L1_S_OFFSET) == 0,
3602             ("pmap_pv_promote_section: pa is not 1mpage aligned"));
3603
3604         /*
3605          * Transfer the first page's pv entry for this mapping to the
3606          * 1mpage's pv list.  Aside from avoiding the cost of a call
3607          * to get_pv_entry(), a transfer avoids the possibility that
3608          * get_pv_entry() calls pmap_pv_reclaim() and that pmap_pv_reclaim()
3609          * removes one of the mappings that is being promoted.
3610          */
3611         m = PHYS_TO_VM_PAGE(pa);
3612         va = trunc_1mpage(va);
3613         pv = pmap_pvh_remove(&m->md, pmap, va);
3614         KASSERT(pv != NULL, ("pmap_pv_promote_section: pv not found"));
3615         pvh = pa_to_pvh(pa);
3616         TAILQ_INSERT_TAIL(&pvh->pv_list, pv, pv_list);
3617         /* Free the remaining pv entries in the newly mapped section pages */
3618         va_last = L2_NEXT_BUCKET(va) - PAGE_SIZE;
3619         do {
3620                 m++;
3621                 va += PAGE_SIZE;
3622                 /*
3623                  * Don't care the flags, first pv contains sufficient
3624                  * information for all of the pages so nothing is really lost.
3625                  */
3626                 pmap_pvh_free(&m->md, pmap, va);
3627         } while (va < va_last);
3628 }
3629
3630 /*
3631  * Tries to create a 1MB page mapping.  Returns TRUE if successful and
3632  * FALSE otherwise.  Fails if (1) page is unmanageg, kernel pmap or vectors
3633  * page, (2) a mapping already exists at the specified virtual address, or
3634  * (3) a pv entry cannot be allocated without reclaiming another pv entry.
3635  */
3636 static boolean_t
3637 pmap_enter_section(pmap_t pmap, vm_offset_t va, vm_page_t m, vm_prot_t prot)
3638 {
3639         pd_entry_t *pl1pd;
3640         vm_offset_t pa;
3641         struct l2_bucket *l2b;
3642
3643         rw_assert(&pvh_global_lock, RA_WLOCKED);
3644         PMAP_ASSERT_LOCKED(pmap);
3645
3646         /* Skip kernel, vectors page and unmanaged mappings */
3647         if ((pmap == pmap_kernel()) || (L1_IDX(va) == L1_IDX(vector_page)) ||
3648             ((m->oflags & VPO_UNMANAGED) != 0)) {
3649                 CTR2(KTR_PMAP, "pmap_enter_section: failure for va %#lx"
3650                     " in pmap %p", va, pmap);
3651                 return (FALSE);
3652         }
3653         /*
3654          * Check whether this is a valid section superpage entry or
3655          * there is a l2_bucket associated with that L1 page directory.
3656          */
3657         va = trunc_1mpage(va);
3658         pl1pd = &pmap->pm_l1->l1_kva[L1_IDX(va)];
3659         l2b = pmap_get_l2_bucket(pmap, va);
3660         if ((*pl1pd & L1_S_PROTO) || (l2b != NULL)) {
3661                 CTR2(KTR_PMAP, "pmap_enter_section: failure for va %#lx"
3662                     " in pmap %p", va, pmap);
3663                 return (FALSE);
3664         }
3665         pa = VM_PAGE_TO_PHYS(m);
3666         /*
3667          * Abort this mapping if its PV entry could not be created.
3668          */
3669         if (!pmap_pv_insert_section(pmap, va, VM_PAGE_TO_PHYS(m))) {
3670                 CTR2(KTR_PMAP, "pmap_enter_section: failure for va %#lx"
3671                     " in pmap %p", va, pmap);
3672                 return (FALSE);
3673         }
3674         /*
3675          * Increment counters.
3676          */
3677         pmap->pm_stats.resident_count += L2_PTE_NUM_TOTAL;
3678         /*
3679          * Despite permissions, mark the superpage read-only.
3680          */
3681         prot &= ~VM_PROT_WRITE;
3682         /*
3683          * Map the superpage.
3684          */
3685         pmap_map_section(pmap, va, pa, prot, FALSE);
3686
3687         pmap_section_mappings++;
3688         CTR2(KTR_PMAP, "pmap_enter_section: success for va %#lx"
3689             " in pmap %p", va, pmap);
3690         return (TRUE);
3691 }
3692
3693 /*
3694  * pmap_remove_section: do the things to unmap a superpage in a process
3695  */
3696 static void
3697 pmap_remove_section(pmap_t pmap, vm_offset_t sva)
3698 {
3699         struct md_page *pvh;
3700         struct l2_bucket *l2b;
3701         pd_entry_t *pl1pd, l1pd;
3702         vm_offset_t eva, va;
3703         vm_page_t m;
3704
3705         PMAP_ASSERT_LOCKED(pmap);
3706         if ((pmap == pmap_kernel()) || (L1_IDX(sva) == L1_IDX(vector_page)))
3707                 return;
3708
3709         KASSERT((sva & L1_S_OFFSET) == 0,
3710             ("pmap_remove_section: sva is not 1mpage aligned"));
3711
3712         pl1pd = &pmap->pm_l1->l1_kva[L1_IDX(sva)];
3713         l1pd = *pl1pd;
3714
3715         m = PHYS_TO_VM_PAGE(l1pd & L1_S_FRAME);
3716         KASSERT((m != NULL && ((m->oflags & VPO_UNMANAGED) == 0)),
3717             ("pmap_remove_section: no corresponding vm_page or "
3718             "page unmanaged"));
3719
3720         pmap->pm_stats.resident_count -= L2_PTE_NUM_TOTAL;
3721         pvh = pa_to_pvh(l1pd & L1_S_FRAME);
3722         pmap_pvh_free(pvh, pmap, sva);
3723         eva = L2_NEXT_BUCKET(sva);
3724         for (va = sva, m = PHYS_TO_VM_PAGE(l1pd & L1_S_FRAME);
3725             va < eva; va += PAGE_SIZE, m++) {
3726                 /*
3727                  * Mark base pages referenced but skip marking them dirty.
3728                  * If the superpage is writeable, hence all base pages were
3729                  * already marked as dirty in pmap_fault_fixup() before
3730                  * promotion. Reference bit however, might not have been set
3731                  * for each base page when the superpage was created at once,
3732                  * not as a result of promotion.
3733                  */
3734                 if (L1_S_REFERENCED(l1pd))
3735                         vm_page_aflag_set(m, PGA_REFERENCED);
3736                 if (TAILQ_EMPTY(&m->md.pv_list) &&
3737                     TAILQ_EMPTY(&pvh->pv_list))
3738                         vm_page_aflag_clear(m, PGA_WRITEABLE);
3739         }
3740
3741         l2b = pmap_get_l2_bucket(pmap, sva);
3742         if (l2b != NULL) {
3743                 KASSERT(l2b->l2b_occupancy == L2_PTE_NUM_TOTAL,
3744                     ("pmap_remove_section: l2_bucket occupancy error"));
3745                 pmap_free_l2_bucket(pmap, l2b, L2_PTE_NUM_TOTAL);
3746         }
3747         /* Now invalidate L1 slot */
3748         *pl1pd = 0;
3749         PTE_SYNC(pl1pd);
3750         if (L1_S_EXECUTABLE(l1pd))
3751                 cpu_tlb_flushID_SE(sva);
3752         else
3753                 cpu_tlb_flushD_SE(sva);
3754         cpu_cpwait();
3755 }
3756
3757 /*
3758  * Tries to promote the 256, contiguous 4KB page mappings that are
3759  * within a single l2_bucket to a single 1MB section mapping.
3760  * For promotion to occur, two conditions must be met: (1) the 4KB page
3761  * mappings must map aligned, contiguous physical memory and (2) the 4KB page
3762  * mappings must have identical characteristics.
3763  */
3764 static void
3765 pmap_promote_section(pmap_t pmap, vm_offset_t va)
3766 {
3767         pt_entry_t *firstptep, firstpte, oldpte, pa, *pte;
3768         vm_page_t m, oldm;
3769         vm_offset_t first_va, old_va;
3770         struct l2_bucket *l2b = NULL;
3771         vm_prot_t prot;
3772         struct pv_entry *pve, *first_pve;
3773
3774         PMAP_ASSERT_LOCKED(pmap);
3775
3776         prot = VM_PROT_ALL;
3777         /*
3778          * Skip promoting kernel pages. This is justified by following:
3779          * 1. Kernel is already mapped using section mappings in each pmap
3780          * 2. Managed mappings within the kernel are not to be promoted anyway
3781          */
3782         if (pmap == pmap_kernel()) {
3783                 pmap_section_p_failures++;
3784                 CTR2(KTR_PMAP, "pmap_promote_section: failure for va %#x"
3785                     " in pmap %p", va, pmap);
3786                 return;
3787         }
3788         /* Do not attemp to promote vectors pages */
3789         if (L1_IDX(va) == L1_IDX(vector_page)) {
3790                 pmap_section_p_failures++;
3791                 CTR2(KTR_PMAP, "pmap_promote_section: failure for va %#x"
3792                     " in pmap %p", va, pmap);
3793                 return;
3794         }
3795         /*
3796          * Examine the first PTE in the specified l2_bucket. Abort if this PTE
3797          * is either invalid, unused, or does not map the first 4KB physical
3798          * page within 1MB page.
3799          */
3800         first_va = trunc_1mpage(va);
3801         l2b = pmap_get_l2_bucket(pmap, first_va);
3802         KASSERT(l2b != NULL, ("pmap_promote_section: trying to promote "
3803             "not existing l2 bucket"));
3804         firstptep = &l2b->l2b_kva[0];
3805
3806         firstpte = *firstptep;
3807         if ((l2pte_pa(firstpte) & L1_S_OFFSET) != 0) {
3808                 pmap_section_p_failures++;
3809                 CTR2(KTR_PMAP, "pmap_promote_section: failure for va %#x"
3810                     " in pmap %p", va, pmap);
3811                 return;
3812         }
3813
3814         if ((firstpte & (L2_S_PROTO | L2_S_REF)) != (L2_S_PROTO | L2_S_REF)) {
3815                 pmap_section_p_failures++;
3816                 CTR2(KTR_PMAP, "pmap_promote_section: failure for va %#x"
3817                     " in pmap %p", va, pmap);
3818                 return;
3819         }
3820         /*
3821          * ARM uses pv_entry to mark particular mapping WIRED so don't promote
3822          * unmanaged pages since it is impossible to determine, whether the
3823          * page is wired or not if there is no corresponding pv_entry.
3824          */
3825         m = PHYS_TO_VM_PAGE(l2pte_pa(firstpte));
3826         if (m && ((m->oflags & VPO_UNMANAGED) != 0)) {
3827                 pmap_section_p_failures++;
3828                 CTR2(KTR_PMAP, "pmap_promote_section: failure for va %#x"
3829                     " in pmap %p", va, pmap);
3830                 return;
3831         }
3832         first_pve = pmap_find_pv(&m->md, pmap, first_va);
3833         /*
3834          * PTE is modified only on write due to modified bit
3835          * emulation. If the entry is referenced and writable
3836          * then it is modified and we don't clear write enable.
3837          * Otherwise, writing is disabled in PTE anyway and
3838          * we just configure protections for the section mapping
3839          * that is going to be created.
3840          */
3841         if ((first_pve->pv_flags & PVF_WRITE) != 0) {
3842                 if (!L2_S_WRITABLE(firstpte)) {
3843                         first_pve->pv_flags &= ~PVF_WRITE;
3844                         prot &= ~VM_PROT_WRITE;
3845                 }
3846         } else
3847                 prot &= ~VM_PROT_WRITE;
3848
3849         if (!L2_S_EXECUTABLE(firstpte))
3850                 prot &= ~VM_PROT_EXECUTE;
3851
3852         /*
3853          * Examine each of the other PTEs in the specified l2_bucket.
3854          * Abort if this PTE maps an unexpected 4KB physical page or
3855          * does not have identical characteristics to the first PTE.
3856          */
3857         pa = l2pte_pa(firstpte) + ((L2_PTE_NUM_TOTAL - 1) * PAGE_SIZE);
3858         old_va = L2_NEXT_BUCKET(first_va) - PAGE_SIZE;
3859
3860         for (pte = (firstptep + L2_PTE_NUM_TOTAL - 1); pte > firstptep; pte--) {
3861                 oldpte = *pte;
3862                 if (l2pte_pa(oldpte) != pa) {
3863                         pmap_section_p_failures++;
3864                         CTR2(KTR_PMAP, "pmap_promote_section: failure for "
3865                             "va %#x in pmap %p", va, pmap);
3866                         return;
3867                 }
3868                 if ((oldpte & L2_S_PROMOTE) != (firstpte & L2_S_PROMOTE)) {
3869                         pmap_section_p_failures++;
3870                         CTR2(KTR_PMAP, "pmap_promote_section: failure for "
3871                             "va %#x in pmap %p", va, pmap);
3872                         return;
3873                 }
3874                 oldm = PHYS_TO_VM_PAGE(l2pte_pa(oldpte));
3875                 if (oldm && ((oldm->oflags & VPO_UNMANAGED) != 0)) {
3876                         pmap_section_p_failures++;
3877                         CTR2(KTR_PMAP, "pmap_promote_section: failure for "
3878                             "va %#x in pmap %p", va, pmap);
3879                         return;
3880                 }
3881
3882                 pve = pmap_find_pv(&oldm->md, pmap, old_va);
3883                 if (pve == NULL) {
3884                         pmap_section_p_failures++;
3885                         CTR2(KTR_PMAP, "pmap_promote_section: failure for "
3886                             "va %#x old_va  %x - no pve", va, old_va);
3887                         return;
3888                 }
3889
3890                 if (!L2_S_WRITABLE(oldpte) && (pve->pv_flags & PVF_WRITE))
3891                         pve->pv_flags &= ~PVF_WRITE;
3892                 if (pve->pv_flags != first_pve->pv_flags) {
3893                         pmap_section_p_failures++;
3894                         CTR2(KTR_PMAP, "pmap_promote_section: failure for "
3895                             "va %#x in pmap %p", va, pmap);
3896                         return;
3897                 }
3898
3899                 old_va -= PAGE_SIZE;
3900                 pa -= PAGE_SIZE;
3901         }
3902         /*
3903          * Promote the pv entries.
3904          */
3905         pmap_pv_promote_section(pmap, first_va, l2pte_pa(firstpte));
3906         /*
3907          * Map the superpage.
3908          */
3909         pmap_map_section(pmap, first_va, l2pte_pa(firstpte), prot, TRUE);
3910         /*
3911          * Invalidate all possible TLB mappings for small
3912          * pages within the newly created superpage.
3913          * Rely on the first PTE's attributes since they
3914          * have to be consistent across all of the base pages
3915          * within the superpage. If page is not executable it
3916          * is at least referenced.
3917          * The fastest way to do that is to invalidate whole
3918          * TLB at once instead of executing 256 CP15 TLB
3919          * invalidations by single entry. TLBs usually maintain
3920          * several dozen entries so loss of unrelated entries is
3921          * still a less agresive approach.
3922          */
3923         if (L2_S_EXECUTABLE(firstpte))
3924                 cpu_tlb_flushID();
3925         else
3926                 cpu_tlb_flushD();
3927         cpu_cpwait();
3928
3929         pmap_section_promotions++;
3930         CTR2(KTR_PMAP, "pmap_promote_section: success for va %#x"
3931             " in pmap %p", first_va, pmap);
3932 }
3933
3934 /*
3935  * Fills a l2_bucket with mappings to consecutive physical pages.
3936  */
3937 static void
3938 pmap_fill_l2b(struct l2_bucket *l2b, pt_entry_t newpte)
3939 {
3940         pt_entry_t *ptep;
3941         int i;
3942
3943         for (i = 0; i < L2_PTE_NUM_TOTAL; i++) {
3944                 ptep = &l2b->l2b_kva[i];
3945                 *ptep = newpte;
3946                 PTE_SYNC(ptep);
3947
3948                 newpte += PAGE_SIZE;
3949         }
3950
3951         l2b->l2b_occupancy = L2_PTE_NUM_TOTAL;
3952 }
3953
3954 /*
3955  * Tries to demote a 1MB section mapping. If demotion fails, the
3956  * 1MB section mapping is invalidated.
3957  */
3958 static boolean_t
3959 pmap_demote_section(pmap_t pmap, vm_offset_t va)
3960 {
3961         struct l2_bucket *l2b;
3962         struct pv_entry *l1pdpve;
3963         struct md_page *pvh;
3964         pd_entry_t *pl1pd, l1pd, newl1pd;
3965         pt_entry_t *firstptep, newpte;
3966         vm_offset_t pa;
3967         vm_page_t m;
3968
3969         PMAP_ASSERT_LOCKED(pmap);
3970         /*
3971          * According to assumptions described in pmap_promote_section,
3972          * kernel is and always should be mapped using 1MB section mappings.
3973          * What more, managed kernel pages were not to be promoted.
3974          */
3975         KASSERT(pmap != pmap_kernel() && L1_IDX(va) != L1_IDX(vector_page),
3976             ("pmap_demote_section: forbidden section mapping"));
3977
3978         va = trunc_1mpage(va);
3979         pl1pd = &pmap->pm_l1->l1_kva[L1_IDX(va)];
3980         l1pd = *pl1pd;
3981         KASSERT((l1pd & L1_TYPE_MASK) == L1_S_PROTO,
3982             ("pmap_demote_section: not section or invalid section"));
3983
3984         pa = l1pd & L1_S_FRAME;
3985         m = PHYS_TO_VM_PAGE(pa);
3986         KASSERT((m != NULL && (m->oflags & VPO_UNMANAGED) == 0),
3987             ("pmap_demote_section: no vm_page for selected superpage or"
3988              "unmanaged"));
3989
3990         pvh = pa_to_pvh(pa);
3991         l1pdpve = pmap_find_pv(pvh, pmap, va);
3992         KASSERT(l1pdpve != NULL, ("pmap_demote_section: no pv entry for "
3993             "managed page"));
3994
3995         l2b = pmap_get_l2_bucket(pmap, va);
3996         if (l2b == NULL) {
3997                 KASSERT((l1pdpve->pv_flags & PVF_WIRED) == 0,
3998                     ("pmap_demote_section: No l2_bucket for wired mapping"));
3999                 /*
4000                  * Invalidate the 1MB section mapping and return
4001                  * "failure" if the mapping was never accessed or the
4002                  * allocation of the new l2_bucket fails.
4003                  */
4004                 if (!L1_S_REFERENCED(l1pd) ||
4005                     (l2b = pmap_alloc_l2_bucket(pmap, va)) == NULL) {
4006                         /* Unmap and invalidate superpage. */
4007                         pmap_remove_section(pmap, trunc_1mpage(va));
4008                         CTR2(KTR_PMAP, "pmap_demote_section: failure for "
4009                             "va %#x in pmap %p", va, pmap);
4010                         return (FALSE);
4011                 }
4012         }
4013
4014         /*
4015          * Now we should have corresponding l2_bucket available.
4016          * Let's process it to recreate 256 PTEs for each base page
4017          * within superpage.
4018          */
4019         newpte = pa | L1_S_DEMOTE(l1pd);
4020         if (m->md.pv_memattr != VM_MEMATTR_UNCACHEABLE)
4021                 newpte |= pte_l2_s_cache_mode;
4022
4023         /*
4024          * If the l2_bucket is new, initialize it.
4025          */
4026         if (l2b->l2b_occupancy == 0)
4027                 pmap_fill_l2b(l2b, newpte);
4028         else {
4029                 firstptep = &l2b->l2b_kva[0];
4030                 KASSERT(l2pte_pa(*firstptep) == (pa),
4031                     ("pmap_demote_section: firstpte and newpte map different "
4032                      "physical addresses"));
4033                 /*
4034                  * If the mapping has changed attributes, update the page table
4035                  * entries.
4036                  */
4037                 if ((*firstptep & L2_S_PROMOTE) != (L1_S_DEMOTE(l1pd)))
4038                         pmap_fill_l2b(l2b, newpte);
4039         }
4040         /* Demote PV entry */
4041         pmap_pv_demote_section(pmap, va, pa);
4042
4043         /* Now fix-up L1 */
4044         newl1pd = l2b->l2b_phys | L1_C_DOM(pmap->pm_domain) | L1_C_PROTO;
4045         *pl1pd = newl1pd;
4046         PTE_SYNC(pl1pd);
4047         /* Invalidate old TLB mapping */
4048         if (L1_S_EXECUTABLE(l1pd))
4049                 cpu_tlb_flushID_SE(va);
4050         else if (L1_S_REFERENCED(l1pd))
4051                 cpu_tlb_flushD_SE(va);
4052         cpu_cpwait();
4053
4054         pmap_section_demotions++;
4055         CTR2(KTR_PMAP, "pmap_demote_section: success for va %#x"
4056             " in pmap %p", va, pmap);
4057         return (TRUE);
4058 }
4059
4060 /***************************************************
4061  * page management routines.
4062  ***************************************************/
4063
4064 /*
4065  * We are in a serious low memory condition.  Resort to
4066  * drastic measures to free some pages so we can allocate
4067  * another pv entry chunk.
4068  */
4069 static vm_page_t
4070 pmap_pv_reclaim(pmap_t locked_pmap)
4071 {
4072         struct pch newtail;
4073         struct pv_chunk *pc;
4074         struct l2_bucket *l2b = NULL;
4075         pmap_t pmap;
4076         pd_entry_t *pl1pd;
4077         pt_entry_t *ptep;
4078         pv_entry_t pv;
4079         vm_offset_t va;
4080         vm_page_t free, m, m_pc;
4081         uint32_t inuse;
4082         int bit, field, freed, idx;
4083
4084         PMAP_ASSERT_LOCKED(locked_pmap);
4085         pmap = NULL;
4086         free = m_pc = NULL;
4087         TAILQ_INIT(&newtail);
4088         while ((pc = TAILQ_FIRST(&pv_chunks)) != NULL && (pv_vafree == 0 ||
4089             free == NULL)) {
4090                 TAILQ_REMOVE(&pv_chunks, pc, pc_lru);
4091                 if (pmap != pc->pc_pmap) {
4092                         if (pmap != NULL) {
4093                                 cpu_tlb_flushID();
4094                                 cpu_cpwait();
4095                                 if (pmap != locked_pmap)
4096                                         PMAP_UNLOCK(pmap);
4097                         }
4098                         pmap = pc->pc_pmap;
4099                         /* Avoid deadlock and lock recursion. */
4100                         if (pmap > locked_pmap)
4101                                 PMAP_LOCK(pmap);
4102                         else if (pmap != locked_pmap && !PMAP_TRYLOCK(pmap)) {
4103                                 pmap = NULL;
4104                                 TAILQ_INSERT_TAIL(&newtail, pc, pc_lru);
4105                                 continue;
4106                         }
4107                 }
4108
4109                 /*
4110                  * Destroy every non-wired, 4 KB page mapping in the chunk.
4111                  */
4112                 freed = 0;
4113                 for (field = 0; field < _NPCM; field++) {
4114                         for (inuse = ~pc->pc_map[field] & pc_freemask[field];
4115                             inuse != 0; inuse &= ~(1UL << bit)) {
4116                                 bit = ffs(inuse) - 1;
4117                                 idx = field * sizeof(inuse) * NBBY + bit;
4118                                 pv = &pc->pc_pventry[idx];
4119                                 va = pv->pv_va;
4120
4121                                 pl1pd = &pmap->pm_l1->l1_kva[L1_IDX(va)];
4122                                 if ((*pl1pd & L1_TYPE_MASK) == L1_S_PROTO)
4123                                         continue;
4124                                 if (pv->pv_flags & PVF_WIRED)
4125                                         continue;
4126
4127                                 l2b = pmap_get_l2_bucket(pmap, va);
4128                                 KASSERT(l2b != NULL, ("No l2 bucket"));
4129                                 ptep = &l2b->l2b_kva[l2pte_index(va)];
4130                                 m = PHYS_TO_VM_PAGE(l2pte_pa(*ptep));
4131                                 KASSERT((vm_offset_t)m >= KERNBASE,
4132                                     ("Trying to access non-existent page "
4133                                      "va %x pte %x", va, *ptep));
4134                                 *ptep = 0;
4135                                 PTE_SYNC(ptep);
4136                                 TAILQ_REMOVE(&m->md.pv_list, pv, pv_list);
4137                                 if (TAILQ_EMPTY(&m->md.pv_list))
4138                                         vm_page_aflag_clear(m, PGA_WRITEABLE);
4139                                 pc->pc_map[field] |= 1UL << bit;
4140                                 freed++;
4141                         }
4142                 }
4143
4144                 if (freed == 0) {
4145                         TAILQ_INSERT_TAIL(&newtail, pc, pc_lru);
4146                         continue;
4147                 }
4148                 /* Every freed mapping is for a 4 KB page. */
4149                 pmap->pm_stats.resident_count -= freed;
4150                 PV_STAT(pv_entry_frees += freed);
4151                 PV_STAT(pv_entry_spare += freed);
4152                 pv_entry_count -= freed;
4153                 TAILQ_REMOVE(&pmap->pm_pvchunk, pc, pc_list);
4154                 for (field = 0; field < _NPCM; field++)
4155                         if (pc->pc_map[field] != pc_freemask[field]) {
4156                                 TAILQ_INSERT_HEAD(&pmap->pm_pvchunk, pc,
4157                                     pc_list);
4158                                 TAILQ_INSERT_TAIL(&newtail, pc, pc_lru);
4159
4160                                 /*
4161                                  * One freed pv entry in locked_pmap is
4162                                  * sufficient.
4163                                  */
4164                                 if (pmap == locked_pmap)
4165                                         goto out;
4166                                 break;
4167                         }
4168                 if (field == _NPCM) {
4169                         PV_STAT(pv_entry_spare -= _NPCPV);
4170                         PV_STAT(pc_chunk_count--);
4171                         PV_STAT(pc_chunk_frees++);
4172                         /* Entire chunk is free; return it. */
4173                         m_pc = PHYS_TO_VM_PAGE(pmap_kextract((vm_offset_t)pc));
4174                         pmap_qremove((vm_offset_t)pc, 1);
4175                         pmap_ptelist_free(&pv_vafree, (vm_offset_t)pc);
4176                         break;
4177                 }
4178         }
4179 out:
4180         TAILQ_CONCAT(&pv_chunks, &newtail, pc_lru);
4181         if (pmap != NULL) {
4182                 cpu_tlb_flushID();
4183                 cpu_cpwait();
4184                 if (pmap != locked_pmap)
4185                         PMAP_UNLOCK(pmap);
4186         }
4187         return (m_pc);
4188 }
4189
4190 /*
4191  * free the pv_entry back to the free list
4192  */
4193 static void
4194 pmap_free_pv_entry(pmap_t pmap, pv_entry_t pv)
4195 {
4196         struct pv_chunk *pc;
4197         int bit, field, idx;
4198
4199         rw_assert(&pvh_global_lock, RA_WLOCKED);
4200         PMAP_ASSERT_LOCKED(pmap);
4201         PV_STAT(pv_entry_frees++);
4202         PV_STAT(pv_entry_spare++);
4203         pv_entry_count--;
4204         pc = pv_to_chunk(pv);
4205         idx = pv - &pc->pc_pventry[0];
4206         field = idx / (sizeof(u_long) * NBBY);
4207         bit = idx % (sizeof(u_long) * NBBY);
4208         pc->pc_map[field] |= 1ul << bit;
4209         for (idx = 0; idx < _NPCM; idx++)
4210                 if (pc->pc_map[idx] != pc_freemask[idx]) {
4211                         /*
4212                          * 98% of the time, pc is already at the head of the
4213                          * list.  If it isn't already, move it to the head.
4214                          */
4215                         if (__predict_false(TAILQ_FIRST(&pmap->pm_pvchunk) !=
4216                             pc)) {
4217                                 TAILQ_REMOVE(&pmap->pm_pvchunk, pc, pc_list);
4218                                 TAILQ_INSERT_HEAD(&pmap->pm_pvchunk, pc,
4219                                     pc_list);
4220                         }
4221                         return;
4222                 }
4223         TAILQ_REMOVE(&pmap->pm_pvchunk, pc, pc_list);
4224         pmap_free_pv_chunk(pc);
4225 }
4226
4227 static void
4228 pmap_free_pv_chunk(struct pv_chunk *pc)
4229 {
4230         vm_page_t m;
4231
4232         TAILQ_REMOVE(&pv_chunks, pc, pc_lru);
4233         PV_STAT(pv_entry_spare -= _NPCPV);
4234         PV_STAT(pc_chunk_count--);
4235         PV_STAT(pc_chunk_frees++);
4236         /* entire chunk is free, return it */
4237         m = PHYS_TO_VM_PAGE(pmap_kextract((vm_offset_t)pc));
4238         pmap_qremove((vm_offset_t)pc, 1);
4239         vm_page_unwire(m, PQ_INACTIVE);
4240         vm_page_free(m);
4241         pmap_ptelist_free(&pv_vafree, (vm_offset_t)pc);
4242
4243 }
4244
4245 static pv_entry_t
4246 pmap_get_pv_entry(pmap_t pmap, boolean_t try)
4247 {
4248         static const struct timeval printinterval = { 60, 0 };
4249         static struct timeval lastprint;
4250         struct pv_chunk *pc;
4251         pv_entry_t pv;
4252         vm_page_t m;
4253         int bit, field, idx;
4254
4255         rw_assert(&pvh_global_lock, RA_WLOCKED);
4256         PMAP_ASSERT_LOCKED(pmap);
4257         PV_STAT(pv_entry_allocs++);
4258         pv_entry_count++;
4259
4260         if (pv_entry_count > pv_entry_high_water)
4261                 if (ratecheck(&lastprint, &printinterval))
4262                         printf("%s: Approaching the limit on PV entries.\n",
4263                             __func__);
4264 retry:
4265         pc = TAILQ_FIRST(&pmap->pm_pvchunk);
4266         if (pc != NULL) {
4267                 for (field = 0; field < _NPCM; field++) {
4268                         if (pc->pc_map[field]) {
4269                                 bit = ffs(pc->pc_map[field]) - 1;
4270                                 break;
4271                         }
4272                 }
4273                 if (field < _NPCM) {
4274                         idx = field * sizeof(pc->pc_map[field]) * NBBY + bit;
4275                         pv = &pc->pc_pventry[idx];
4276                         pc->pc_map[field] &= ~(1ul << bit);
4277                         /* If this was the last item, move it to tail */
4278                         for (field = 0; field < _NPCM; field++)
4279                                 if (pc->pc_map[field] != 0) {
4280                                         PV_STAT(pv_entry_spare--);
4281                                         return (pv);    /* not full, return */
4282                                 }
4283                         TAILQ_REMOVE(&pmap->pm_pvchunk, pc, pc_list);
4284                         TAILQ_INSERT_TAIL(&pmap->pm_pvchunk, pc, pc_list);
4285                         PV_STAT(pv_entry_spare--);
4286                         return (pv);
4287                 }
4288         }
4289         /*
4290          * Access to the ptelist "pv_vafree" is synchronized by the pvh
4291          * global lock.  If "pv_vafree" is currently non-empty, it will
4292          * remain non-empty until pmap_ptelist_alloc() completes.
4293          */
4294         if (pv_vafree == 0 || (m = vm_page_alloc(NULL, 0, VM_ALLOC_NORMAL |
4295             VM_ALLOC_NOOBJ | VM_ALLOC_WIRED)) == NULL) {
4296                 if (try) {
4297                         pv_entry_count--;
4298                         PV_STAT(pc_chunk_tryfail++);
4299                         return (NULL);
4300                 }
4301                 m = pmap_pv_reclaim(pmap);
4302                 if (m == NULL)
4303                         goto retry;
4304         }
4305         PV_STAT(pc_chunk_count++);
4306         PV_STAT(pc_chunk_allocs++);
4307         pc = (struct pv_chunk *)pmap_ptelist_alloc(&pv_vafree);
4308         pmap_qenter((vm_offset_t)pc, &m, 1);
4309         pc->pc_pmap = pmap;
4310         pc->pc_map[0] = pc_freemask[0] & ~1ul;  /* preallocated bit 0 */
4311         for (field = 1; field < _NPCM; field++)
4312                 pc->pc_map[field] = pc_freemask[field];
4313         TAILQ_INSERT_TAIL(&pv_chunks, pc, pc_lru);
4314         pv = &pc->pc_pventry[0];
4315         TAILQ_INSERT_HEAD(&pmap->pm_pvchunk, pc, pc_list);
4316         PV_STAT(pv_entry_spare += _NPCPV - 1);
4317         return (pv);
4318 }
4319
4320 /*
4321  *      Remove the given range of addresses from the specified map.
4322  *
4323  *      It is assumed that the start and end are properly
4324  *      rounded to the page size.
4325  */
4326 #define PMAP_REMOVE_CLEAN_LIST_SIZE     3
4327 void
4328 pmap_remove(pmap_t pmap, vm_offset_t sva, vm_offset_t eva)
4329 {
4330         struct l2_bucket *l2b;
4331         vm_offset_t next_bucket;
4332         pd_entry_t l1pd;
4333         pt_entry_t *ptep;
4334         u_int total;
4335         u_int mappings, is_exec, is_refd;
4336         int flushall = 0;
4337
4338
4339         /*
4340          * we lock in the pmap => pv_head direction
4341          */
4342
4343         rw_wlock(&pvh_global_lock);
4344         PMAP_LOCK(pmap);
4345         total = 0;
4346         while (sva < eva) {
4347                 next_bucket = L2_NEXT_BUCKET(sva);
4348
4349                 /*
4350                  * Check for large page.
4351                  */
4352                 l1pd = pmap->pm_l1->l1_kva[L1_IDX(sva)];
4353                 if ((l1pd & L1_TYPE_MASK) == L1_S_PROTO) {
4354                         KASSERT((l1pd & L1_S_DOM_MASK) !=
4355                             L1_S_DOM(PMAP_DOMAIN_KERNEL), ("pmap_remove: "
4356                             "Trying to remove kernel section mapping"));
4357                         /*
4358                          * Are we removing the entire large page?  If not,
4359                          * demote the mapping and fall through.
4360                          */
4361                         if (sva + L1_S_SIZE == next_bucket &&
4362                             eva >= next_bucket) {
4363                                 pmap_remove_section(pmap, sva);
4364                                 sva = next_bucket;
4365                                 continue;
4366                         } else if (!pmap_demote_section(pmap, sva)) {
4367                                 /* The large page mapping was destroyed. */
4368                                 sva = next_bucket;
4369                                 continue;
4370                         }
4371                 }
4372                 /*
4373                  * Do one L2 bucket's worth at a time.
4374                  */
4375                 if (next_bucket > eva)
4376                         next_bucket = eva;
4377
4378                 l2b = pmap_get_l2_bucket(pmap, sva);
4379                 if (l2b == NULL) {
4380                         sva = next_bucket;
4381                         continue;
4382                 }
4383
4384                 ptep = &l2b->l2b_kva[l2pte_index(sva)];
4385                 mappings = 0;
4386
4387                 while (sva < next_bucket) {
4388                         struct vm_page *m;
4389                         pt_entry_t pte;
4390                         vm_paddr_t pa;
4391
4392                         pte = *ptep;
4393
4394                         if (pte == 0) {
4395                                 /*
4396                                  * Nothing here, move along
4397                                  */
4398                                 sva += PAGE_SIZE;
4399                                 ptep++;
4400                                 continue;
4401                         }
4402
4403                         pmap->pm_stats.resident_count--;
4404                         pa = l2pte_pa(pte);
4405                         is_exec = 0;
4406                         is_refd = 1;
4407
4408                         /*
4409                          * Update flags. In a number of circumstances,
4410                          * we could cluster a lot of these and do a
4411                          * number of sequential pages in one go.
4412                          */
4413                         if ((m = PHYS_TO_VM_PAGE(pa)) != NULL) {
4414                                 struct pv_entry *pve;
4415
4416                                 pve = pmap_remove_pv(m, pmap, sva);
4417                                 if (pve) {
4418                                         is_exec = PTE_BEEN_EXECD(pte);
4419                                         is_refd = PTE_BEEN_REFD(pte);
4420                                         pmap_free_pv_entry(pmap, pve);
4421                                 }
4422                         }
4423
4424                         *ptep = 0;
4425                         PTE_SYNC(ptep);
4426                         if (pmap_is_current(pmap)) {
4427                                 total++;
4428                                 if (total < PMAP_REMOVE_CLEAN_LIST_SIZE) {
4429                                         if (is_exec)
4430                                                 cpu_tlb_flushID_SE(sva);
4431                                         else if (is_refd)
4432                                                 cpu_tlb_flushD_SE(sva);
4433                                 } else if (total == PMAP_REMOVE_CLEAN_LIST_SIZE)
4434                                         flushall = 1;
4435                         }
4436
4437                         sva += PAGE_SIZE;
4438                         ptep++;
4439                         mappings++;
4440                 }
4441
4442                 pmap_free_l2_bucket(pmap, l2b, mappings);
4443         }
4444
4445         rw_wunlock(&pvh_global_lock);
4446         if (flushall)
4447                 cpu_tlb_flushID();
4448         cpu_cpwait();
4449
4450         PMAP_UNLOCK(pmap);
4451 }
4452
4453 /*
4454  * pmap_zero_page()
4455  *
4456  * Zero a given physical page by mapping it at a page hook point.
4457  * In doing the zero page op, the page we zero is mapped cachable, as with
4458  * StrongARM accesses to non-cached pages are non-burst making writing
4459  * _any_ bulk data very slow.
4460  */
4461 static void
4462 pmap_zero_page_gen(vm_page_t m, int off, int size)
4463 {
4464         struct czpages *czp;
4465
4466         KASSERT(TAILQ_EMPTY(&m->md.pv_list),
4467             ("pmap_zero_page_gen: page has mappings"));
4468
4469         vm_paddr_t phys = VM_PAGE_TO_PHYS(m);
4470
4471         sched_pin();
4472         czp = &cpu_czpages[PCPU_GET(cpuid)];
4473         mtx_lock(&czp->lock);
4474
4475         /*
4476          * Hook in the page, zero it.
4477          */
4478         *czp->dstptep = L2_S_PROTO | phys | pte_l2_s_cache_mode | L2_S_REF;
4479         pmap_set_prot(czp->dstptep, VM_PROT_WRITE, 0);
4480         PTE_SYNC(czp->dstptep);
4481         cpu_tlb_flushD_SE(czp->dstva);
4482         cpu_cpwait();
4483
4484         if (off || size != PAGE_SIZE)
4485                 bzero((void *)(czp->dstva + off), size);
4486         else
4487                 bzero_page(czp->dstva);
4488
4489         /*
4490          * Although aliasing is not possible, if we use temporary mappings with
4491          * memory that will be mapped later as non-cached or with write-through
4492          * caches, we might end up overwriting it when calling wbinv_all.  So
4493          * make sure caches are clean after the operation.
4494          */
4495         cpu_idcache_wbinv_range(czp->dstva, size);
4496         pmap_l2cache_wbinv_range(czp->dstva, phys, size);
4497
4498         mtx_unlock(&czp->lock);
4499         sched_unpin();
4500 }
4501
4502 /*
4503  *      pmap_zero_page zeros the specified hardware page by mapping
4504  *      the page into KVM and using bzero to clear its contents.
4505  */
4506 void
4507 pmap_zero_page(vm_page_t m)
4508 {
4509         pmap_zero_page_gen(m, 0, PAGE_SIZE);
4510 }
4511
4512
4513 /*
4514  *      pmap_zero_page_area zeros the specified hardware page by mapping
4515  *      the page into KVM and using bzero to clear its contents.
4516  *
4517  *      off and size may not cover an area beyond a single hardware page.
4518  */
4519 void
4520 pmap_zero_page_area(vm_page_t m, int off, int size)
4521 {
4522
4523         pmap_zero_page_gen(m, off, size);
4524 }
4525
4526
4527 /*
4528  *      pmap_zero_page_idle zeros the specified hardware page by mapping
4529  *      the page into KVM and using bzero to clear its contents.  This
4530  *      is intended to be called from the vm_pagezero process only and
4531  *      outside of Giant.
4532  */
4533 void
4534 pmap_zero_page_idle(vm_page_t m)
4535 {
4536
4537         pmap_zero_page(m);
4538 }
4539
4540 /*
4541  *      pmap_copy_page copies the specified (machine independent)
4542  *      page by mapping the page into virtual memory and using
4543  *      bcopy to copy the page, one machine dependent page at a
4544  *      time.
4545  */
4546
4547 /*
4548  * pmap_copy_page()
4549  *
4550  * Copy one physical page into another, by mapping the pages into
4551  * hook points. The same comment regarding cachability as in
4552  * pmap_zero_page also applies here.
4553  */
4554 void
4555 pmap_copy_page_generic(vm_paddr_t src, vm_paddr_t dst)
4556 {
4557         struct czpages *czp;
4558
4559         sched_pin();
4560         czp = &cpu_czpages[PCPU_GET(cpuid)];
4561         mtx_lock(&czp->lock);
4562
4563         /*
4564          * Map the pages into the page hook points, copy them, and purge the
4565          * cache for the appropriate page.
4566          */
4567         *czp->srcptep = L2_S_PROTO | src | pte_l2_s_cache_mode | L2_S_REF;
4568         pmap_set_prot(czp->srcptep, VM_PROT_READ, 0);
4569         PTE_SYNC(czp->srcptep);
4570         cpu_tlb_flushD_SE(czp->srcva);
4571         *czp->dstptep = L2_S_PROTO | dst | pte_l2_s_cache_mode | L2_S_REF;
4572         pmap_set_prot(czp->dstptep, VM_PROT_READ | VM_PROT_WRITE, 0);
4573         PTE_SYNC(czp->dstptep);
4574         cpu_tlb_flushD_SE(czp->dstva);
4575         cpu_cpwait();
4576
4577         bcopy_page(czp->srcva, czp->dstva);
4578
4579         /*
4580          * Although aliasing is not possible, if we use temporary mappings with
4581          * memory that will be mapped later as non-cached or with write-through
4582          * caches, we might end up overwriting it when calling wbinv_all.  So
4583          * make sure caches are clean after the operation.
4584          */
4585         cpu_idcache_wbinv_range(czp->dstva, PAGE_SIZE);
4586         pmap_l2cache_wbinv_range(czp->dstva, dst, PAGE_SIZE);
4587
4588         mtx_unlock(&czp->lock);
4589         sched_unpin();
4590 }
4591
4592 int unmapped_buf_allowed = 1;
4593
4594 void
4595 pmap_copy_pages(vm_page_t ma[], vm_offset_t a_offset, vm_page_t mb[],
4596     vm_offset_t b_offset, int xfersize)
4597 {
4598         vm_page_t a_pg, b_pg;
4599         vm_offset_t a_pg_offset, b_pg_offset;
4600         int cnt;
4601         struct czpages *czp;
4602
4603         sched_pin();
4604         czp = &cpu_czpages[PCPU_GET(cpuid)];
4605         mtx_lock(&czp->lock);
4606
4607         while (xfersize > 0) {
4608                 a_pg = ma[a_offset >> PAGE_SHIFT];
4609                 a_pg_offset = a_offset & PAGE_MASK;
4610                 cnt = min(xfersize, PAGE_SIZE - a_pg_offset);
4611                 b_pg = mb[b_offset >> PAGE_SHIFT];
4612                 b_pg_offset = b_offset & PAGE_MASK;
4613                 cnt = min(cnt, PAGE_SIZE - b_pg_offset);
4614                 *czp->srcptep = L2_S_PROTO | VM_PAGE_TO_PHYS(a_pg) |
4615                     pte_l2_s_cache_mode | L2_S_REF;
4616                 pmap_set_prot(czp->srcptep, VM_PROT_READ, 0);
4617                 PTE_SYNC(czp->srcptep);
4618                 cpu_tlb_flushD_SE(czp->srcva);
4619                 *czp->dstptep = L2_S_PROTO | VM_PAGE_TO_PHYS(b_pg) |
4620                     pte_l2_s_cache_mode | L2_S_REF;
4621                 pmap_set_prot(czp->dstptep, VM_PROT_READ | VM_PROT_WRITE, 0);
4622                 PTE_SYNC(czp->dstptep);
4623                 cpu_tlb_flushD_SE(czp->dstva);
4624                 cpu_cpwait();
4625                 bcopy((char *)czp->srcva + a_pg_offset, (char *)czp->dstva + b_pg_offset,
4626                     cnt);
4627                 cpu_idcache_wbinv_range(czp->dstva + b_pg_offset, cnt);
4628                 pmap_l2cache_wbinv_range(czp->dstva + b_pg_offset,
4629                     VM_PAGE_TO_PHYS(b_pg) + b_pg_offset, cnt);
4630                 xfersize -= cnt;
4631                 a_offset += cnt;
4632                 b_offset += cnt;
4633         }
4634
4635         mtx_unlock(&czp->lock);
4636         sched_unpin();
4637 }
4638
4639 void
4640 pmap_copy_page(vm_page_t src, vm_page_t dst)
4641 {
4642
4643         if (_arm_memcpy && PAGE_SIZE >= _min_memcpy_size &&
4644             _arm_memcpy((void *)VM_PAGE_TO_PHYS(dst),
4645             (void *)VM_PAGE_TO_PHYS(src), PAGE_SIZE, IS_PHYSICAL) == 0)
4646                 return;
4647
4648         pmap_copy_page_generic(VM_PAGE_TO_PHYS(src), VM_PAGE_TO_PHYS(dst));
4649 }
4650
4651 /*
4652  * this routine returns true if a physical page resides
4653  * in the given pmap.
4654  */
4655 boolean_t
4656 pmap_page_exists_quick(pmap_t pmap, vm_page_t m)
4657 {
4658         struct md_page *pvh;
4659         pv_entry_t pv;
4660         int loops = 0;
4661         boolean_t rv;
4662
4663         KASSERT((m->oflags & VPO_UNMANAGED) == 0,
4664             ("pmap_page_exists_quick: page %p is not managed", m));
4665         rv = FALSE;
4666         rw_wlock(&pvh_global_lock);
4667         TAILQ_FOREACH(pv, &m->md.pv_list, pv_list) {
4668                 if (PV_PMAP(pv) == pmap) {
4669                         rv = TRUE;
4670                         break;
4671                 }
4672                 loops++;
4673                 if (loops >= 16)
4674                         break;
4675         }
4676         if (!rv && loops < 16 && (m->flags & PG_FICTITIOUS) == 0) {
4677                 pvh = pa_to_pvh(VM_PAGE_TO_PHYS(m));
4678                 TAILQ_FOREACH(pv, &pvh->pv_list, pv_list) {
4679                         if (PV_PMAP(pv) == pmap) {
4680                                 rv = TRUE;
4681                                 break;
4682                         }
4683                         loops++;
4684                         if (loops >= 16)
4685                                 break;
4686                 }
4687         }
4688         rw_wunlock(&pvh_global_lock);
4689         return (rv);
4690 }
4691
4692 /*
4693  *      pmap_page_wired_mappings:
4694  *
4695  *      Return the number of managed mappings to the given physical page
4696  *      that are wired.
4697  */
4698 int
4699 pmap_page_wired_mappings(vm_page_t m)
4700 {
4701         int count;
4702
4703         count = 0;
4704         if ((m->oflags & VPO_UNMANAGED) != 0)
4705                 return (count);
4706         rw_wlock(&pvh_global_lock);
4707         count = pmap_pvh_wired_mappings(&m->md, count);
4708         if ((m->flags & PG_FICTITIOUS) == 0) {
4709             count = pmap_pvh_wired_mappings(pa_to_pvh(VM_PAGE_TO_PHYS(m)),
4710                 count);
4711         }
4712         rw_wunlock(&pvh_global_lock);
4713         return (count);
4714 }
4715
4716 /*
4717  *      pmap_pvh_wired_mappings:
4718  *
4719  *      Return the updated number "count" of managed mappings that are wired.
4720  */
4721 static int
4722 pmap_pvh_wired_mappings(struct md_page *pvh, int count)
4723 {
4724         pv_entry_t pv;
4725
4726         rw_assert(&pvh_global_lock, RA_WLOCKED);
4727         TAILQ_FOREACH(pv, &pvh->pv_list, pv_list) {
4728                 if ((pv->pv_flags & PVF_WIRED) != 0)
4729                         count++;
4730         }
4731         return (count);
4732 }
4733
4734 /*
4735  * Returns TRUE if any of the given mappings were referenced and FALSE
4736  * otherwise.  Both page and section mappings are supported.
4737  */
4738 static boolean_t
4739 pmap_is_referenced_pvh(struct md_page *pvh)
4740 {
4741         struct l2_bucket *l2b;
4742         pv_entry_t pv;
4743         pd_entry_t *pl1pd;
4744         pt_entry_t *ptep;
4745         pmap_t pmap;
4746         boolean_t rv;
4747
4748         rw_assert(&pvh_global_lock, RA_WLOCKED);
4749         rv = FALSE;
4750         TAILQ_FOREACH(pv, &pvh->pv_list, pv_list) {
4751                 pmap = PV_PMAP(pv);
4752                 PMAP_LOCK(pmap);
4753                 pl1pd = &pmap->pm_l1->l1_kva[L1_IDX(pv->pv_va)];
4754                 if ((*pl1pd & L1_TYPE_MASK) == L1_S_PROTO)
4755                         rv = L1_S_REFERENCED(*pl1pd);
4756                 else {
4757                         l2b = pmap_get_l2_bucket(pmap, pv->pv_va);
4758                         ptep = &l2b->l2b_kva[l2pte_index(pv->pv_va)];
4759                         rv = L2_S_REFERENCED(*ptep);
4760                 }
4761                 PMAP_UNLOCK(pmap);
4762                 if (rv)
4763                         break;
4764         }
4765         return (rv);
4766 }
4767
4768 /*
4769  *      pmap_is_referenced:
4770  *
4771  *      Return whether or not the specified physical page was referenced
4772  *      in any physical maps.
4773  */
4774 boolean_t
4775 pmap_is_referenced(vm_page_t m)
4776 {
4777         boolean_t rv;
4778
4779         KASSERT((m->oflags & VPO_UNMANAGED) == 0,
4780             ("pmap_is_referenced: page %p is not managed", m));
4781         rw_wlock(&pvh_global_lock);
4782         rv = pmap_is_referenced_pvh(&m->md) ||
4783             ((m->flags & PG_FICTITIOUS) == 0 &&
4784             pmap_is_referenced_pvh(pa_to_pvh(VM_PAGE_TO_PHYS(m))));
4785         rw_wunlock(&pvh_global_lock);
4786         return (rv);
4787 }
4788
4789 /*
4790  *      pmap_ts_referenced:
4791  *
4792  *      Return the count of reference bits for a page, clearing all of them.
4793  */
4794 int
4795 pmap_ts_referenced(vm_page_t m)
4796 {
4797
4798         KASSERT((m->oflags & VPO_UNMANAGED) == 0,
4799             ("pmap_ts_referenced: page %p is not managed", m));
4800         return (pmap_clearbit(m, PVF_REF));
4801 }
4802
4803 /*
4804  * Returns TRUE if any of the given mappings were used to modify
4805  * physical memory. Otherwise, returns FALSE. Both page and 1MB section
4806  * mappings are supported.
4807  */
4808 static boolean_t
4809 pmap_is_modified_pvh(struct md_page *pvh)
4810 {
4811         pd_entry_t *pl1pd;
4812         struct l2_bucket *l2b;
4813         pv_entry_t pv;
4814         pt_entry_t *ptep;
4815         pmap_t pmap;
4816         boolean_t rv;
4817
4818         rw_assert(&pvh_global_lock, RA_WLOCKED);
4819         rv = FALSE;
4820
4821         TAILQ_FOREACH(pv, &pvh->pv_list, pv_list) {
4822                 pmap = PV_PMAP(pv);
4823                 PMAP_LOCK(pmap);
4824                 pl1pd = &pmap->pm_l1->l1_kva[L1_IDX(pv->pv_va)];
4825                 if ((*pl1pd & L1_TYPE_MASK) == L1_S_PROTO)
4826                         rv = L1_S_WRITABLE(*pl1pd);
4827                 else {
4828                         l2b = pmap_get_l2_bucket(pmap, pv->pv_va);
4829                         ptep = &l2b->l2b_kva[l2pte_index(pv->pv_va)];
4830                         rv = L2_S_WRITABLE(*ptep);
4831                 }
4832                 PMAP_UNLOCK(pmap);
4833                 if (rv)
4834                         break;
4835         }
4836
4837         return (rv);
4838 }
4839
4840 boolean_t
4841 pmap_is_modified(vm_page_t m)
4842 {
4843         boolean_t rv;
4844
4845         KASSERT((m->oflags & VPO_UNMANAGED) == 0,
4846             ("pmap_is_modified: page %p is not managed", m));
4847         /*
4848          * If the page is not exclusive busied, then PGA_WRITEABLE cannot be
4849          * concurrently set while the object is locked.  Thus, if PGA_WRITEABLE
4850          * is clear, no PTEs can have APX cleared.
4851          */
4852         VM_OBJECT_ASSERT_WLOCKED(m->object);
4853         if (!vm_page_xbusied(m) && (m->aflags & PGA_WRITEABLE) == 0)
4854                 return (FALSE);
4855         rw_wlock(&pvh_global_lock);
4856         rv = pmap_is_modified_pvh(&m->md) ||
4857             ((m->flags & PG_FICTITIOUS) == 0 &&
4858             pmap_is_modified_pvh(pa_to_pvh(VM_PAGE_TO_PHYS(m))));
4859         rw_wunlock(&pvh_global_lock);
4860         return (rv);
4861 }
4862
4863 /*
4864  *      Apply the given advice to the specified range of addresses within the
4865  *      given pmap.  Depending on the advice, clear the referenced and/or
4866  *      modified flags in each mapping.
4867  */
4868 void
4869 pmap_advise(pmap_t pmap, vm_offset_t sva, vm_offset_t eva, int advice)
4870 {
4871         struct l2_bucket *l2b;
4872         struct pv_entry *pve;
4873         pd_entry_t l1pd;
4874         pt_entry_t *ptep, opte, pte;
4875         vm_offset_t next_bucket;
4876         vm_page_t m;
4877
4878         if (advice != MADV_DONTNEED && advice != MADV_FREE)
4879                 return;
4880         rw_wlock(&pvh_global_lock);
4881         PMAP_LOCK(pmap);
4882         for (; sva < eva; sva = next_bucket) {
4883                 next_bucket = L2_NEXT_BUCKET(sva);
4884                 if (next_bucket < sva)
4885                         next_bucket = eva;
4886                 l1pd = pmap->pm_l1->l1_kva[L1_IDX(sva)];
4887                 if ((l1pd & L1_TYPE_MASK) == L1_S_PROTO) {
4888                         if (pmap == pmap_kernel())
4889                                 continue;
4890                         if (!pmap_demote_section(pmap, sva)) {
4891                                 /*
4892                                  * The large page mapping was destroyed.
4893                                  */
4894                                 continue;
4895                         }
4896                         /*
4897                          * Unless the page mappings are wired, remove the
4898                          * mapping to a single page so that a subsequent
4899                          * access may repromote. Since the underlying
4900                          * l2_bucket is fully populated, this removal
4901                          * never frees an entire l2_bucket.
4902                          */
4903                         l2b = pmap_get_l2_bucket(pmap, sva);
4904                         KASSERT(l2b != NULL,
4905                             ("pmap_advise: no l2 bucket for "
4906                              "va 0x%#x, pmap 0x%p", sva, pmap));
4907                         ptep = &l2b->l2b_kva[l2pte_index(sva)];
4908                         opte = *ptep;
4909                         m = PHYS_TO_VM_PAGE(l2pte_pa(*ptep));
4910                         KASSERT(m != NULL,
4911                             ("pmap_advise: no vm_page for demoted superpage"));
4912                         pve = pmap_find_pv(&m->md, pmap, sva);
4913                         KASSERT(pve != NULL,
4914                             ("pmap_advise: no PV entry for managed mapping"));
4915                         if ((pve->pv_flags & PVF_WIRED) == 0) {
4916                                 pmap_free_l2_bucket(pmap, l2b, 1);
4917                                 pve = pmap_remove_pv(m, pmap, sva);
4918                                 pmap_free_pv_entry(pmap, pve);
4919                                 *ptep = 0;
4920                                 PTE_SYNC(ptep);
4921                                 if (pmap_is_current(pmap)) {
4922                                         if (PTE_BEEN_EXECD(opte))
4923                                                 cpu_tlb_flushID_SE(sva);
4924                                         else if (PTE_BEEN_REFD(opte))
4925                                                 cpu_tlb_flushD_SE(sva);
4926                                 }
4927                         }
4928                 }
4929                 if (next_bucket > eva)
4930                         next_bucket = eva;
4931                 l2b = pmap_get_l2_bucket(pmap, sva);
4932                 if (l2b == NULL)
4933                         continue;
4934                 for (ptep = &l2b->l2b_kva[l2pte_index(sva)];
4935                     sva != next_bucket; ptep++, sva += PAGE_SIZE) {
4936                         opte = pte = *ptep;
4937                         if ((opte & L2_S_PROTO) == 0)
4938                                 continue;
4939                         m = PHYS_TO_VM_PAGE(l2pte_pa(opte));
4940                         if (m == NULL || (m->oflags & VPO_UNMANAGED) != 0)
4941                                 continue;
4942                         else if (L2_S_WRITABLE(opte)) {
4943                                 if (advice == MADV_DONTNEED) {
4944                                         /*
4945                                          * Don't need to mark the page
4946                                          * dirty as it was already marked as
4947                                          * such in pmap_fault_fixup() or
4948                                          * pmap_enter_locked().
4949                                          * Just clear the state.
4950                                          */
4951                                 } else
4952                                         pte |= L2_APX;
4953
4954                                 pte &= ~L2_S_REF;
4955                                 *ptep = pte;
4956                                 PTE_SYNC(ptep);
4957                         } else if (L2_S_REFERENCED(opte)) {
4958                                 pte &= ~L2_S_REF;
4959                                 *ptep = pte;
4960                                 PTE_SYNC(ptep);
4961                         } else
4962                                 continue;
4963                         if (pmap_is_current(pmap)) {
4964                                 if (PTE_BEEN_EXECD(opte))
4965                                         cpu_tlb_flushID_SE(sva);
4966                                 else if (PTE_BEEN_REFD(opte))
4967                                         cpu_tlb_flushD_SE(sva);
4968                         }
4969                 }
4970         }
4971         cpu_cpwait();
4972         rw_wunlock(&pvh_global_lock);
4973         PMAP_UNLOCK(pmap);
4974 }
4975
4976 /*
4977  *      Clear the modify bits on the specified physical page.
4978  */
4979 void
4980 pmap_clear_modify(vm_page_t m)
4981 {
4982
4983         KASSERT((m->oflags & VPO_UNMANAGED) == 0,
4984             ("pmap_clear_modify: page %p is not managed", m));
4985         VM_OBJECT_ASSERT_WLOCKED(m->object);
4986         KASSERT(!vm_page_xbusied(m),
4987             ("pmap_clear_modify: page %p is exclusive busied", m));
4988
4989         /*
4990          * If the page is not PGA_WRITEABLE, then no mappings can be modified.
4991          * If the object containing the page is locked and the page is not
4992          * exclusive busied, then PGA_WRITEABLE cannot be concurrently set.
4993          */
4994         if ((m->aflags & PGA_WRITEABLE) == 0)
4995                 return;
4996         if (pmap_is_modified(m))
4997                 pmap_clearbit(m, PVF_MOD);
4998 }
4999
5000
5001 /*
5002  * Clear the write and modified bits in each of the given page's mappings.
5003  */
5004 void
5005 pmap_remove_write(vm_page_t m)
5006 {
5007         KASSERT((m->oflags & VPO_UNMANAGED) == 0,
5008             ("pmap_remove_write: page %p is not managed", m));
5009
5010         /*
5011          * If the page is not exclusive busied, then PGA_WRITEABLE cannot be
5012          * set by another thread while the object is locked.  Thus,
5013          * if PGA_WRITEABLE is clear, no page table entries need updating.
5014          */
5015         VM_OBJECT_ASSERT_WLOCKED(m->object);
5016         if (vm_page_xbusied(m) || (m->aflags & PGA_WRITEABLE) != 0)
5017                 pmap_clearbit(m, PVF_WRITE);
5018 }
5019
5020
5021 /*
5022  * perform the pmap work for mincore
5023  */
5024 int
5025 pmap_mincore(pmap_t pmap, vm_offset_t addr, vm_paddr_t *locked_pa)
5026 {
5027         struct l2_bucket *l2b;
5028         pd_entry_t *pl1pd, l1pd;
5029         pt_entry_t *ptep, pte;
5030         vm_paddr_t pa;
5031         vm_page_t m;
5032         int val;
5033         boolean_t managed;
5034
5035         PMAP_LOCK(pmap);
5036 retry:
5037         pl1pd = &pmap->pm_l1->l1_kva[L1_IDX(addr)];
5038         l1pd = *pl1pd;
5039         if ((l1pd & L1_TYPE_MASK) == L1_S_PROTO) {
5040                 pa = (l1pd & L1_S_FRAME);
5041                 val = MINCORE_SUPER | MINCORE_INCORE;
5042                 if (L1_S_WRITABLE(l1pd))
5043                         val |= MINCORE_MODIFIED | MINCORE_MODIFIED_OTHER;
5044                 managed = FALSE;
5045                 m = PHYS_TO_VM_PAGE(pa);
5046                 if (m != NULL && (m->oflags & VPO_UNMANAGED) == 0)
5047                         managed = TRUE;
5048                 if (managed) {
5049                         if (L1_S_REFERENCED(l1pd))
5050                                 val |= MINCORE_REFERENCED |
5051                                     MINCORE_REFERENCED_OTHER;
5052                 }
5053         } else {
5054                 l2b = pmap_get_l2_bucket(pmap, addr);
5055                 if (l2b == NULL) {
5056                         val = 0;
5057                         goto out;
5058                 }
5059                 ptep = &l2b->l2b_kva[l2pte_index(addr)];
5060                 pte = *ptep;
5061                 if (!l2pte_valid(pte)) {
5062                         val = 0;
5063                         goto out;
5064                 }
5065                 val = MINCORE_INCORE;
5066                 if (L2_S_WRITABLE(pte))
5067                         val |= MINCORE_MODIFIED | MINCORE_MODIFIED_OTHER;
5068                 managed = FALSE;
5069                 pa = l2pte_pa(pte);
5070                 m = PHYS_TO_VM_PAGE(pa);
5071                 if (m != NULL && (m->oflags & VPO_UNMANAGED) == 0)
5072                         managed = TRUE;
5073                 if (managed) {
5074                         if (L2_S_REFERENCED(pte))
5075                                 val |= MINCORE_REFERENCED |
5076                                     MINCORE_REFERENCED_OTHER;
5077                 }
5078         }
5079         if ((val & (MINCORE_MODIFIED_OTHER | MINCORE_REFERENCED_OTHER)) !=
5080             (MINCORE_MODIFIED_OTHER | MINCORE_REFERENCED_OTHER) && managed) {
5081                 /* Ensure that "PHYS_TO_VM_PAGE(pa)->object" doesn't change. */
5082                 if (vm_page_pa_tryrelock(pmap, pa, locked_pa))
5083                         goto retry;
5084         } else
5085 out:
5086                 PA_UNLOCK_COND(*locked_pa);
5087         PMAP_UNLOCK(pmap);
5088         return (val);
5089 }
5090
5091 void
5092 pmap_sync_icache(pmap_t pmap, vm_offset_t va, vm_size_t sz)
5093 {
5094 }
5095
5096 /*
5097  *      Increase the starting virtual address of the given mapping if a
5098  *      different alignment might result in more superpage mappings.
5099  */
5100 void
5101 pmap_align_superpage(vm_object_t object, vm_ooffset_t offset,
5102     vm_offset_t *addr, vm_size_t size)
5103 {
5104         vm_offset_t superpage_offset;
5105
5106         if (size < NBPDR)
5107                 return;
5108         if (object != NULL && (object->flags & OBJ_COLORED) != 0)
5109                 offset += ptoa(object->pg_color);
5110         superpage_offset = offset & PDRMASK;
5111         if (size - ((NBPDR - superpage_offset) & PDRMASK) < NBPDR ||
5112             (*addr & PDRMASK) == superpage_offset)
5113                 return;
5114         if ((*addr & PDRMASK) < superpage_offset)
5115                 *addr = (*addr & ~PDRMASK) + superpage_offset;
5116         else
5117                 *addr = ((*addr + PDRMASK) & ~PDRMASK) + superpage_offset;
5118 }
5119
5120 /*
5121  * pmap_map_section:
5122  *
5123  *      Create a single section mapping.
5124  */
5125 void
5126 pmap_map_section(pmap_t pmap, vm_offset_t va, vm_offset_t pa, vm_prot_t prot,
5127     boolean_t ref)
5128 {
5129         pd_entry_t *pl1pd, l1pd;
5130         pd_entry_t fl;
5131
5132         KASSERT(((va | pa) & L1_S_OFFSET) == 0,
5133             ("Not a valid section mapping"));
5134
5135         fl = pte_l1_s_cache_mode;
5136
5137         pl1pd = &pmap->pm_l1->l1_kva[L1_IDX(va)];
5138         l1pd = L1_S_PROTO | pa | L1_S_PROT(PTE_USER, prot) | fl |
5139             L1_S_DOM(pmap->pm_domain);
5140
5141         /* Mark page referenced if this section is a result of a promotion. */
5142         if (ref == TRUE)
5143                 l1pd |= L1_S_REF;
5144 #ifdef SMP
5145         l1pd |= L1_SHARED;
5146 #endif
5147         *pl1pd = l1pd;
5148         PTE_SYNC(pl1pd);
5149 }
5150
5151 /*
5152  * pmap_link_l2pt:
5153  *
5154  *      Link the L2 page table specified by l2pv.pv_pa into the L1
5155  *      page table at the slot for "va".
5156  */
5157 void
5158 pmap_link_l2pt(vm_offset_t l1pt, vm_offset_t va, struct pv_addr *l2pv)
5159 {
5160         pd_entry_t *pde = (pd_entry_t *) l1pt, proto;
5161         u_int slot = va >> L1_S_SHIFT;
5162
5163         proto = L1_S_DOM(PMAP_DOMAIN_KERNEL) | L1_C_PROTO;
5164
5165 #ifdef VERBOSE_INIT_ARM
5166         printf("pmap_link_l2pt: pa=0x%x va=0x%x\n", l2pv->pv_pa, l2pv->pv_va);
5167 #endif
5168
5169         pde[slot + 0] = proto | (l2pv->pv_pa + 0x000);
5170         PTE_SYNC(&pde[slot]);
5171
5172         SLIST_INSERT_HEAD(&kernel_pt_list, l2pv, pv_list);
5173
5174 }
5175
5176 /*
5177  * pmap_map_entry
5178  *
5179  *      Create a single page mapping.
5180  */
5181 void
5182 pmap_map_entry(vm_offset_t l1pt, vm_offset_t va, vm_offset_t pa, int prot,
5183     int cache)
5184 {
5185         pd_entry_t *pde = (pd_entry_t *) l1pt;
5186         pt_entry_t fl;
5187         pt_entry_t *ptep;
5188
5189         KASSERT(((va | pa) & PAGE_MASK) == 0, ("ouin"));
5190
5191         fl = l2s_mem_types[cache];
5192
5193         if ((pde[va >> L1_S_SHIFT] & L1_TYPE_MASK) != L1_TYPE_C)
5194                 panic("pmap_map_entry: no L2 table for VA 0x%08x", va);
5195
5196         ptep = (pt_entry_t *)kernel_pt_lookup(pde[L1_IDX(va)] & L1_C_ADDR_MASK);
5197
5198         if (ptep == NULL)
5199                 panic("pmap_map_entry: can't find L2 table for VA 0x%08x", va);
5200
5201         ptep[l2pte_index(va)] = L2_S_PROTO | pa | fl | L2_S_REF;
5202         pmap_set_prot(&ptep[l2pte_index(va)], prot, 0);
5203         PTE_SYNC(&ptep[l2pte_index(va)]);
5204 }
5205
5206 /*
5207  * pmap_map_chunk:
5208  *
5209  *      Map a chunk of memory using the most efficient mappings
5210  *      possible (section. large page, small page) into the
5211  *      provided L1 and L2 tables at the specified virtual address.
5212  */
5213 vm_size_t
5214 pmap_map_chunk(vm_offset_t l1pt, vm_offset_t va, vm_offset_t pa,
5215     vm_size_t size, int prot, int type)
5216 {
5217         pd_entry_t *pde = (pd_entry_t *) l1pt;
5218         pt_entry_t *ptep, f1, f2s, f2l;
5219         vm_size_t resid;
5220         int i;
5221
5222         resid = (size + (PAGE_SIZE - 1)) & ~(PAGE_SIZE - 1);
5223
5224         if (l1pt == 0)
5225                 panic("pmap_map_chunk: no L1 table provided");
5226
5227 #ifdef VERBOSE_INIT_ARM
5228         printf("pmap_map_chunk: pa=0x%x va=0x%x size=0x%x resid=0x%x "
5229             "prot=0x%x type=%d\n", pa, va, size, resid, prot, type);
5230 #endif
5231
5232         f1 = l1_mem_types[type];
5233         f2l = l2l_mem_types[type];
5234         f2s = l2s_mem_types[type];
5235
5236         size = resid;
5237
5238         while (resid > 0) {
5239                 /* See if we can use a section mapping. */
5240                 if (L1_S_MAPPABLE_P(va, pa, resid)) {
5241 #ifdef VERBOSE_INIT_ARM
5242                         printf("S");
5243 #endif
5244                         pde[va >> L1_S_SHIFT] = L1_S_PROTO | pa |
5245                             L1_S_PROT(PTE_KERNEL, prot | VM_PROT_EXECUTE) |
5246                             f1 | L1_S_DOM(PMAP_DOMAIN_KERNEL) | L1_S_REF;
5247                         PTE_SYNC(&pde[va >> L1_S_SHIFT]);
5248                         va += L1_S_SIZE;
5249                         pa += L1_S_SIZE;
5250                         resid -= L1_S_SIZE;
5251                         continue;
5252                 }
5253
5254                 /*
5255                  * Ok, we're going to use an L2 table.  Make sure
5256                  * one is actually in the corresponding L1 slot
5257                  * for the current VA.
5258                  */
5259                 if ((pde[va >> L1_S_SHIFT] & L1_TYPE_MASK) != L1_TYPE_C)
5260                         panic("pmap_map_chunk: no L2 table for VA 0x%08x", va);
5261
5262                 ptep = (pt_entry_t *) kernel_pt_lookup(
5263                     pde[L1_IDX(va)] & L1_C_ADDR_MASK);
5264                 if (ptep == NULL)
5265                         panic("pmap_map_chunk: can't find L2 table for VA"
5266                             "0x%08x", va);
5267                 /* See if we can use a L2 large page mapping. */
5268                 if (L2_L_MAPPABLE_P(va, pa, resid)) {
5269 #ifdef VERBOSE_INIT_ARM
5270                         printf("L");
5271 #endif
5272                         for (i = 0; i < 16; i++) {
5273                                 ptep[l2pte_index(va) + i] =
5274                                     L2_L_PROTO | pa |
5275                                     L2_L_PROT(PTE_KERNEL, prot) | f2l;
5276                                 PTE_SYNC(&ptep[l2pte_index(va) + i]);
5277                         }
5278                         va += L2_L_SIZE;
5279                         pa += L2_L_SIZE;
5280                         resid -= L2_L_SIZE;
5281                         continue;
5282                 }
5283
5284                 /* Use a small page mapping. */
5285 #ifdef VERBOSE_INIT_ARM
5286                 printf("P");
5287 #endif
5288                 ptep[l2pte_index(va)] = L2_S_PROTO | pa | f2s | L2_S_REF;
5289                 pmap_set_prot(&ptep[l2pte_index(va)], prot, 0);
5290                 PTE_SYNC(&ptep[l2pte_index(va)]);
5291                 va += PAGE_SIZE;
5292                 pa += PAGE_SIZE;
5293                 resid -= PAGE_SIZE;
5294         }
5295 #ifdef VERBOSE_INIT_ARM
5296         printf("\n");
5297 #endif
5298         return (size);
5299
5300 }
5301
5302 int
5303 pmap_dmap_iscurrent(pmap_t pmap)
5304 {
5305         return(pmap_is_current(pmap));
5306 }
5307
5308 void
5309 pmap_page_set_memattr(vm_page_t m, vm_memattr_t ma)
5310 {
5311         /*
5312          * Remember the memattr in a field that gets used to set the appropriate
5313          * bits in the PTEs as mappings are established.
5314          */
5315         m->md.pv_memattr = ma;
5316
5317         /*
5318          * It appears that this function can only be called before any mappings
5319          * for the page are established on ARM.  If this ever changes, this code
5320          * will need to walk the pv_list and make each of the existing mappings
5321          * uncacheable, being careful to sync caches and PTEs (and maybe
5322          * invalidate TLB?) for any current mapping it modifies.
5323          */
5324         if (TAILQ_FIRST(&m->md.pv_list) != NULL)
5325                 panic("Can't change memattr on page with existing mappings");
5326 }