kvm_mmu.h 15 KB

123456789101112131415161718192021222324252627282930313233343536373839404142434445464748495051525354555657585960616263646566676869707172737475767778798081828384858687888990919293949596979899100101102103104105106107108109110111112113114115116117118119120121122123124125126127128129130131132133134135136137138139140141142143144145146147148149150151152153154155156157158159160161162163164165166167168169170171172173174175176177178179180181182183184185186187188189190191192193194195196197198199200201202203204205206207208209210211212213214215216217218219220221222223224225226227228229230231232233234235236237238239240241242243244245246247248249250251252253254255256257258259260261262263264265266267268269270271272273274275276277278279280281282283284285286287288289290291292293294295296297298299300301302303304305306307308309310311312313314315316317318319320321322323324325326327328329330331332333334335336337338339340341342343344345346347348349350351352353354355356357358359360361362363364365366367368369370371372373374375376377378379380381382383384385386387388389390391392393394395396397398399400401402403404405406407408409410411412413414415416417418419420421422423424425426427428429430431432433434435436437438439440441442443444445446447448449450451452453454455456457458459460461462463464465466467468469470471472473474475476477478479480481482483484485486487488489490491492493494495496497498499500501502503504505506507508509510511512513514515516517518519520521522523524525526527528529530531532
  1. /*
  2. * Copyright (C) 2012,2013 - ARM Ltd
  3. * Author: Marc Zyngier <marc.zyngier@arm.com>
  4. *
  5. * This program is free software; you can redistribute it and/or modify
  6. * it under the terms of the GNU General Public License version 2 as
  7. * published by the Free Software Foundation.
  8. *
  9. * This program is distributed in the hope that it will be useful,
  10. * but WITHOUT ANY WARRANTY; without even the implied warranty of
  11. * MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
  12. * GNU General Public License for more details.
  13. *
  14. * You should have received a copy of the GNU General Public License
  15. * along with this program. If not, see <http://www.gnu.org/licenses/>.
  16. */
  17. #ifndef __ARM64_KVM_MMU_H__
  18. #define __ARM64_KVM_MMU_H__
  19. #include <asm/page.h>
  20. #include <asm/memory.h>
  21. #include <asm/cpufeature.h>
  22. /*
  23. * As ARMv8.0 only has the TTBR0_EL2 register, we cannot express
  24. * "negative" addresses. This makes it impossible to directly share
  25. * mappings with the kernel.
  26. *
  27. * Instead, give the HYP mode its own VA region at a fixed offset from
  28. * the kernel by just masking the top bits (which are all ones for a
  29. * kernel address). We need to find out how many bits to mask.
  30. *
  31. * We want to build a set of page tables that cover both parts of the
  32. * idmap (the trampoline page used to initialize EL2), and our normal
  33. * runtime VA space, at the same time.
  34. *
  35. * Given that the kernel uses VA_BITS for its entire address space,
  36. * and that half of that space (VA_BITS - 1) is used for the linear
  37. * mapping, we can also limit the EL2 space to (VA_BITS - 1).
  38. *
  39. * The main question is "Within the VA_BITS space, does EL2 use the
  40. * top or the bottom half of that space to shadow the kernel's linear
  41. * mapping?". As we need to idmap the trampoline page, this is
  42. * determined by the range in which this page lives.
  43. *
  44. * If the page is in the bottom half, we have to use the top half. If
  45. * the page is in the top half, we have to use the bottom half:
  46. *
  47. * T = __pa_symbol(__hyp_idmap_text_start)
  48. * if (T & BIT(VA_BITS - 1))
  49. * HYP_VA_MIN = 0 //idmap in upper half
  50. * else
  51. * HYP_VA_MIN = 1 << (VA_BITS - 1)
  52. * HYP_VA_MAX = HYP_VA_MIN + (1 << (VA_BITS - 1)) - 1
  53. *
  54. * This of course assumes that the trampoline page exists within the
  55. * VA_BITS range. If it doesn't, then it means we're in the odd case
  56. * where the kernel idmap (as well as HYP) uses more levels than the
  57. * kernel runtime page tables (as seen when the kernel is configured
  58. * for 4k pages, 39bits VA, and yet memory lives just above that
  59. * limit, forcing the idmap to use 4 levels of page tables while the
  60. * kernel itself only uses 3). In this particular case, it doesn't
  61. * matter which side of VA_BITS we use, as we're guaranteed not to
  62. * conflict with anything.
  63. *
  64. * When using VHE, there are no separate hyp mappings and all KVM
  65. * functionality is already mapped as part of the main kernel
  66. * mappings, and none of this applies in that case.
  67. */
  68. #ifdef __ASSEMBLY__
  69. #include <asm/alternative.h>
  70. /*
  71. * Convert a kernel VA into a HYP VA.
  72. * reg: VA to be converted.
  73. *
  74. * The actual code generation takes place in kvm_update_va_mask, and
  75. * the instructions below are only there to reserve the space and
  76. * perform the register allocation (kvm_update_va_mask uses the
  77. * specific registers encoded in the instructions).
  78. */
  79. .macro kern_hyp_va reg
  80. alternative_cb kvm_update_va_mask
  81. and \reg, \reg, #1 /* mask with va_mask */
  82. ror \reg, \reg, #1 /* rotate to the first tag bit */
  83. add \reg, \reg, #0 /* insert the low 12 bits of the tag */
  84. add \reg, \reg, #0, lsl 12 /* insert the top 12 bits of the tag */
  85. ror \reg, \reg, #63 /* rotate back */
  86. alternative_cb_end
  87. .endm
  88. #else
  89. #include <asm/pgalloc.h>
  90. #include <asm/cache.h>
  91. #include <asm/cacheflush.h>
  92. #include <asm/mmu_context.h>
  93. #include <asm/pgtable.h>
  94. void kvm_update_va_mask(struct alt_instr *alt,
  95. __le32 *origptr, __le32 *updptr, int nr_inst);
  96. static inline unsigned long __kern_hyp_va(unsigned long v)
  97. {
  98. asm volatile(ALTERNATIVE_CB("and %0, %0, #1\n"
  99. "ror %0, %0, #1\n"
  100. "add %0, %0, #0\n"
  101. "add %0, %0, #0, lsl 12\n"
  102. "ror %0, %0, #63\n",
  103. kvm_update_va_mask)
  104. : "+r" (v));
  105. return v;
  106. }
  107. #define kern_hyp_va(v) ((typeof(v))(__kern_hyp_va((unsigned long)(v))))
  108. /*
  109. * Obtain the PC-relative address of a kernel symbol
  110. * s: symbol
  111. *
  112. * The goal of this macro is to return a symbol's address based on a
  113. * PC-relative computation, as opposed to a loading the VA from a
  114. * constant pool or something similar. This works well for HYP, as an
  115. * absolute VA is guaranteed to be wrong. Only use this if trying to
  116. * obtain the address of a symbol (i.e. not something you obtained by
  117. * following a pointer).
  118. */
  119. #define hyp_symbol_addr(s) \
  120. ({ \
  121. typeof(s) *addr; \
  122. asm("adrp %0, %1\n" \
  123. "add %0, %0, :lo12:%1\n" \
  124. : "=r" (addr) : "S" (&s)); \
  125. addr; \
  126. })
  127. /*
  128. * We currently only support a 40bit IPA.
  129. */
  130. #define KVM_PHYS_SHIFT (40)
  131. #define KVM_PHYS_SIZE (1UL << KVM_PHYS_SHIFT)
  132. #define KVM_PHYS_MASK (KVM_PHYS_SIZE - 1UL)
  133. #include <asm/stage2_pgtable.h>
  134. int create_hyp_mappings(void *from, void *to, pgprot_t prot);
  135. int create_hyp_io_mappings(phys_addr_t phys_addr, size_t size,
  136. void __iomem **kaddr,
  137. void __iomem **haddr);
  138. int create_hyp_exec_mappings(phys_addr_t phys_addr, size_t size,
  139. void **haddr);
  140. void free_hyp_pgds(void);
  141. void stage2_unmap_vm(struct kvm *kvm);
  142. int kvm_alloc_stage2_pgd(struct kvm *kvm);
  143. void kvm_free_stage2_pgd(struct kvm *kvm);
  144. int kvm_phys_addr_ioremap(struct kvm *kvm, phys_addr_t guest_ipa,
  145. phys_addr_t pa, unsigned long size, bool writable);
  146. int kvm_handle_guest_abort(struct kvm_vcpu *vcpu, struct kvm_run *run);
  147. void kvm_mmu_free_memory_caches(struct kvm_vcpu *vcpu);
  148. phys_addr_t kvm_mmu_get_httbr(void);
  149. phys_addr_t kvm_get_idmap_vector(void);
  150. int kvm_mmu_init(void);
  151. void kvm_clear_hyp_idmap(void);
  152. #define kvm_mk_pmd(ptep) \
  153. __pmd(__phys_to_pmd_val(__pa(ptep)) | PMD_TYPE_TABLE)
  154. #define kvm_mk_pud(pmdp) \
  155. __pud(__phys_to_pud_val(__pa(pmdp)) | PMD_TYPE_TABLE)
  156. #define kvm_mk_pgd(pudp) \
  157. __pgd(__phys_to_pgd_val(__pa(pudp)) | PUD_TYPE_TABLE)
  158. static inline pte_t kvm_s2pte_mkwrite(pte_t pte)
  159. {
  160. pte_val(pte) |= PTE_S2_RDWR;
  161. return pte;
  162. }
  163. static inline pmd_t kvm_s2pmd_mkwrite(pmd_t pmd)
  164. {
  165. pmd_val(pmd) |= PMD_S2_RDWR;
  166. return pmd;
  167. }
  168. static inline pte_t kvm_s2pte_mkexec(pte_t pte)
  169. {
  170. pte_val(pte) &= ~PTE_S2_XN;
  171. return pte;
  172. }
  173. static inline pmd_t kvm_s2pmd_mkexec(pmd_t pmd)
  174. {
  175. pmd_val(pmd) &= ~PMD_S2_XN;
  176. return pmd;
  177. }
  178. static inline void kvm_set_s2pte_readonly(pte_t *ptep)
  179. {
  180. pteval_t old_pteval, pteval;
  181. pteval = READ_ONCE(pte_val(*ptep));
  182. do {
  183. old_pteval = pteval;
  184. pteval &= ~PTE_S2_RDWR;
  185. pteval |= PTE_S2_RDONLY;
  186. pteval = cmpxchg_relaxed(&pte_val(*ptep), old_pteval, pteval);
  187. } while (pteval != old_pteval);
  188. }
  189. static inline bool kvm_s2pte_readonly(pte_t *ptep)
  190. {
  191. return (READ_ONCE(pte_val(*ptep)) & PTE_S2_RDWR) == PTE_S2_RDONLY;
  192. }
  193. static inline bool kvm_s2pte_exec(pte_t *ptep)
  194. {
  195. return !(READ_ONCE(pte_val(*ptep)) & PTE_S2_XN);
  196. }
  197. static inline void kvm_set_s2pmd_readonly(pmd_t *pmdp)
  198. {
  199. kvm_set_s2pte_readonly((pte_t *)pmdp);
  200. }
  201. static inline bool kvm_s2pmd_readonly(pmd_t *pmdp)
  202. {
  203. return kvm_s2pte_readonly((pte_t *)pmdp);
  204. }
  205. static inline bool kvm_s2pmd_exec(pmd_t *pmdp)
  206. {
  207. return !(READ_ONCE(pmd_val(*pmdp)) & PMD_S2_XN);
  208. }
  209. static inline bool kvm_page_empty(void *ptr)
  210. {
  211. struct page *ptr_page = virt_to_page(ptr);
  212. return page_count(ptr_page) == 1;
  213. }
  214. #define hyp_pte_table_empty(ptep) kvm_page_empty(ptep)
  215. #ifdef __PAGETABLE_PMD_FOLDED
  216. #define hyp_pmd_table_empty(pmdp) (0)
  217. #else
  218. #define hyp_pmd_table_empty(pmdp) kvm_page_empty(pmdp)
  219. #endif
  220. #ifdef __PAGETABLE_PUD_FOLDED
  221. #define hyp_pud_table_empty(pudp) (0)
  222. #else
  223. #define hyp_pud_table_empty(pudp) kvm_page_empty(pudp)
  224. #endif
  225. struct kvm;
  226. #define kvm_flush_dcache_to_poc(a,l) __flush_dcache_area((a), (l))
  227. static inline bool vcpu_has_cache_enabled(struct kvm_vcpu *vcpu)
  228. {
  229. return (vcpu_read_sys_reg(vcpu, SCTLR_EL1) & 0b101) == 0b101;
  230. }
  231. static inline void __clean_dcache_guest_page(kvm_pfn_t pfn, unsigned long size)
  232. {
  233. void *va = page_address(pfn_to_page(pfn));
  234. /*
  235. * With FWB, we ensure that the guest always accesses memory using
  236. * cacheable attributes, and we don't have to clean to PoC when
  237. * faulting in pages. Furthermore, FWB implies IDC, so cleaning to
  238. * PoU is not required either in this case.
  239. */
  240. if (cpus_have_const_cap(ARM64_HAS_STAGE2_FWB))
  241. return;
  242. kvm_flush_dcache_to_poc(va, size);
  243. }
  244. static inline void __invalidate_icache_guest_page(kvm_pfn_t pfn,
  245. unsigned long size)
  246. {
  247. if (icache_is_aliasing()) {
  248. /* any kind of VIPT cache */
  249. __flush_icache_all();
  250. } else if (is_kernel_in_hyp_mode() || !icache_is_vpipt()) {
  251. /* PIPT or VPIPT at EL2 (see comment in __kvm_tlb_flush_vmid_ipa) */
  252. void *va = page_address(pfn_to_page(pfn));
  253. invalidate_icache_range((unsigned long)va,
  254. (unsigned long)va + size);
  255. }
  256. }
  257. static inline void __kvm_flush_dcache_pte(pte_t pte)
  258. {
  259. if (!cpus_have_const_cap(ARM64_HAS_STAGE2_FWB)) {
  260. struct page *page = pte_page(pte);
  261. kvm_flush_dcache_to_poc(page_address(page), PAGE_SIZE);
  262. }
  263. }
  264. static inline void __kvm_flush_dcache_pmd(pmd_t pmd)
  265. {
  266. if (!cpus_have_const_cap(ARM64_HAS_STAGE2_FWB)) {
  267. struct page *page = pmd_page(pmd);
  268. kvm_flush_dcache_to_poc(page_address(page), PMD_SIZE);
  269. }
  270. }
  271. static inline void __kvm_flush_dcache_pud(pud_t pud)
  272. {
  273. if (!cpus_have_const_cap(ARM64_HAS_STAGE2_FWB)) {
  274. struct page *page = pud_page(pud);
  275. kvm_flush_dcache_to_poc(page_address(page), PUD_SIZE);
  276. }
  277. }
  278. #define kvm_virt_to_phys(x) __pa_symbol(x)
  279. void kvm_set_way_flush(struct kvm_vcpu *vcpu);
  280. void kvm_toggle_cache(struct kvm_vcpu *vcpu, bool was_enabled);
  281. static inline bool __kvm_cpu_uses_extended_idmap(void)
  282. {
  283. return __cpu_uses_extended_idmap_level();
  284. }
  285. static inline unsigned long __kvm_idmap_ptrs_per_pgd(void)
  286. {
  287. return idmap_ptrs_per_pgd;
  288. }
  289. /*
  290. * Can't use pgd_populate here, because the extended idmap adds an extra level
  291. * above CONFIG_PGTABLE_LEVELS (which is 2 or 3 if we're using the extended
  292. * idmap), and pgd_populate is only available if CONFIG_PGTABLE_LEVELS = 4.
  293. */
  294. static inline void __kvm_extend_hypmap(pgd_t *boot_hyp_pgd,
  295. pgd_t *hyp_pgd,
  296. pgd_t *merged_hyp_pgd,
  297. unsigned long hyp_idmap_start)
  298. {
  299. int idmap_idx;
  300. u64 pgd_addr;
  301. /*
  302. * Use the first entry to access the HYP mappings. It is
  303. * guaranteed to be free, otherwise we wouldn't use an
  304. * extended idmap.
  305. */
  306. VM_BUG_ON(pgd_val(merged_hyp_pgd[0]));
  307. pgd_addr = __phys_to_pgd_val(__pa(hyp_pgd));
  308. merged_hyp_pgd[0] = __pgd(pgd_addr | PMD_TYPE_TABLE);
  309. /*
  310. * Create another extended level entry that points to the boot HYP map,
  311. * which contains an ID mapping of the HYP init code. We essentially
  312. * merge the boot and runtime HYP maps by doing so, but they don't
  313. * overlap anyway, so this is fine.
  314. */
  315. idmap_idx = hyp_idmap_start >> VA_BITS;
  316. VM_BUG_ON(pgd_val(merged_hyp_pgd[idmap_idx]));
  317. pgd_addr = __phys_to_pgd_val(__pa(boot_hyp_pgd));
  318. merged_hyp_pgd[idmap_idx] = __pgd(pgd_addr | PMD_TYPE_TABLE);
  319. }
  320. static inline unsigned int kvm_get_vmid_bits(void)
  321. {
  322. int reg = read_sanitised_ftr_reg(SYS_ID_AA64MMFR1_EL1);
  323. return (cpuid_feature_extract_unsigned_field(reg, ID_AA64MMFR1_VMIDBITS_SHIFT) == 2) ? 16 : 8;
  324. }
  325. /*
  326. * We are not in the kvm->srcu critical section most of the time, so we take
  327. * the SRCU read lock here. Since we copy the data from the user page, we
  328. * can immediately drop the lock again.
  329. */
  330. static inline int kvm_read_guest_lock(struct kvm *kvm,
  331. gpa_t gpa, void *data, unsigned long len)
  332. {
  333. int srcu_idx = srcu_read_lock(&kvm->srcu);
  334. int ret = kvm_read_guest(kvm, gpa, data, len);
  335. srcu_read_unlock(&kvm->srcu, srcu_idx);
  336. return ret;
  337. }
  338. static inline int kvm_write_guest_lock(struct kvm *kvm, gpa_t gpa,
  339. const void *data, unsigned long len)
  340. {
  341. int srcu_idx = srcu_read_lock(&kvm->srcu);
  342. int ret = kvm_write_guest(kvm, gpa, data, len);
  343. srcu_read_unlock(&kvm->srcu, srcu_idx);
  344. return ret;
  345. }
  346. #ifdef CONFIG_KVM_INDIRECT_VECTORS
  347. /*
  348. * EL2 vectors can be mapped and rerouted in a number of ways,
  349. * depending on the kernel configuration and CPU present:
  350. *
  351. * - If the CPU has the ARM64_HARDEN_BRANCH_PREDICTOR cap, the
  352. * hardening sequence is placed in one of the vector slots, which is
  353. * executed before jumping to the real vectors.
  354. *
  355. * - If the CPU has both the ARM64_HARDEN_EL2_VECTORS cap and the
  356. * ARM64_HARDEN_BRANCH_PREDICTOR cap, the slot containing the
  357. * hardening sequence is mapped next to the idmap page, and executed
  358. * before jumping to the real vectors.
  359. *
  360. * - If the CPU only has the ARM64_HARDEN_EL2_VECTORS cap, then an
  361. * empty slot is selected, mapped next to the idmap page, and
  362. * executed before jumping to the real vectors.
  363. *
  364. * Note that ARM64_HARDEN_EL2_VECTORS is somewhat incompatible with
  365. * VHE, as we don't have hypervisor-specific mappings. If the system
  366. * is VHE and yet selects this capability, it will be ignored.
  367. */
  368. #include <asm/mmu.h>
  369. extern void *__kvm_bp_vect_base;
  370. extern int __kvm_harden_el2_vector_slot;
  371. static inline void *kvm_get_hyp_vector(void)
  372. {
  373. struct bp_hardening_data *data = arm64_get_bp_hardening_data();
  374. void *vect = kern_hyp_va(kvm_ksym_ref(__kvm_hyp_vector));
  375. int slot = -1;
  376. if (cpus_have_const_cap(ARM64_HARDEN_BRANCH_PREDICTOR) && data->fn) {
  377. vect = kern_hyp_va(kvm_ksym_ref(__bp_harden_hyp_vecs_start));
  378. slot = data->hyp_vectors_slot;
  379. }
  380. if (this_cpu_has_cap(ARM64_HARDEN_EL2_VECTORS) && !has_vhe()) {
  381. vect = __kvm_bp_vect_base;
  382. if (slot == -1)
  383. slot = __kvm_harden_el2_vector_slot;
  384. }
  385. if (slot != -1)
  386. vect += slot * SZ_2K;
  387. return vect;
  388. }
  389. /* This is only called on a !VHE system */
  390. static inline int kvm_map_vectors(void)
  391. {
  392. /*
  393. * HBP = ARM64_HARDEN_BRANCH_PREDICTOR
  394. * HEL2 = ARM64_HARDEN_EL2_VECTORS
  395. *
  396. * !HBP + !HEL2 -> use direct vectors
  397. * HBP + !HEL2 -> use hardened vectors in place
  398. * !HBP + HEL2 -> allocate one vector slot and use exec mapping
  399. * HBP + HEL2 -> use hardened vertors and use exec mapping
  400. */
  401. if (cpus_have_const_cap(ARM64_HARDEN_BRANCH_PREDICTOR)) {
  402. __kvm_bp_vect_base = kvm_ksym_ref(__bp_harden_hyp_vecs_start);
  403. __kvm_bp_vect_base = kern_hyp_va(__kvm_bp_vect_base);
  404. }
  405. if (cpus_have_const_cap(ARM64_HARDEN_EL2_VECTORS)) {
  406. phys_addr_t vect_pa = __pa_symbol(__bp_harden_hyp_vecs_start);
  407. unsigned long size = (__bp_harden_hyp_vecs_end -
  408. __bp_harden_hyp_vecs_start);
  409. /*
  410. * Always allocate a spare vector slot, as we don't
  411. * know yet which CPUs have a BP hardening slot that
  412. * we can reuse.
  413. */
  414. __kvm_harden_el2_vector_slot = atomic_inc_return(&arm64_el2_vector_last_slot);
  415. BUG_ON(__kvm_harden_el2_vector_slot >= BP_HARDEN_EL2_SLOTS);
  416. return create_hyp_exec_mappings(vect_pa, size,
  417. &__kvm_bp_vect_base);
  418. }
  419. return 0;
  420. }
  421. #else
  422. static inline void *kvm_get_hyp_vector(void)
  423. {
  424. return kern_hyp_va(kvm_ksym_ref(__kvm_hyp_vector));
  425. }
  426. static inline int kvm_map_vectors(void)
  427. {
  428. return 0;
  429. }
  430. #endif
  431. #ifdef CONFIG_ARM64_SSBD
  432. DECLARE_PER_CPU_READ_MOSTLY(u64, arm64_ssbd_callback_required);
  433. static inline int hyp_map_aux_data(void)
  434. {
  435. int cpu, err;
  436. for_each_possible_cpu(cpu) {
  437. u64 *ptr;
  438. ptr = per_cpu_ptr(&arm64_ssbd_callback_required, cpu);
  439. err = create_hyp_mappings(ptr, ptr + 1, PAGE_HYP);
  440. if (err)
  441. return err;
  442. }
  443. return 0;
  444. }
  445. #else
  446. static inline int hyp_map_aux_data(void)
  447. {
  448. return 0;
  449. }
  450. #endif
  451. #define kvm_phys_to_vttbr(addr) phys_to_ttbr(addr)
  452. #endif /* __ASSEMBLY__ */
  453. #endif /* __ARM64_KVM_MMU_H__ */