Skip to main content

ostd/mm/kspace/
mod.rs

1// SPDX-License-Identifier: MPL-2.0
2
3//! Kernel memory space management.
4//!
5//! The kernel memory space is currently managed as follows, if the
6//! address width is 48 bits (with 47 bits kernel space).
7//!
8//! TODO: the cap of linear mapping (the start of vm alloc) are raised
9//! to workaround for high IO in TDX. We need actual vm alloc API to have
10//! a proper fix.
11//!
12//! ```text
13//! +-+ <- the highest used address (0xffff_ffff_ffff_0000)
14//! | |         For the kernel code, 1 GiB.
15//! +-+ <- 0xffff_ffff_8000_0000
16//! | |
17//! | |         Unused hole.
18//! +-+ <- 0xffff_e100_0000_0000
19//! | |         For frame metadata, 1 TiB.
20//! +-+ <- 0xffff_e000_0000_0000
21//! | |         For [`KVirtArea`], 32 TiB.
22//! +-+ <- the middle of the higher half (0xffff_c000_0000_0000)
23//! | |
24//! | |
25//! | |
26//! | |         For linear mappings, 64 TiB.
27//! | |         Mapped physical addresses are untracked.
28//! | |
29//! | |
30//! | |
31//! +-+ <- the base of high canonical address (0xffff_8000_0000_0000)
32//! ```
33//!
34//! If the address width is (according to [`crate::arch::mm::PagingConsts`])
35//! 39 bits or 57 bits, the memory space just adjust proportionally.
36
37#![cfg_attr(target_arch = "loongarch64", expect(unused_imports))]
38
39pub(crate) mod kvirt_area;
40
41use core::ops::Range;
42
43use spin::Once;
44
45#[cfg(ktest)]
46mod test;
47
48use super::{
49    Frame, HasSize, Paddr, PagingConstsTrait, Vaddr,
50    frame::{
51        Segment,
52        meta::{AnyFrameMeta, MetaPageMeta, mapping},
53    },
54    page_prop::{CachePolicy, PageFlags, PageProperty, PrivilegedPageFlags},
55    page_table::{PageTable, PageTableConfig, largest_pages, max_page_level},
56};
57use crate::{
58    arch::mm::{PageTableEntry, PagingConsts},
59    boot::memory_region::MemoryRegionType,
60    const_assert, info,
61    mm::{HasPaddr, PAGE_SIZE, PagingLevel, frame::FrameRef},
62    task::disable_preempt,
63};
64
65// The shortest supported address width is 39 bits. So the literal
66// values are written for 39 bits address width and we adjust the values
67// by arithmetic left shift.
68const_assert!(PagingConsts::ADDRESS_WIDTH >= 39);
69const ADDR_WIDTH_SHIFT: usize = PagingConsts::ADDRESS_WIDTH - 39;
70
71/// Start of the kernel address space.
72#[cfg(not(target_arch = "loongarch64"))]
73pub(super) const KERNEL_BASE_VADDR: Vaddr = 0xffff_ffc0_0000_0000 << ADDR_WIDTH_SHIFT;
74#[cfg(target_arch = "loongarch64")]
75pub(super) const KERNEL_BASE_VADDR: Vaddr = 0x9000_0000_0000_0000;
76/// End of the kernel address space (non inclusive).
77pub(super) const KERNEL_END_VADDR: Vaddr = 0xffff_ffff_ffff_0000;
78
79/// The maximum virtual address of user space (non inclusive).
80///
81/// A typical way to reserve half of the address space for the kernel is
82/// to use the highest `ADDRESS_WIDTH`-bit virtual address space.
83///
84/// Also, the top page is not regarded as usable since it's a workaround
85/// for some x86_64 CPUs' bugs. See
86/// <https://github.com/torvalds/linux/blob/480e035fc4c714fb5536e64ab9db04fedc89e910/arch/x86/include/asm/page_64.h#L68-L78>
87/// for the rationale.
88pub const MAX_USERSPACE_VADDR: Vaddr = (0x0000_0040_0000_0000 << ADDR_WIDTH_SHIFT) - PAGE_SIZE;
89
90/// The kernel address space.
91///
92/// They are the high canonical addresses (i.e., the negative part of the
93/// address space, with the most significant bits in the addresses set).
94pub const KERNEL_VADDR_RANGE: Range<Vaddr> = KERNEL_BASE_VADDR..KERNEL_END_VADDR;
95
96/// The kernel code is linear mapped to this address.
97///
98/// FIXME: This offset should be randomly chosen by the loader or the
99/// boot compatibility layer. But we disabled it because OSTD
100/// doesn't support relocatable kernel yet.
101pub(crate) fn kernel_loaded_offset() -> usize {
102    KERNEL_CODE_BASE_VADDR
103}
104
105#[cfg(target_arch = "x86_64")]
106const KERNEL_CODE_BASE_VADDR: usize = 0xffff_ffff_8000_0000;
107#[cfg(any(target_arch = "riscv64", target_arch = "aarch64"))]
108const KERNEL_CODE_BASE_VADDR: usize = 0xffff_ffff_0000_0000;
109#[cfg(target_arch = "loongarch64")]
110const KERNEL_CODE_BASE_VADDR: usize = 0x9000_0000_0000_0000;
111
112const FRAME_METADATA_CAP_VADDR: Vaddr = 0xffff_fff0_8000_0000 << ADDR_WIDTH_SHIFT;
113const FRAME_METADATA_BASE_VADDR: Vaddr = 0xffff_fff0_0000_0000 << ADDR_WIDTH_SHIFT;
114pub(super) const FRAME_METADATA_RANGE: Range<Vaddr> =
115    FRAME_METADATA_BASE_VADDR..FRAME_METADATA_CAP_VADDR;
116
117const VMALLOC_BASE_VADDR: Vaddr = 0xffff_ffe0_0000_0000 << ADDR_WIDTH_SHIFT;
118pub(super) const VMALLOC_VADDR_RANGE: Range<Vaddr> = VMALLOC_BASE_VADDR..FRAME_METADATA_BASE_VADDR;
119
120/// The base address of the linear mapping of all physical
121/// memory in the kernel address space.
122#[cfg(not(target_arch = "loongarch64"))]
123pub(crate) const LINEAR_MAPPING_BASE_VADDR: Vaddr = 0xffff_ffc0_0000_0000 << ADDR_WIDTH_SHIFT;
124#[cfg(target_arch = "loongarch64")]
125pub(crate) const LINEAR_MAPPING_BASE_VADDR: Vaddr = 0x9000_0000_0000_0000;
126pub(crate) const LINEAR_MAPPING_VADDR_RANGE: Range<Vaddr> =
127    LINEAR_MAPPING_BASE_VADDR..VMALLOC_BASE_VADDR;
128
129/// Convert physical address to virtual address using offset, only available inside `ostd`
130pub(crate) fn paddr_to_vaddr(pa: Paddr) -> usize {
131    debug_assert!(pa < VMALLOC_BASE_VADDR - LINEAR_MAPPING_BASE_VADDR);
132    pa + LINEAR_MAPPING_BASE_VADDR
133}
134
135/// The kernel page table instance.
136///
137/// It manages the kernel mapping of all address spaces by sharing the kernel part. And it
138/// is unlikely to be activated.
139pub(super) static KERNEL_PAGE_TABLE: Once<PageTable<KernelPtConfig>> = Once::new();
140
141#[derive(Clone, Debug)]
142pub(super) struct KernelPtConfig {}
143
144// We use the first available PTE bit to mark the frame as tracked.
145// SAFETY: `item_raw_info`, `item_into_raw`, `item_from_raw`, and
146// `item_ref_from_raw` are correctly implemented with respect to the `Item` and
147// `ItemRef` types.
148unsafe impl PageTableConfig for KernelPtConfig {
149    const TOP_LEVEL_INDEX_RANGE: Range<usize> = 256..512;
150    const TOP_LEVEL_CAN_UNMAP: bool = false;
151
152    type E = PageTableEntry;
153    type C = PagingConsts;
154
155    type Item = MappedItem;
156    type ItemRef<'a> = MappedItemRef<'a>;
157
158    fn item_raw_info(item: &Self::Item) -> (Paddr, PagingLevel, PageProperty) {
159        match *item {
160            MappedItem::Tracked(ref frame, mut prop) => {
161                debug_assert!(!prop.priv_flags.contains(PrivilegedPageFlags::AVAIL1));
162                prop.priv_flags |= PrivilegedPageFlags::AVAIL1;
163                let level = frame.map_level();
164                let paddr = frame.paddr();
165                (paddr, level, prop)
166            }
167            MappedItem::Untracked(ref pa, ref level, mut prop) => {
168                debug_assert!(!prop.priv_flags.contains(PrivilegedPageFlags::AVAIL1));
169                prop.priv_flags -= PrivilegedPageFlags::AVAIL1;
170                (*pa, *level, prop)
171            }
172        }
173    }
174
175    unsafe fn item_from_raw(paddr: Paddr, level: PagingLevel, prop: PageProperty) -> Self::Item {
176        if prop.priv_flags.contains(PrivilegedPageFlags::AVAIL1) {
177            debug_assert_eq!(level, 1);
178            // SAFETY: The caller ensures safety.
179            let frame = unsafe { Frame::<dyn AnyFrameMeta>::from_raw(paddr) };
180            MappedItem::Tracked(frame, prop)
181        } else {
182            MappedItem::Untracked(paddr, level, prop)
183        }
184    }
185
186    unsafe fn item_ref_from_raw<'a>(
187        paddr: Paddr,
188        level: PagingLevel,
189        prop: PageProperty,
190    ) -> Self::ItemRef<'a> {
191        if prop.priv_flags.contains(PrivilegedPageFlags::AVAIL1) {
192            debug_assert_eq!(level, 1);
193            // SAFETY: The caller ensures that the frame outlives `'a` and that
194            // the type matches the frame.
195            let frame = unsafe { FrameRef::<dyn AnyFrameMeta>::borrow_paddr(paddr) };
196            MappedItemRef::Tracked(frame, prop)
197        } else {
198            MappedItemRef::Untracked(paddr, level, prop)
199        }
200    }
201}
202
203#[derive(Clone, Debug, Eq, PartialEq)]
204pub(super) enum MappedItem {
205    Tracked(Frame<dyn AnyFrameMeta>, PageProperty),
206    Untracked(Paddr, PagingLevel, PageProperty),
207}
208
209#[derive(Debug)]
210pub(crate) enum MappedItemRef<'a> {
211    #[cfg_attr(not(ktest), expect(dead_code))]
212    Tracked(FrameRef<'a, dyn AnyFrameMeta>, PageProperty),
213    #[cfg_attr(not(ktest), expect(dead_code))]
214    Untracked(Paddr, PagingLevel, PageProperty),
215}
216
217/// Initializes the kernel page table.
218///
219/// This function should be called after:
220///  - the page allocator and the heap allocator are initialized;
221///  - the memory regions are initialized.
222///
223/// This function should be called before:
224///  - any initializer that modifies the kernel page table.
225pub(crate) fn init_kernel_page_table(meta_pages: Segment<MetaPageMeta>) {
226    info!("Initializing the kernel page table");
227
228    // Start to initialize the kernel page table.
229    let kpt = PageTable::<KernelPtConfig>::new_kernel_page_table();
230    let preempt_guard = disable_preempt();
231
232    // In LoongArch64, we don't need to do linear mappings for the kernel because of DMW0.
233    #[cfg(not(target_arch = "loongarch64"))]
234    // Do linear mappings for the kernel.
235    {
236        let max_paddr = crate::mm::frame::max_paddr();
237        let from = LINEAR_MAPPING_BASE_VADDR..LINEAR_MAPPING_BASE_VADDR + max_paddr;
238        let prop = PageProperty {
239            flags: PageFlags::RW,
240            cache: CachePolicy::Writeback,
241            priv_flags: PrivilegedPageFlags::GLOBAL,
242        };
243        let min_level = max_page_level::<KernelPtConfig>(from.len());
244        let mut cursor = kpt
245            .cursor_mut_with_min_level(&preempt_guard, &from, min_level)
246            .unwrap();
247        for (pa, level) in largest_pages::<KernelPtConfig>(from.start, 0, max_paddr) {
248            // SAFETY: we are doing the linear mapping for the kernel.
249            unsafe { cursor.map(MappedItem::Untracked(pa, level, prop)) };
250        }
251    }
252
253    // Map the metadata pages.
254    {
255        let start_va = mapping::frame_to_meta::<PagingConsts>(crate::mm::frame::min_paddr());
256        let from = start_va..start_va + meta_pages.size();
257        let prop = PageProperty {
258            flags: PageFlags::RW,
259            cache: CachePolicy::Writeback,
260            priv_flags: PrivilegedPageFlags::GLOBAL,
261        };
262        let min_level = max_page_level::<KernelPtConfig>(from.len());
263        let mut cursor = kpt
264            .cursor_mut_with_min_level(&preempt_guard, &from, min_level)
265            .unwrap();
266        // We use untracked mapping so that we can benefit from huge pages.
267        // We won't unmap them anyway, so there's no leaking problem yet.
268        let pa_range = meta_pages.into_raw();
269        for (pa, level) in
270            largest_pages::<KernelPtConfig>(from.start, pa_range.start, pa_range.len())
271        {
272            // SAFETY: We are doing the metadata mappings for the kernel.
273            unsafe { cursor.map(MappedItem::Untracked(pa, level, prop)) };
274        }
275    }
276
277    // In LoongArch64, we don't need to do linear mappings for the kernel code because of DMW0.
278    #[cfg(not(target_arch = "loongarch64"))]
279    // Map for the kernel code itself.
280    // TODO: set separated permissions for each segments in the kernel.
281    {
282        let regions = &crate::boot::EARLY_INFO.get().unwrap().memory_regions;
283        let region = regions
284            .iter()
285            .find(|r| r.typ() == MemoryRegionType::Kernel)
286            .unwrap();
287        let offset = kernel_loaded_offset();
288        let from = region.base() + offset..region.end() + offset;
289        let prop = PageProperty {
290            flags: PageFlags::RWX,
291            cache: CachePolicy::Writeback,
292            priv_flags: PrivilegedPageFlags::GLOBAL,
293        };
294        let min_level = max_page_level::<KernelPtConfig>(from.len());
295        let mut cursor = kpt
296            .cursor_mut_with_min_level(&preempt_guard, &from, min_level)
297            .unwrap();
298        for (pa, level) in largest_pages::<KernelPtConfig>(from.start, region.base(), from.len()) {
299            // SAFETY: we are doing the kernel code mapping.
300            unsafe { cursor.map(MappedItem::Untracked(pa, level, prop)) };
301        }
302    }
303
304    KERNEL_PAGE_TABLE.call_once(|| kpt);
305}
306
307/// Activates the kernel page table.
308///
309/// All address translation of symbols in the boot sections must be manually
310/// done from now on.
311///
312/// # Safety
313///
314/// This function must only be called once per CPU.
315pub(crate) unsafe fn activate_kernel_page_table() {
316    let kpt = KERNEL_PAGE_TABLE
317        .get()
318        .expect("The kernel page table is not initialized yet");
319    // SAFETY: the kernel page table is initialized properly.
320    unsafe {
321        kpt.first_activate_unchecked();
322        crate::arch::mm::tlb_flush_all_including_global();
323    }
324}