diff --git a/arch/x86/include/asm/xen/page.h b/arch/x86/include/asm/xen/page.h index 59f642a94b9d90..80b3eda2fa01b6 100644 --- a/arch/x86/include/asm/xen/page.h +++ b/arch/x86/include/asm/xen/page.h @@ -55,6 +55,11 @@ extern unsigned long xen_max_p2m_pfn; extern int xen_alloc_p2m_entry(unsigned long pfn); +extern int xen_prealloc_p2m_range(unsigned long pfn, unsigned long count); +extern int xen_remap_contig_pfns(unsigned long pfn, unsigned long mfn, + unsigned long count); +extern int xen_zap_contig_pfns(unsigned long pfn, unsigned long count); + extern unsigned long get_phys_to_machine(unsigned long pfn); extern bool set_phys_to_machine(unsigned long pfn, unsigned long mfn); extern bool __set_phys_to_machine(unsigned long pfn, unsigned long mfn); diff --git a/arch/x86/xen/mmu_pv.c b/arch/x86/xen/mmu_pv.c index 2a4a8deaf612f5..d61aa464f46df2 100644 --- a/arch/x86/xen/mmu_pv.c +++ b/arch/x86/xen/mmu_pv.c @@ -2313,6 +2313,114 @@ static void xen_remap_exchanged_ptes(unsigned long vaddr, int order, xen_mc_issue(0); } +/* + * A p2m leaf covers this many pfns, so a run touches one leaf per stride plus + * whichever leaf its last pfn falls in. + */ +#define P2M_PFNS_PER_LEAF (PAGE_SIZE / sizeof(unsigned long)) + +/* + * Make every p2m leaf that a run of @count pfns from @pfn writes through + * present, so that xen_remap_contig_pfns() over the run cannot fail for want + * of one. Callers run this before asking Xen to populate the run, while a + * failure is still easy to back out of. + */ +int xen_prealloc_p2m_range(unsigned long pfn, unsigned long count) +{ + unsigned long i; + int ret; + + for (i = 0; i < count; i += P2M_PFNS_PER_LEAF) { + ret = xen_alloc_p2m_entry(pfn + i); + if (ret < 0) + return ret; + } + + return xen_alloc_p2m_entry(pfn + count - 1); +} +EXPORT_SYMBOL_GPL(xen_prealloc_p2m_range); + +/* PTE updates per mmu_update hypercall, few enough to batch on the stack. */ +#define CONTIG_PTE_BATCH 32UL + +/* + * Point the direct map PTEs of @count contiguous pfns from @pfn at the frames + * counting up from @mfn when @map, or clear them when not, and set their p2m + * entries to match. Returns 0, or the error for the first PTE or p2m entry + * that could not be written. + */ +static int xen_set_contig_ptes(unsigned long pfn, unsigned long mfn, + unsigned long count, bool map) +{ + struct mmu_update u[CONTIG_PTE_BATCH]; + + while (count) { + unsigned long vaddr = (unsigned long)__va(pfn << PAGE_SHIFT); + unsigned long batch = min(count, CONTIG_PTE_BATCH); + unsigned long i; + unsigned int level; + pte_t *ptep; + int ret; + + /* Stay within one PTE page, so that its entries are consecutive. */ + batch = min(batch, (PMD_SIZE - (vaddr & ~PMD_MASK)) >> PAGE_SHIFT); + + ptep = lookup_address(vaddr, &level); + if (!ptep || level != PG_LEVEL_4K) + return -EINVAL; + + for (i = 0; i < batch; i++) { + unsigned long frame = map ? mfn + i : INVALID_P2M_ENTRY; + pte_t pte = map ? mfn_pte(frame, PAGE_KERNEL) : VOID_PTE; + + if (!__set_phys_to_machine(pfn + i, frame)) + return -ENOMEM; + + u[i].ptr = virt_to_machine(ptep + i).maddr | + MMU_NORMAL_PT_UPDATE; + u[i].val = pte_val_ma(pte); + } + + ret = HYPERVISOR_mmu_update(u, batch, NULL, DOMID_SELF); + if (ret < 0) + return ret; + + pfn += batch; + mfn += batch; + count -= batch; + } + + return 0; +} + +/* + * Point a run of @count contiguous pfns from @pfn at the equally contiguous + * machine frames from @mfn. The caller must have run xen_prealloc_p2m_range() + * over the same run. + * + * No TLB flush is done: a run is only ever mapped over PTEs that + * xen_zap_contig_pfns() already cleared, so there is no stale entry to shoot + * down. + */ +int xen_remap_contig_pfns(unsigned long pfn, unsigned long mfn, + unsigned long count) +{ + return xen_set_contig_ptes(pfn, mfn, count, true); +} +EXPORT_SYMBOL_GPL(xen_remap_contig_pfns); + +/* + * Unmap a run of @count contiguous pfns from @pfn and invalidate their p2m + * entries. Writing an invalid entry never needs a new leaf, so unlike the + * remap above this needs no preallocation. The caller flushes the TLB before + * the frames are handed back to Xen. + */ +int xen_zap_contig_pfns(unsigned long pfn, unsigned long count) +{ + return xen_set_contig_ptes(pfn, INVALID_P2M_ENTRY, count, false); +} +EXPORT_SYMBOL_GPL(xen_zap_contig_pfns); + /* * Perform the hypercall to exchange a region of our pfns to point to * memory with the required contiguous alignment. Takes the pfns as diff --git a/drivers/xen/balloon.c b/drivers/xen/balloon.c index 40c55082867ecd..692dca2eccbc58 100644 --- a/drivers/xen/balloon.c +++ b/drivers/xen/balloon.c @@ -107,6 +107,25 @@ static const struct ctl_table balloon_table[] = { */ #define EXTENT_ORDER (fls(XEN_PFN_PER_PAGE) - 1) +/* + * A balloon block is 2 MiB of Xen pages, the largest extent Xen lets a domU + * request for itself: CONFIG_DOMU_MAX_ORDER is PAGETABLE_ORDER on x86. + */ +#define BALLOON_BLOCK_XEN_ORDER 9 +#define BALLOON_BLOCK_ORDER (BALLOON_BLOCK_XEN_ORDER - EXTENT_ORDER) +#define BALLOON_BLOCK_NR_PAGES (1UL << BALLOON_BLOCK_ORDER) + +/* + * Most blocks populated by one deflate. On PV each block also costs a page + * table update per page, batched but done under balloon_mutex. Capping the + * batch keeps a large target change from holding the mutex against a later + * target update, or against the OOM notifier, which can only trylock it. + */ +#define BALLOON_BLOCK_BATCH 32 + +static bool __read_mostly balloon_superpages = true; +module_param(balloon_superpages, bool, 0444); + /* * balloon_thread() state: * @@ -139,11 +158,21 @@ static xen_pfn_t frame_list[PAGE_SIZE / sizeof(xen_pfn_t)]; static LIST_HEAD(ballooned_pages); static DECLARE_WAIT_QUEUE_HEAD(balloon_wq); +/* List of ballooned blocks, threaded through each block's head page. */ +static LIST_HEAD(ballooned_blocks); + /* When ballooning out (allocating memory to return to Xen) we don't really want the kernel to try too hard since that can trigger the oom killer. */ #define GFP_BALLOON \ (GFP_HIGHUSER | __GFP_NOWARN | __GFP_NORETRY | __GFP_NOMEMALLOC) +/* + * A block is only taken when the guest has a whole free 2 MiB run to spare: + * the balloon never reclaims or compacts to make one, and inflates a page at + * a time instead. + */ +#define GFP_BALLOON_BLOCK (GFP_BALLOON & ~__GFP_RECLAIM) + /* balloon_append: add the given page to the balloon. */ static void balloon_append(struct page *page) { @@ -163,12 +192,87 @@ static void balloon_append(struct page *page) wake_up(&balloon_wq); } +/* + * A block is accounted as one unit, which the low/high split of the page list + * cannot express. That only matters with highmem, which the 64-bit + * configurations Xen guests run on do not have. + */ +static bool balloon_blocks_usable(void) +{ + return balloon_superpages && !IS_ENABLED(CONFIG_HIGHMEM); +} + +/* balloon_append_block: park a whole block on the balloon. */ +static void balloon_append_block(struct page *head) +{ + unsigned long i; + + for (i = 0; i < BALLOON_BLOCK_NR_PAGES; i++) + __SetPageOffline(head + i); + + list_add(&head->lru, &ballooned_blocks); + balloon_stats.balloon_blocks++; + mod_node_page_state(page_pgdat(head), NR_BALLOON_PAGES, + BALLOON_BLOCK_NR_PAGES); + + wake_up(&balloon_wq); +} + +/* balloon_retrieve_block: rescue a whole block from the balloon. */ +static struct page *balloon_retrieve_block(void) +{ + struct page *head; + unsigned long i; + + head = list_first_entry_or_null(&ballooned_blocks, struct page, lru); + if (!head) + return NULL; + + list_del(&head->lru); + balloon_stats.balloon_blocks--; + mod_node_page_state(page_pgdat(head), NR_BALLOON_PAGES, + -(long)BALLOON_BLOCK_NR_PAGES); + + for (i = 0; i < BALLOON_BLOCK_NR_PAGES; i++) + __ClearPageOffline(head + i); + + return head; +} + +static struct page *balloon_next_block(struct page *head) +{ + struct list_head *next = head->lru.next; + + if (next == &ballooned_blocks) + return NULL; + return list_entry(next, struct page, lru); +} + +/* + * Spill one block onto the page list so the order-0 paths can consume it. + * This is one way: the pages go back to being ballooned individually, and + * only form a block again if a later inflate allocates the same run whole. + */ +static bool balloon_break_block(void) +{ + struct page *head = balloon_retrieve_block(); + unsigned long i; + + if (!head) + return false; + + for (i = 0; i < BALLOON_BLOCK_NR_PAGES; i++) + balloon_append(head + i); + + return true; +} + /* balloon_retrieve: rescue a page from the balloon, if it is not empty. */ static struct page *balloon_retrieve(bool require_lowmem) { struct page *page; - if (list_empty(&ballooned_pages)) + if (list_empty(&ballooned_pages) && !balloon_break_block()) return NULL; page = list_entry(ballooned_pages.next, struct page, lru); @@ -385,7 +489,8 @@ static long current_credit(void) static bool balloon_is_inflated(void) { - return balloon_stats.balloon_low || balloon_stats.balloon_high; + return balloon_stats.balloon_low || balloon_stats.balloon_high || + balloon_stats.balloon_blocks; } static enum bp_state increase_reservation(unsigned long nr_pages) @@ -397,6 +502,14 @@ static enum bp_state increase_reservation(unsigned long nr_pages) if (nr_pages > ARRAY_SIZE(frame_list)) nr_pages = ARRAY_SIZE(frame_list); + /* + * Memory parked as whole blocks is invisible to the page list walk + * below. Break a block onto the list rather than find nothing to + * populate while the balloon still holds memory. + */ + if (list_empty(&ballooned_pages)) + balloon_break_block(); + page = list_first_entry_or_null(&ballooned_pages, struct page, lru); for (i = 0; i < nr_pages; i++) { if (!page) { @@ -431,6 +544,94 @@ static enum bp_state increase_reservation(unsigned long nr_pages) return BP_DONE; } +/* + * Populate whole parked blocks, one order-9 extent each, so that Xen backs + * each block with a single machine-contiguous 2 MiB run. On PVH Xen maps that + * run with one 2 MiB EPT entry rather than 512 4 KiB ones. + */ +static enum bp_state increase_reservation_blocks(unsigned long nr_blocks) +{ + struct page *head; + unsigned long i, j; + int rc; + + if (nr_blocks > BALLOON_BLOCK_BATCH) + nr_blocks = BALLOON_BLOCK_BATCH; + + head = list_first_entry_or_null(&ballooned_blocks, struct page, lru); + for (i = 0; i < nr_blocks; i++) { + if (!head) { + nr_blocks = i; + break; + } + + /* + * Has to happen before the hypercall: mapping the block in + * below cannot allocate, and once Xen has populated an extent + * there is no way to hand it back. A short batch is fine; what + * is left is deflated on a later pass. + */ + if (xenmem_reservation_p2m_prealloc(BALLOON_BLOCK_NR_PAGES, + head)) { + nr_blocks = i; + break; + } + + frame_list[i] = page_to_xen_pfn(head); + head = balloon_next_block(head); + } + + if (!nr_blocks) + return BP_EAGAIN; + + rc = xenmem_reservation_increase_order(nr_blocks, frame_list, + BALLOON_BLOCK_ORDER); + if (rc <= 0) + return BP_EAGAIN; + + for (i = 0; i < rc; i++) { + head = balloon_retrieve_block(); + BUG_ON(head == NULL); + + /* + * A PV domain gets back only the base machine frame of each + * extent, the rest of the block following it; on PVH there is + * nothing to map. + */ + xenmem_reservation_va_mapping_update_contig( + BALLOON_BLOCK_NR_PAGES, head, frame_list[i]); + + for (j = 0; j < BALLOON_BLOCK_NR_PAGES; j++) + free_reserved_page(head + j); + } + + balloon_stats.current_pages += rc * BALLOON_BLOCK_NR_PAGES; + + return BP_DONE; +} + +/* + * Deflate by whole blocks while Xen can supply a 2 MiB extent, a page at a + * time otherwise. Xen refuses a block when the host has no free 2 MiB run, + * or when the domain is less than 2 MiB below its max_pages ceiling, and + * neither must stop the guest growing by what it can still get. + */ +static enum bp_state balloon_deflate(unsigned long nr_pages) +{ + if (balloon_blocks_usable() && balloon_stats.balloon_blocks && + nr_pages >= BALLOON_BLOCK_NR_PAGES) { + unsigned long before = balloon_stats.current_pages; + enum bp_state state; + + state = increase_reservation_blocks(nr_pages >> + BALLOON_BLOCK_ORDER); + if (balloon_stats.current_pages != before) + return state; + } + + return increase_reservation(nr_pages); +} + static enum bp_state decrease_reservation(unsigned long nr_pages, gfp_t gfp) { enum bp_state state = BP_DONE; @@ -488,6 +689,94 @@ static enum bp_state decrease_reservation(unsigned long nr_pages, gfp_t gfp) return state; } +static enum bp_state decrease_reservation_blocks(unsigned long nr_blocks) +{ + enum bp_state state = BP_DONE; + unsigned long i, j, nr_pages; + struct page *head, *tmp; + LIST_HEAD(blocks); + int ret; + + /* + * One extent per native page even though the frames form whole blocks. + * XENMEM_decrease_reservation releases each frame individually + * whatever the extent order, so an ordered extent would change nothing + * here while asking Xen to trust that the block is machine-contiguous, + * a claim it cannot check for a PV domain, where the frame number it + * is handed is already the machine frame. What lets Xen merge the + * frames back into a 2 MiB buddy is that they are a whole aligned run, + * which they are whenever the block is backed by one machine-contiguous + * allocation. + */ + if (nr_blocks > ARRAY_SIZE(frame_list) / BALLOON_BLOCK_NR_PAGES) + nr_blocks = ARRAY_SIZE(frame_list) / BALLOON_BLOCK_NR_PAGES; + + for (i = 0; i < nr_blocks; i++) { + head = alloc_pages(GFP_BALLOON_BLOCK, BALLOON_BLOCK_ORDER); + if (head == NULL) { + nr_blocks = i; + state = BP_EAGAIN; + break; + } + + split_page(head, BALLOON_BLOCK_ORDER); + adjust_managed_page_count(head, -(long)BALLOON_BLOCK_NR_PAGES); + + for (j = 0; j < BALLOON_BLOCK_NR_PAGES; j++) + xenmem_reservation_scrub_page(head + j); + + list_add(&head->lru, &blocks); + } + + if (!nr_blocks) + return state; + + kmap_flush_unused(); + + i = 0; + list_for_each_entry_safe(head, tmp, &blocks, lru) { + for (j = 0; j < BALLOON_BLOCK_NR_PAGES; j++) + frame_list[i++] = xen_page_to_gfn(head + j); + + xenmem_reservation_va_mapping_reset_contig( + BALLOON_BLOCK_NR_PAGES, head); + + list_del(&head->lru); + + balloon_append_block(head); + } + + flush_tlb_all(); + + nr_pages = nr_blocks * BALLOON_BLOCK_NR_PAGES; + ret = xenmem_reservation_decrease(nr_pages, frame_list); + BUG_ON(ret != nr_pages); + + balloon_stats.current_pages -= nr_pages; + + return state; +} + +/* + * Inflate by whole blocks while the guest has one to spare, a page at a time + * otherwise. A guest with plenty of free memory can still have no whole free + * 2 MiB run left, and that must not stop it shrinking. + */ +static enum bp_state balloon_inflate(unsigned long nr_pages) +{ + if (balloon_blocks_usable() && nr_pages >= BALLOON_BLOCK_NR_PAGES) { + unsigned long before = balloon_stats.current_pages; + enum bp_state state; + + state = decrease_reservation_blocks(nr_pages >> + BALLOON_BLOCK_ORDER); + if (balloon_stats.current_pages != before) + return state; + } + + return decrease_reservation(nr_pages, GFP_BALLOON); +} + /* * Stop waiting if either state is BP_DONE and ballooning action is * needed, or if the credit has changed while state is not BP_DONE. @@ -540,7 +829,7 @@ static int balloon_thread(void *unused) if (credit > 0) { if (balloon_is_inflated()) - balloon_state = increase_reservation(credit); + balloon_state = balloon_deflate(credit); else balloon_state = reserve_additional_memory(); } @@ -549,8 +838,7 @@ static int balloon_thread(void *unused) long n_pages; n_pages = min(-credit, si_mem_available()); - balloon_state = decrease_reservation(n_pages, - GFP_BALLOON); + balloon_state = balloon_inflate(n_pages); if (balloon_state == BP_DONE && n_pages != -credit && n_pages < totalreserve_pages) balloon_state = BP_EAGAIN; @@ -696,8 +984,9 @@ static int balloon_oom_notify(struct notifier_block *nb, unsigned long dummy, if (!mutex_trylock(&balloon_mutex)) return NOTIFY_OK; - /* nr: number of pages parked on the balloon list that we can reclaim */ - nr = balloon_stats.balloon_low + balloon_stats.balloon_high; + /* nr: number of pages parked on the balloon that we can reclaim */ + nr = balloon_stats.balloon_low + balloon_stats.balloon_high + + balloon_stats.balloon_blocks * BALLOON_BLOCK_NR_PAGES; /* clamp, so we don't release all of them at once */ nr = min_t(unsigned long, nr, BALLOON_OOM_DEFLATE_BATCH); @@ -709,10 +998,10 @@ static int balloon_oom_notify(struct notifier_block *nb, unsigned long dummy, /* runs XENMEM_populate_physmap, updates current_pages */ if (nr > 0) - increase_reservation(nr); + balloon_deflate(nr); /* - * increase_reservation() bumped current_pages by however many Xen + * balloon_deflate() bumped current_pages by however many Xen * actually populated (possibly fewer than asked, or zero if the host is * out or we are already at max_pages). Report that to the OOM killer so * it retries instead of killing when we made progress. @@ -795,6 +1084,14 @@ static int __init balloon_init(void) if (!xen_domain()) return -ENODEV; + /* A block must span at least a native page, and be allocatable. */ + BUILD_BUG_ON(BALLOON_BLOCK_XEN_ORDER < EXTENT_ORDER); + BUILD_BUG_ON(BALLOON_BLOCK_ORDER > MAX_PAGE_ORDER); + /* Inflating a block takes one frame_list entry per page. */ + BUILD_BUG_ON(BALLOON_BLOCK_NR_PAGES > ARRAY_SIZE(frame_list)); + /* Deflating takes one frame_list entry per block. */ + BUILD_BUG_ON(BALLOON_BLOCK_BATCH > ARRAY_SIZE(frame_list)); + pr_info("Initialising balloon driver\n"); if (xen_initial_domain()) diff --git a/drivers/xen/mem-reservation.c b/drivers/xen/mem-reservation.c index 24648836e0d445..621606935c132b 100644 --- a/drivers/xen/mem-reservation.c +++ b/drivers/xen/mem-reservation.c @@ -80,14 +80,57 @@ void __xenmem_reservation_va_mapping_reset(unsigned long count, } } EXPORT_SYMBOL_GPL(__xenmem_reservation_va_mapping_reset); + +void __xenmem_reservation_va_mapping_update_contig(unsigned long count, + struct page *page, + xen_pfn_t frame) +{ + BUILD_BUG_ON(XEN_PAGE_SIZE != PAGE_SIZE); + + /* + * Fatal, as in the per-page helper: callers treat the run as mapped + * once this returns, and a page without a mapping faults wherever it + * is next used, far from here. + */ + BUG_ON(xen_remap_contig_pfns(page_to_pfn(page), frame, count)); +} +EXPORT_SYMBOL_GPL(__xenmem_reservation_va_mapping_update_contig); + +void __xenmem_reservation_va_mapping_reset_contig(unsigned long count, + struct page *page) +{ + BUILD_BUG_ON(XEN_PAGE_SIZE != PAGE_SIZE); + + /* + * Unlike the per-page helper, only a warning: a frame left mapped is a + * leak, but Xen will not hand it to another domain while any mapping + * of it remains. + */ + WARN_ON_ONCE(xen_zap_contig_pfns(page_to_pfn(page), count)); +} +EXPORT_SYMBOL_GPL(__xenmem_reservation_va_mapping_reset_contig); + +int __xenmem_reservation_p2m_prealloc(unsigned long count, struct page *page) +{ + BUILD_BUG_ON(XEN_PAGE_SIZE != PAGE_SIZE); + + return xen_prealloc_p2m_range(page_to_pfn(page), count); +} +EXPORT_SYMBOL_GPL(__xenmem_reservation_p2m_prealloc); #endif /* CONFIG_XEN_HAVE_PVMMU */ -/* @frames is an array of PFNs */ -int xenmem_reservation_increase(int count, xen_pfn_t *frames) +/* + * @frames is an array of PFNs, one per extent. Each extent covers + * 1 << @order native pages. For a PV domain Xen reports back only the base + * machine frame of each extent; the rest of the extent follows it + * contiguously. + */ +int xenmem_reservation_increase_order(int count, xen_pfn_t *frames, + unsigned int order) { struct xen_memory_reservation reservation = { .address_bits = 0, - .extent_order = EXTENT_ORDER, + .extent_order = EXTENT_ORDER + order, .domid = DOMID_SELF }; @@ -96,6 +139,13 @@ int xenmem_reservation_increase(int count, xen_pfn_t *frames) reservation.nr_extents = count; return HYPERVISOR_memory_op(XENMEM_populate_physmap, &reservation); } +EXPORT_SYMBOL_GPL(xenmem_reservation_increase_order); + +/* @frames is an array of PFNs */ +int xenmem_reservation_increase(int count, xen_pfn_t *frames) +{ + return xenmem_reservation_increase_order(count, frames, 0); +} EXPORT_SYMBOL_GPL(xenmem_reservation_increase); /* @frames is an array of GFNs */ diff --git a/drivers/xen/xen-balloon.c b/drivers/xen/xen-balloon.c index b293d7652f1559..bf538f12781261 100644 --- a/drivers/xen/xen-balloon.c +++ b/drivers/xen/xen-balloon.c @@ -145,6 +145,7 @@ EXPORT_SYMBOL_GPL(xen_balloon_init); BALLOON_SHOW(current_kb, "%lu\n", PAGES2KB(balloon_stats.current_pages)); BALLOON_SHOW(low_kb, "%lu\n", PAGES2KB(balloon_stats.balloon_low)); BALLOON_SHOW(high_kb, "%lu\n", PAGES2KB(balloon_stats.balloon_high)); +BALLOON_SHOW(blocks, "%lu\n", balloon_stats.balloon_blocks); static DEVICE_ULONG_ATTR(schedule_delay, 0444, balloon_stats.schedule_delay); static DEVICE_ULONG_ATTR(max_schedule_delay, 0644, balloon_stats.max_schedule_delay); @@ -223,6 +224,7 @@ static struct attribute *balloon_info_attrs[] = { &dev_attr_current_kb.attr, &dev_attr_low_kb.attr, &dev_attr_high_kb.attr, + &dev_attr_blocks.attr, NULL }; diff --git a/include/xen/balloon.h b/include/xen/balloon.h index f78a6cc94f1aea..4702026337e554 100644 --- a/include/xen/balloon.h +++ b/include/xen/balloon.h @@ -15,6 +15,8 @@ struct balloon_stats { /* Number of pages in high- and low-memory balloons. */ unsigned long balloon_low; unsigned long balloon_high; + /* Number of whole superpage blocks parked on the balloon. */ + unsigned long balloon_blocks; unsigned long total_pages; unsigned long schedule_delay; unsigned long max_schedule_delay; diff --git a/include/xen/mem-reservation.h b/include/xen/mem-reservation.h index 3cbe3df0dfd4e9..5f481259ae1501 100644 --- a/include/xen/mem-reservation.h +++ b/include/xen/mem-reservation.h @@ -32,6 +32,15 @@ void __xenmem_reservation_va_mapping_update(unsigned long count, void __xenmem_reservation_va_mapping_reset(unsigned long count, struct page **pages); + +void __xenmem_reservation_va_mapping_update_contig(unsigned long count, + struct page *page, + xen_pfn_t frame); + +void __xenmem_reservation_va_mapping_reset_contig(unsigned long count, + struct page *page); + +int __xenmem_reservation_p2m_prealloc(unsigned long count, struct page *page); #endif static inline void xenmem_reservation_va_mapping_update(unsigned long count, @@ -53,8 +62,53 @@ static inline void xenmem_reservation_va_mapping_reset(unsigned long count, #endif } +/* + * Prepare a run of @count physically contiguous pages starting at @page to be + * mapped by xenmem_reservation_va_mapping_update_contig(), which cannot + * allocate. Call this before asking Xen to populate the run: it is the only + * step that can fail, and failing it after the populate would strand the + * frames Xen handed over. + */ +static inline int xenmem_reservation_p2m_prealloc(unsigned long count, + struct page *page) +{ +#ifdef CONFIG_XEN_HAVE_PVMMU + if (xen_pv_domain()) + return __xenmem_reservation_p2m_prealloc(count, page); +#endif + return 0; +} + +/* + * Variants for a run of @count physically contiguous pages starting at @page, + * backed by the equally contiguous machine frames starting at @frame. Xen + * reports only the base frame of a multi-page extent, so the ordered populate + * path has no frame array to pass. + */ +static inline void xenmem_reservation_va_mapping_update_contig(unsigned long count, + struct page *page, + xen_pfn_t frame) +{ +#ifdef CONFIG_XEN_HAVE_PVMMU + if (xen_pv_domain()) + __xenmem_reservation_va_mapping_update_contig(count, page, frame); +#endif +} + +static inline void xenmem_reservation_va_mapping_reset_contig(unsigned long count, + struct page *page) +{ +#ifdef CONFIG_XEN_HAVE_PVMMU + if (xen_pv_domain()) + __xenmem_reservation_va_mapping_reset_contig(count, page); +#endif +} + int xenmem_reservation_increase(int count, xen_pfn_t *frames); +int xenmem_reservation_increase_order(int count, xen_pfn_t *frames, + unsigned int order); + int xenmem_reservation_decrease(int count, xen_pfn_t *frames); #endif