2007-07-26 10:41:02 -07:00
|
|
|
/*P:400 This contains run_guest() which actually calls into the Host<->Guest
|
|
|
|
* Switcher and analyzes the return, such as determining if the Guest wants the
|
|
|
|
* Host to do something. This file also contains useful helper routines, and a
|
|
|
|
* couple of non-obvious setup and teardown pieces which were implemented after
|
|
|
|
* days of debugging pain. :*/
|
2007-07-19 01:49:23 -07:00
|
|
|
#include <linux/module.h>
|
|
|
|
#include <linux/stringify.h>
|
|
|
|
#include <linux/stddef.h>
|
|
|
|
#include <linux/io.h>
|
|
|
|
#include <linux/mm.h>
|
|
|
|
#include <linux/vmalloc.h>
|
|
|
|
#include <linux/cpu.h>
|
|
|
|
#include <linux/freezer.h>
|
2007-10-22 11:03:28 +10:00
|
|
|
#include <linux/highmem.h>
|
2007-07-19 01:49:23 -07:00
|
|
|
#include <asm/paravirt.h>
|
|
|
|
#include <asm/pgtable.h>
|
|
|
|
#include <asm/uaccess.h>
|
|
|
|
#include <asm/poll.h>
|
|
|
|
#include <asm/asm-offsets.h>
|
|
|
|
#include "lg.h"
|
|
|
|
|
|
|
|
|
|
|
|
static struct vm_struct *switcher_vma;
|
|
|
|
static struct page **switcher_page;
|
|
|
|
|
|
|
|
/* This One Big lock protects all inter-guest data structures. */
|
|
|
|
DEFINE_MUTEX(lguest_lock);
|
|
|
|
|
2007-07-26 10:41:04 -07:00
|
|
|
/*H:010 We need to set up the Switcher at a high virtual address. Remember the
|
|
|
|
* Switcher is a few hundred bytes of assembler code which actually changes the
|
|
|
|
* CPU to run the Guest, and then changes back to the Host when a trap or
|
|
|
|
* interrupt happens.
|
|
|
|
*
|
|
|
|
* The Switcher code must be at the same virtual address in the Guest as the
|
|
|
|
* Host since it will be running as the switchover occurs.
|
|
|
|
*
|
|
|
|
* Trying to map memory at a particular address is an unusual thing to do, so
|
2007-10-22 11:03:28 +10:00
|
|
|
* it's not a simple one-liner. */
|
2007-07-19 01:49:23 -07:00
|
|
|
static __init int map_switcher(void)
|
|
|
|
{
|
|
|
|
int i, err;
|
|
|
|
struct page **pagep;
|
|
|
|
|
2007-07-26 10:41:04 -07:00
|
|
|
/*
|
|
|
|
* Map the Switcher in to high memory.
|
|
|
|
*
|
|
|
|
* It turns out that if we choose the address 0xFFC00000 (4MB under the
|
|
|
|
* top virtual address), it makes setting up the page tables really
|
|
|
|
* easy.
|
|
|
|
*/
|
|
|
|
|
|
|
|
/* We allocate an array of "struct page"s. map_vm_area() wants the
|
|
|
|
* pages in this form, rather than just an array of pointers. */
|
2007-07-19 01:49:23 -07:00
|
|
|
switcher_page = kmalloc(sizeof(switcher_page[0])*TOTAL_SWITCHER_PAGES,
|
|
|
|
GFP_KERNEL);
|
|
|
|
if (!switcher_page) {
|
|
|
|
err = -ENOMEM;
|
|
|
|
goto out;
|
|
|
|
}
|
|
|
|
|
2007-07-26 10:41:04 -07:00
|
|
|
/* Now we actually allocate the pages. The Guest will see these pages,
|
|
|
|
* so we make sure they're zeroed. */
|
2007-07-19 01:49:23 -07:00
|
|
|
for (i = 0; i < TOTAL_SWITCHER_PAGES; i++) {
|
|
|
|
unsigned long addr = get_zeroed_page(GFP_KERNEL);
|
|
|
|
if (!addr) {
|
|
|
|
err = -ENOMEM;
|
|
|
|
goto free_some_pages;
|
|
|
|
}
|
|
|
|
switcher_page[i] = virt_to_page(addr);
|
|
|
|
}
|
|
|
|
|
2007-07-26 10:41:04 -07:00
|
|
|
/* Now we reserve the "virtual memory area" we want: 0xFFC00000
|
|
|
|
* (SWITCHER_ADDR). We might not get it in theory, but in practice
|
|
|
|
* it's worked so far. */
|
2007-07-19 01:49:23 -07:00
|
|
|
switcher_vma = __get_vm_area(TOTAL_SWITCHER_PAGES * PAGE_SIZE,
|
|
|
|
VM_ALLOC, SWITCHER_ADDR, VMALLOC_END);
|
|
|
|
if (!switcher_vma) {
|
|
|
|
err = -ENOMEM;
|
|
|
|
printk("lguest: could not map switcher pages high\n");
|
|
|
|
goto free_pages;
|
|
|
|
}
|
|
|
|
|
2007-07-26 10:41:04 -07:00
|
|
|
/* This code actually sets up the pages we've allocated to appear at
|
|
|
|
* SWITCHER_ADDR. map_vm_area() takes the vma we allocated above, the
|
|
|
|
* kind of pages we're mapping (kernel pages), and a pointer to our
|
|
|
|
* array of struct pages. It increments that pointer, but we don't
|
|
|
|
* care. */
|
2007-07-19 01:49:23 -07:00
|
|
|
pagep = switcher_page;
|
|
|
|
err = map_vm_area(switcher_vma, PAGE_KERNEL, &pagep);
|
|
|
|
if (err) {
|
|
|
|
printk("lguest: map_vm_area failed: %i\n", err);
|
|
|
|
goto free_vma;
|
|
|
|
}
|
2007-07-26 10:41:04 -07:00
|
|
|
|
2007-10-22 11:03:28 +10:00
|
|
|
/* Now the Switcher is mapped at the right address, we can't fail!
|
|
|
|
* Copy in the compiled-in Switcher code (from <arch>_switcher.S). */
|
2007-07-19 01:49:23 -07:00
|
|
|
memcpy(switcher_vma->addr, start_switcher_text,
|
|
|
|
end_switcher_text - start_switcher_text);
|
|
|
|
|
|
|
|
printk(KERN_INFO "lguest: mapped switcher at %p\n",
|
|
|
|
switcher_vma->addr);
|
2007-07-26 10:41:04 -07:00
|
|
|
/* And we succeeded... */
|
2007-07-19 01:49:23 -07:00
|
|
|
return 0;
|
|
|
|
|
|
|
|
free_vma:
|
|
|
|
vunmap(switcher_vma->addr);
|
|
|
|
free_pages:
|
|
|
|
i = TOTAL_SWITCHER_PAGES;
|
|
|
|
free_some_pages:
|
|
|
|
for (--i; i >= 0; i--)
|
|
|
|
__free_pages(switcher_page[i], 0);
|
|
|
|
kfree(switcher_page);
|
|
|
|
out:
|
|
|
|
return err;
|
|
|
|
}
|
2007-07-26 10:41:04 -07:00
|
|
|
/*:*/
|
2007-07-19 01:49:23 -07:00
|
|
|
|
2007-07-26 10:41:04 -07:00
|
|
|
/* Cleaning up the mapping when the module is unloaded is almost...
|
|
|
|
* too easy. */
|
2007-07-19 01:49:23 -07:00
|
|
|
static void unmap_switcher(void)
|
|
|
|
{
|
|
|
|
unsigned int i;
|
|
|
|
|
2007-07-26 10:41:04 -07:00
|
|
|
/* vunmap() undoes *both* map_vm_area() and __get_vm_area(). */
|
2007-07-19 01:49:23 -07:00
|
|
|
vunmap(switcher_vma->addr);
|
2007-07-26 10:41:04 -07:00
|
|
|
/* Now we just need to free the pages we copied the switcher into */
|
2007-07-19 01:49:23 -07:00
|
|
|
for (i = 0; i < TOTAL_SWITCHER_PAGES; i++)
|
|
|
|
__free_pages(switcher_page[i], 0);
|
|
|
|
}
|
|
|
|
|
2007-07-26 10:41:03 -07:00
|
|
|
/*L:305
|
|
|
|
* Dealing With Guest Memory.
|
|
|
|
*
|
|
|
|
* When the Guest gives us (what it thinks is) a physical address, we can use
|
2007-10-22 11:03:26 +10:00
|
|
|
* the normal copy_from_user() & copy_to_user() on the corresponding place in
|
|
|
|
* the memory region allocated by the Launcher.
|
2007-07-26 10:41:03 -07:00
|
|
|
*
|
|
|
|
* But we can't trust the Guest: it might be trying to access the Launcher
|
|
|
|
* code. We have to check that the range is below the pfn_limit the Launcher
|
|
|
|
* gave us. We have to make sure that addr + len doesn't give us a false
|
|
|
|
* positive by overflowing, too. */
|
2007-07-19 01:49:23 -07:00
|
|
|
int lguest_address_ok(const struct lguest *lg,
|
|
|
|
unsigned long addr, unsigned long len)
|
|
|
|
{
|
|
|
|
return (addr+len) / PAGE_SIZE < lg->pfn_limit && (addr+len >= addr);
|
|
|
|
}
|
|
|
|
|
2007-07-26 10:41:03 -07:00
|
|
|
/* This is a convenient routine to get a 32-bit value from the Guest (a very
|
|
|
|
* common operation). Here we can see how useful the kill_lguest() routine we
|
|
|
|
* met in the Launcher can be: we return a random value (0) instead of needing
|
|
|
|
* to return an error. */
|
2007-07-19 01:49:23 -07:00
|
|
|
u32 lgread_u32(struct lguest *lg, unsigned long addr)
|
|
|
|
{
|
|
|
|
u32 val = 0;
|
|
|
|
|
2007-07-26 10:41:03 -07:00
|
|
|
/* Don't let them access lguest binary. */
|
2007-07-19 01:49:23 -07:00
|
|
|
if (!lguest_address_ok(lg, addr, sizeof(val))
|
2007-10-22 11:03:26 +10:00
|
|
|
|| get_user(val, (u32 *)(lg->mem_base + addr)) != 0)
|
|
|
|
kill_guest(lg, "bad read address %#lx: pfn_limit=%u membase=%p", addr, lg->pfn_limit, lg->mem_base);
|
2007-07-19 01:49:23 -07:00
|
|
|
return val;
|
|
|
|
}
|
|
|
|
|
2007-07-26 10:41:03 -07:00
|
|
|
/* Same thing for writing a value. */
|
2007-07-19 01:49:23 -07:00
|
|
|
void lgwrite_u32(struct lguest *lg, unsigned long addr, u32 val)
|
|
|
|
{
|
|
|
|
if (!lguest_address_ok(lg, addr, sizeof(val))
|
2007-10-22 11:03:26 +10:00
|
|
|
|| put_user(val, (u32 *)(lg->mem_base + addr)) != 0)
|
2007-07-19 01:49:23 -07:00
|
|
|
kill_guest(lg, "bad write address %#lx", addr);
|
|
|
|
}
|
|
|
|
|
2007-07-26 10:41:03 -07:00
|
|
|
/* This routine is more generic, and copies a range of Guest bytes into a
|
|
|
|
* buffer. If the copy_from_user() fails, we fill the buffer with zeroes, so
|
|
|
|
* the caller doesn't end up using uninitialized kernel memory. */
|
2007-07-19 01:49:23 -07:00
|
|
|
void lgread(struct lguest *lg, void *b, unsigned long addr, unsigned bytes)
|
|
|
|
{
|
|
|
|
if (!lguest_address_ok(lg, addr, bytes)
|
2007-10-22 11:03:26 +10:00
|
|
|
|| copy_from_user(b, lg->mem_base + addr, bytes) != 0) {
|
2007-07-19 01:49:23 -07:00
|
|
|
/* copy_from_user should do this, but as we rely on it... */
|
|
|
|
memset(b, 0, bytes);
|
|
|
|
kill_guest(lg, "bad read address %#lx len %u", addr, bytes);
|
|
|
|
}
|
|
|
|
}
|
|
|
|
|
2007-07-26 10:41:03 -07:00
|
|
|
/* Similarly, our generic routine to copy into a range of Guest bytes. */
|
2007-07-19 01:49:23 -07:00
|
|
|
void lgwrite(struct lguest *lg, unsigned long addr, const void *b,
|
|
|
|
unsigned bytes)
|
|
|
|
{
|
|
|
|
if (!lguest_address_ok(lg, addr, bytes)
|
2007-10-22 11:03:26 +10:00
|
|
|
|| copy_to_user(lg->mem_base + addr, b, bytes) != 0)
|
2007-07-19 01:49:23 -07:00
|
|
|
kill_guest(lg, "bad write address %#lx len %u", addr, bytes);
|
|
|
|
}
|
2007-07-26 10:41:03 -07:00
|
|
|
/* (end of memory access helper routines) :*/
|
2007-07-19 01:49:23 -07:00
|
|
|
|
2007-07-26 10:41:04 -07:00
|
|
|
/*H:030 Let's jump straight to the the main loop which runs the Guest.
|
|
|
|
* Remember, this is called by the Launcher reading /dev/lguest, and we keep
|
|
|
|
* going around and around until something interesting happens. */
|
2007-07-19 01:49:23 -07:00
|
|
|
int run_guest(struct lguest *lg, unsigned long __user *user)
|
|
|
|
{
|
2007-07-26 10:41:04 -07:00
|
|
|
/* We stop running once the Guest is dead. */
|
2007-07-19 01:49:23 -07:00
|
|
|
while (!lg->dead) {
|
2007-10-22 11:03:30 +10:00
|
|
|
/* First we run any hypercalls the Guest wants done. */
|
|
|
|
if (lg->hcall)
|
|
|
|
do_hypercalls(lg);
|
|
|
|
|
2007-10-22 11:24:10 +10:00
|
|
|
/* It's possible the Guest did a NOTIFY hypercall to the
|
2007-07-26 10:41:04 -07:00
|
|
|
* Launcher, in which case we return from the read() now. */
|
2007-10-22 11:24:10 +10:00
|
|
|
if (lg->pending_notify) {
|
|
|
|
if (put_user(lg->pending_notify, user))
|
2007-07-19 01:49:23 -07:00
|
|
|
return -EFAULT;
|
2007-10-22 11:24:10 +10:00
|
|
|
return sizeof(lg->pending_notify);
|
2007-07-19 01:49:23 -07:00
|
|
|
}
|
|
|
|
|
2007-07-26 10:41:04 -07:00
|
|
|
/* Check for signals */
|
2007-07-19 01:49:23 -07:00
|
|
|
if (signal_pending(current))
|
|
|
|
return -ERESTARTSYS;
|
|
|
|
|
|
|
|
/* If Waker set break_out, return to Launcher. */
|
|
|
|
if (lg->break_out)
|
|
|
|
return -EAGAIN;
|
|
|
|
|
2007-07-26 10:41:04 -07:00
|
|
|
/* Check if there are any interrupts which can be delivered
|
|
|
|
* now: if so, this sets up the hander to be executed when we
|
|
|
|
* next run the Guest. */
|
2007-07-19 01:49:23 -07:00
|
|
|
maybe_do_interrupt(lg);
|
|
|
|
|
2007-07-26 10:41:04 -07:00
|
|
|
/* All long-lived kernel loops need to check with this horrible
|
|
|
|
* thing called the freezer. If the Host is trying to suspend,
|
|
|
|
* it stops us. */
|
2007-07-19 01:49:23 -07:00
|
|
|
try_to_freeze();
|
|
|
|
|
2007-07-26 10:41:04 -07:00
|
|
|
/* Just make absolutely sure the Guest is still alive. One of
|
|
|
|
* those hypercalls could have been fatal, for example. */
|
2007-07-19 01:49:23 -07:00
|
|
|
if (lg->dead)
|
|
|
|
break;
|
|
|
|
|
2007-07-26 10:41:04 -07:00
|
|
|
/* If the Guest asked to be stopped, we sleep. The Guest's
|
|
|
|
* clock timer or LHCALL_BREAK from the Waker will wake us. */
|
2007-07-19 01:49:23 -07:00
|
|
|
if (lg->halted) {
|
|
|
|
set_current_state(TASK_INTERRUPTIBLE);
|
|
|
|
schedule();
|
|
|
|
continue;
|
|
|
|
}
|
|
|
|
|
2007-07-26 10:41:04 -07:00
|
|
|
/* OK, now we're ready to jump into the Guest. First we put up
|
|
|
|
* the "Do Not Disturb" sign: */
|
2007-07-19 01:49:23 -07:00
|
|
|
local_irq_disable();
|
|
|
|
|
2007-10-22 11:03:28 +10:00
|
|
|
/* Actually run the Guest until something happens. */
|
|
|
|
lguest_arch_run_guest(lg);
|
2007-07-26 10:41:04 -07:00
|
|
|
|
|
|
|
/* Now we're ready to be interrupted or moved to other CPUs */
|
2007-07-19 01:49:23 -07:00
|
|
|
local_irq_enable();
|
|
|
|
|
2007-10-22 11:03:28 +10:00
|
|
|
/* Now we deal with whatever happened to the Guest. */
|
|
|
|
lguest_arch_handle_trap(lg);
|
2007-07-19 01:49:23 -07:00
|
|
|
}
|
2007-10-22 11:03:28 +10:00
|
|
|
|
2007-07-26 10:41:04 -07:00
|
|
|
/* The Guest is dead => "No such file or directory" */
|
2007-07-19 01:49:23 -07:00
|
|
|
return -ENOENT;
|
|
|
|
}
|
|
|
|
|
2007-07-26 10:41:04 -07:00
|
|
|
/*H:000
|
|
|
|
* Welcome to the Host!
|
|
|
|
*
|
|
|
|
* By this point your brain has been tickled by the Guest code and numbed by
|
|
|
|
* the Launcher code; prepare for it to be stretched by the Host code. This is
|
|
|
|
* the heart. Let's begin at the initialization routine for the Host's lg
|
|
|
|
* module.
|
|
|
|
*/
|
2007-07-19 01:49:23 -07:00
|
|
|
static int __init init(void)
|
|
|
|
{
|
|
|
|
int err;
|
|
|
|
|
2007-07-26 10:41:04 -07:00
|
|
|
/* Lguest can't run under Xen, VMI or itself. It does Tricky Stuff. */
|
2007-07-19 01:49:23 -07:00
|
|
|
if (paravirt_enabled()) {
|
2007-10-16 11:51:29 -07:00
|
|
|
printk("lguest is afraid of %s\n", pv_info.name);
|
2007-07-19 01:49:23 -07:00
|
|
|
return -EPERM;
|
|
|
|
}
|
|
|
|
|
2007-07-26 10:41:04 -07:00
|
|
|
/* First we put the Switcher up in very high virtual memory. */
|
2007-07-19 01:49:23 -07:00
|
|
|
err = map_switcher();
|
|
|
|
if (err)
|
2007-10-22 11:03:35 +10:00
|
|
|
goto out;
|
2007-07-19 01:49:23 -07:00
|
|
|
|
2007-07-26 10:41:04 -07:00
|
|
|
/* Now we set up the pagetable implementation for the Guests. */
|
2007-07-19 01:49:23 -07:00
|
|
|
err = init_pagetables(switcher_page, SHARED_SWITCHER_PAGES);
|
2007-10-22 11:03:35 +10:00
|
|
|
if (err)
|
|
|
|
goto unmap;
|
2007-07-26 10:41:04 -07:00
|
|
|
|
2007-10-22 11:03:35 +10:00
|
|
|
/* We might need to reserve an interrupt vector. */
|
|
|
|
err = init_interrupts();
|
|
|
|
if (err)
|
|
|
|
goto free_pgtables;
|
|
|
|
|
2007-07-26 10:41:04 -07:00
|
|
|
/* /dev/lguest needs to be registered. */
|
2007-07-19 01:49:23 -07:00
|
|
|
err = lguest_device_init();
|
2007-10-22 11:03:35 +10:00
|
|
|
if (err)
|
|
|
|
goto free_interrupts;
|
2007-07-26 10:41:04 -07:00
|
|
|
|
2007-10-22 11:03:28 +10:00
|
|
|
/* Finally we do some architecture-specific setup. */
|
|
|
|
lguest_arch_host_init();
|
2007-07-26 10:41:04 -07:00
|
|
|
|
|
|
|
/* All good! */
|
2007-07-19 01:49:23 -07:00
|
|
|
return 0;
|
2007-10-22 11:03:35 +10:00
|
|
|
|
|
|
|
free_interrupts:
|
|
|
|
free_interrupts();
|
|
|
|
free_pgtables:
|
|
|
|
free_pagetables();
|
|
|
|
unmap:
|
|
|
|
unmap_switcher();
|
|
|
|
out:
|
|
|
|
return err;
|
2007-07-19 01:49:23 -07:00
|
|
|
}
|
|
|
|
|
2007-07-26 10:41:04 -07:00
|
|
|
/* Cleaning up is just the same code, backwards. With a little French. */
|
2007-07-19 01:49:23 -07:00
|
|
|
static void __exit fini(void)
|
|
|
|
{
|
|
|
|
lguest_device_remove();
|
2007-10-22 11:03:35 +10:00
|
|
|
free_interrupts();
|
2007-07-19 01:49:23 -07:00
|
|
|
free_pagetables();
|
|
|
|
unmap_switcher();
|
2007-07-26 10:41:04 -07:00
|
|
|
|
2007-10-22 11:03:28 +10:00
|
|
|
lguest_arch_host_fini();
|
2007-07-19 01:49:23 -07:00
|
|
|
}
|
2007-10-22 11:03:28 +10:00
|
|
|
/*:*/
|
2007-07-19 01:49:23 -07:00
|
|
|
|
2007-07-26 10:41:04 -07:00
|
|
|
/* The Host side of lguest can be a module. This is a nice way for people to
|
|
|
|
* play with it. */
|
2007-07-19 01:49:23 -07:00
|
|
|
module_init(init);
|
|
|
|
module_exit(fini);
|
|
|
|
MODULE_LICENSE("GPL");
|
|
|
|
MODULE_AUTHOR("Rusty Russell <rusty@rustcorp.com.au>");
|