From 4d919d65098bfcb5a41f8d8d3bc418b592afbd44 Mon Sep 17 00:00:00 2001 From: Nick Hudson Date: Sat, 10 Oct 2026 20:18:08 +0100 Subject: [PATCH 13/13] WIP interrupt --- accel/nvmm/nvmm-accel-ops.c | 2 +- accel/nvmm/nvmm-all.c | 111 ++++++------- accel/nvmm/trace-events | 43 +++++ accel/nvmm/trace.h | 1 + include/system/nvmm_int.h | 210 +++++++++++++++++++++++ meson.build | 2 + target/arm/nvmm/nvmm.c | 311 +++++++++++++++++++++++++++-------- target/arm/nvmm/trace-events | 22 +++ target/arm/nvmm/trace.h | 1 + 9 files changed, 566 insertions(+), 137 deletions(-) create mode 100644 accel/nvmm/trace-events create mode 100644 accel/nvmm/trace.h create mode 100644 include/system/nvmm_int.h create mode 100644 target/arm/nvmm/trace-events create mode 100644 target/arm/nvmm/trace.h diff --git a/accel/nvmm/nvmm-accel-ops.c b/accel/nvmm/nvmm-accel-ops.c index 16adf08b38..349b025770 100644 --- a/accel/nvmm/nvmm-accel-ops.c +++ b/accel/nvmm/nvmm-accel-ops.c @@ -26,7 +26,7 @@ static void *qemu_nvmm_cpu_thread_fn(void *arg) rcu_register_thread(); bql_lock(); - qemu_thread_get_self(cpu->thread); + qemu_thread_get_self(cpu->thread); //XXXNH? cpu->thread_id = qemu_get_thread_id(); current_cpu = cpu; diff --git a/accel/nvmm/nvmm-all.c b/accel/nvmm/nvmm-all.c index fa778e9624..223bd960d2 100644 --- a/accel/nvmm/nvmm-all.c +++ b/accel/nvmm/nvmm-all.c @@ -16,6 +16,7 @@ #include "qapi/error.h" #include "system/nvmm.h" #include "qemu/error-report.h" +#include "trace.h" struct qemu_machine { struct nvmm_capability cap; @@ -65,37 +66,13 @@ static void nvmm_mem_callback(struct nvmm_mem *mem) { -// address_space_rw(&address_space_memory, addr, MEMTXATTRS_UNSPECIFIED, -// val, req->size, rw); -#if 0 - trace_kvm_run_exit(cpu->cpu_index, run->exit_reason); - switch (run->exit_reason) { - case KVM_EXIT_IO: - /* Called outside BQL */ - kvm_handle_io(run->io.port, attrs, - (uint8_t *)run + run->io.data_offset, - run->io.direction, - run->io.size, - run->io.count); - ret = 0; - break; - case KVM_EXIT_MMIO: - /* Called outside BQL */ - address_space_rw(&address_space_memory, - run->mmio.phys_addr, attrs, - run->mmio.data, - run->mmio.len, - run->mmio.is_write); - ret = 0; - break; -#endif -// cpu_physical_memory_rw(mem->gpa, mem->data, mem->size, mem->write); + trace_nvmm_run_exit(mem->vcpu->cpuid, NVMM_VCPU_EXIT_MEMORY); - MemTxAttrs attrs = MEMTXATTRS_UNSPECIFIED; // { 0 }; + trace_nvmm_memory_access(mem->vcpu->cpuid, mem->gpa, mem->size, mem->write); + MemTxAttrs attrs = MEMTXATTRS_UNSPECIFIED; address_space_rw(&address_space_memory, - mem->gpa, - attrs, + mem->gpa, attrs, mem->data, mem->size, mem->write); /* Needed, otherwise infinite loop. */ @@ -199,13 +176,33 @@ nvmm_vcpu_exec(CPUState *cpu) { int ret, fatal; + trace_nvmm_cpu_exec(); + +#if 0 + if (cpu->halted) { + return EXCP_HLT; + } + + bql_unlock(); + cpu_exec_start(cpu); +#endif + while (true) { + +#if 0 + if (!(cpu->singlestep_enabled & SSTEP_NOIRQ) && hvf_inject_interrupts(cpu)) { + return EXCP_INTERRUPT; + } +#endif + // XXXNH this is needed to exit QEMU if (cpu->exception_index >= EXCP_INTERRUPT) { + // Inject interrupts!!! or many it's in nvmm_vcpu_loop ret = cpu->exception_index; cpu->exception_index = -1; break; } + // XXXNH arch dependant loop fatal = nvmm_vcpu_loop(cpu); if (fatal) { @@ -213,6 +210,10 @@ nvmm_vcpu_exec(CPUState *cpu) abort(); } } +#if 0 + cpu_exec_end(); + bql_lock(); +#endif return ret; } @@ -224,6 +225,8 @@ nvmm_update_mapping(hwaddr start_pa, ram_addr_t size, uintptr_t hva, struct nvmm_machine *mach = get_nvmm_mach(); int ret, prot; + trace_nvmm_mapping(start_pa, size, hva, add, rom, name); + if (add) { prot = PROT_READ | PROT_EXEC; if (!rom) { @@ -232,6 +235,8 @@ nvmm_update_mapping(hwaddr start_pa, ram_addr_t size, uintptr_t hva, ret = nvmm_gpa_map(mach, hva, start_pa, size, prot); } else { ret = nvmm_gpa_unmap(mach, hva, start_pa, size); + if (ret < 0 && errno == ESRCH) + return; } if (ret == -1) { @@ -243,7 +248,7 @@ nvmm_update_mapping(hwaddr start_pa, ram_addr_t size, uintptr_t hva, } static void -nvmm_process_section(MemoryRegionSection *section, int add) +nvmm_process_section(MemoryRegionSection *section, bool add) { MemoryRegion *mr = section->mr; hwaddr start_pa = section->offset_within_address_space; @@ -252,26 +257,18 @@ nvmm_process_section(MemoryRegionSection *section, int add) unsigned int delta; uintptr_t hva; - if (!memory_region_is_ram(mr)) { - if (writable /* || !kvm_readonly_mem_allowed */) { + if (!memory_region_is_ram(mr)) { + if (writable) { return; - } else if (!mr->romd_mode) { - /* If the memory device is not in romd_mode, then we actually want - * to remove the kvm memory slot so all accesses will trap. */ - return; -// add = false; + } else if (!memory_region_is_romd(mr)) { + /* + * If the memory device is not in romd_mode, then we actually want + * to remove the hvf memory slot so all accesses will trap. + */ + add = false; } } - - - -#if 0 - if (!memory_region_is_ram(mr)) { - return; - } -#endif - /* Adjust start_pa and size so that they are page-aligned. */ delta = qemu_real_host_page_size() - (start_pa & ~qemu_real_host_page_mask()); delta &= ~qemu_real_host_page_mask(); @@ -288,36 +285,22 @@ nvmm_process_section(MemoryRegionSection *section, int add) hva = (uintptr_t)memory_region_get_ram_ptr(mr) + section->offset_within_region + delta; -printf("%s:%d pa %#lx size %#lx hva %#lx %s rom/'%s'\n", - __func__, __LINE__, start_pa, size, hva, memory_region_is_rom(mr) ? "is" : "not", mr->name); - nvmm_update_mapping(start_pa, size, hva, add, - memory_region_is_rom(mr), mr->name); + bool rom = mr->readonly || + (!memory_region_is_ram(mr) && memory_region_is_romd(mr)); + nvmm_update_mapping(start_pa, size, hva, add, rom, mr->name); } static void nvmm_region_add(MemoryListener *listener, MemoryRegionSection *section) { memory_region_ref(section->mr); - nvmm_process_section(section, 1); - -#if 0 - if (!memory_region_is_ram(mr)) { - if (writable || !kvm_readonly_mem_allowed) { - return; - } else if (!mr->romd_mode) { - /* If the memory device is not in romd_mode, then we actually want - * to remove the kvm memory slot so all accesses will trap. */ - add = false; - } - } -#endif - + nvmm_process_section(section, /* add = */ true); } static void nvmm_region_del(MemoryListener *listener, MemoryRegionSection *section) { - nvmm_process_section(section, 0); + nvmm_process_section(section, /* add = */ false); memory_region_unref(section->mr); } diff --git a/accel/nvmm/trace-events b/accel/nvmm/trace-events new file mode 100644 index 0000000000..a6f50e678b --- /dev/null +++ b/accel/nvmm/trace-events @@ -0,0 +1,43 @@ +# See docs/devel/tracing.rst for syntax documentation. + +# nvmm-all.c + +nvmm_cpu_exec(void) "" +nvmm_run_exit(int cpu_index, uint32_t reason) "cpu_index %d, reason %d" +nvmm_mapping(uint64_t start_pa, uint64_t size, uintptr_t hva, bool add, bool rom, const char *name) "pa 0x%"PRIx64" size 0x%"PRIx64" hva 0x%"PRIxPTR" add %d rom %d name %s" +nvmm_memory_access(int cpuid, uint64_t gpa, size_t size, bool write) "cpuid %d, gpa 0x%"PRIx64", size %zu, write %d" + +#nvmm_ioctl(unsigned long type, void *arg) "type 0x%lx, arg %p" +#nvmm_vm_ioctl(unsigned long type, void *arg) "type 0x%lx, arg %p" +#nvmm_vcpu_ioctl(int cpu_index, unsigned long type, void *arg) "cpu_index %d, type 0x%lx, arg %p" +#nvmm_device_ioctl(int fd, unsigned long type, void *arg) "dev fd %d, type 0x%lx, arg %p" +#nvmm_failed_reg_get(uint64_t id, const char *msg) "Warning: Unable to retrieve ONEREG %" PRIu64 " from nvmm: %s" +#nvmm_failed_reg_set(uint64_t id, const char *msg) "Warning: Unable to set ONEREG %" PRIu64 " to nvmm: %s" +#nvmm_init_vcpu(int cpu_index, unsigned long arch_cpu_id) "index: %d id: %lu" +#nvmm_create_vcpu(int cpu_index, unsigned long arch_cpu_id, int nvmm_fd) "index: %d, id: %lu, nvmm fd: %d" +#nvmm_destroy_vcpu(int cpu_index, unsigned long arch_cpu_id) "index: %d id: %lu" +#nvmm_park_vcpu(int cpu_index, unsigned long arch_cpu_id) "index: %d id: %lu" +#nvmm_unpark_vcpu(unsigned long arch_cpu_id, const char *msg) "id: %lu %s" +#nvmm_irqchip_commit_routes(void) "" +#nvmm_irqchip_add_msi_route(char *name, int vector, int virq) "dev %s vector %d virq %d" +#nvmm_irqchip_update_msi_route(int virq) "Updating MSI route virq=%d" +#nvmm_irqchip_release_virq(int virq) "virq %d" +#nvmm_set_ioeventfd_mmio(int fd, uint64_t addr, uint32_t val, bool assign, uint32_t size, bool datamatch) "fd: %d @0x%" PRIx64 " val=0x%x assign: %d size: %d match: %d" +#nvmm_set_ioeventfd_pio(int fd, uint16_t addr, uint32_t val, bool assign, uint32_t size, bool datamatch) "fd: %d @0x%x val=0x%x assign: %d size: %d match: %d" +#nvmm_set_user_memory(uint16_t as, uint16_t slot, uint32_t flags, uint64_t guest_phys_addr, uint64_t memory_size, uint64_t userspace_addr, uint32_t fd, uint64_t fd_offset, int ret) "AddrSpace#%d Slot#%d flags=0x%x gpa=0x%"PRIx64 " size=0x%"PRIx64 " ua=0x%"PRIx64 " guest_memfd=%d" " guest_memfd_offset=0x%" PRIx64 " ret=%d" +#nvmm_clear_dirty_log(uint32_t slot, uint64_t start, uint32_t size) "slot#%"PRId32" start 0x%"PRIx64" size 0x%"PRIx32 +#nvmm_resample_fd_notify(int gsi) "gsi %d" +#nvmm_dirty_ring_full(int id) "vcpu %d" +#nvmm_dirty_ring_reap_vcpu(int id) "vcpu %d" +#nvmm_dirty_ring_page(int vcpu, uint32_t slot, uint64_t offset) "vcpu %d fetch %"PRIu32" offset 0x%"PRIx64 +#nvmm_dirty_ring_reaper(const char *s) "%s" +#nvmm_dirty_ring_reap(uint64_t count, int64_t t) "reaped %"PRIu64" pages (took %"PRIi64" us)" +#nvmm_dirty_ring_reaper_kick(const char *reason) "%s" +#nvmm_dirty_ring_flush(int finished) "%d" +#nvmm_failed_get_vcpu_mmap_size(void) "" +#nvmm_interrupt_exit_request(void) "" +#nvmm_io_window_exit(void) "" +#nvmm_run_exit_system_event(int cpu_index, uint32_t event_type) "cpu_index %d, system_even_type %"PRIu32 +#nvmm_convert_memory(uint64_t start, uint64_t size, const char *msg) "start 0x%" PRIx64 " size 0x%" PRIx64 " %s" +#nvmm_memory_fault(uint64_t start, uint64_t size, uint64_t flags) "start 0x%" PRIx64 " size 0x%" PRIx64 " flags 0x%" PRIx64 +#nvmm_slots_grow(unsigned int old, unsigned int new) "%u -> %u" diff --git a/accel/nvmm/trace.h b/accel/nvmm/trace.h new file mode 100644 index 0000000000..ecc53ce7c1 --- /dev/null +++ b/accel/nvmm/trace.h @@ -0,0 +1 @@ +#include "trace/trace-accel_nvmm.h" diff --git a/include/system/nvmm_int.h b/include/system/nvmm_int.h new file mode 100644 index 0000000000..b2f9984966 --- /dev/null +++ b/include/system/nvmm_int.h @@ -0,0 +1,210 @@ +/* + * Internal definitions for a target's NVMM support + * + * This work is licensed under the terms of the GNU GPL, version 2 or later. + * See the COPYING file in the top-level directory. + * + */ + +#ifndef QEMU_NVMM_INT_H +#define QEMU_NVMM_INT_H + +#include "system/memory.h" +#include "qapi/qapi-types-common.h" +#include "qemu/accel.h" +#include "qemu/queue.h" +#include "system/nvmm.h" +#include "accel/accel-ops.h" +#include "hw/boards.h" +//#include "hw/i386/topology.h" +#include "io/channel-socket.h" + + +#if 0 +typedef struct NVMMSlot +{ + hwaddr start_addr; + ram_addr_t memory_size; + void *ram; + int slot; + int flags; + int old_flags; + /* Dirty bitmap cache for the slot */ + unsigned long *dirty_bmap; + unsigned long dirty_bmap_size; + /* Cache of the address space ID */ + int as_id; + /* Cache of the offset in ram address space */ + ram_addr_t ram_start_offset; + int guest_memfd; + hwaddr guest_memfd_offset; +} NVMMSlot; + +typedef struct NVMMMemoryUpdate { + QSIMPLEQ_ENTRY(NVMMMemoryUpdate) next; + MemoryRegionSection section; +} NVMMMemoryUpdate; + +typedef struct NVMMMemoryListener { + MemoryListener listener; + NVMMSlot *slots; + unsigned int nr_slots_used; + unsigned int nr_slots_allocated; + int as_id; + QSIMPLEQ_HEAD(, NVMMMemoryUpdate) transaction_add; + QSIMPLEQ_HEAD(, NVMMMemoryUpdate) transaction_del; +} NVMMMemoryListener; + +#define NVMM_MSI_HASHTAB_SIZE 256 + +typedef struct NVMMHostTopoInfo { + /* Number of package on the Host */ + unsigned int maxpkgs; + /* Number of cpus on the Host */ + unsigned int maxcpus; + /* Number of cpus on each different package */ + unsigned int *pkg_cpu_count; + /* Each package can have different maxticks */ + unsigned int *maxticks; +} NVMMHostTopoInfo; + +struct NVMMMsrEnergy { + pid_t pid; + bool enable; + char *socket_path; + QIOChannelSocket *sioc; + QemuThread msr_thr; + unsigned int guest_vcpus; + unsigned int guest_vsockets; + X86CPUTopoInfo guest_topo_info; + NVMMHostTopoInfo host_topo; + const CPUArchIdList *guest_cpu_list; + uint64_t *msr_value; + uint64_t msr_unit; + uint64_t msr_limit; + uint64_t msr_info; +}; + +enum NVMMDirtyRingReaperState { + NVMM_DIRTY_RING_REAPER_NONE = 0, + /* The reaper is sleeping */ + NVMM_DIRTY_RING_REAPER_WAIT, + /* The reaper is reaping for dirty pages */ + NVMM_DIRTY_RING_REAPER_REAPING, +}; + +/* + * NVMM reaper instance, responsible for collecting the NVMM dirty bits + * via the dirty ring. + */ +struct NVMMDirtyRingReaper { + /* The reaper thread */ + QemuThread reaper_thr; + volatile uint64_t reaper_iteration; /* iteration number of reaper thr */ + volatile enum NVMMDirtyRingReaperState reaper_state; /* reap thr state */ +}; + +#endif + + + +struct NVMMState +{ + AccelState parent_obj; + + + +#if 0 + /* Max number of NVMM slots supported */ + int nr_slots_max; + int fd; + int vmfd; + int coalesced_mmio; + int coalesced_pio; + struct kvm_coalesced_mmio_ring *coalesced_mmio_ring; + bool coalesced_flush_in_progress; + int vcpu_events; +#ifdef TARGET_NVMM_HAVE_GUEST_DEBUG + QTAILQ_HEAD(, kvm_sw_breakpoint) kvm_sw_breakpoints; +#endif + int max_nested_state_len; + int kvm_shadow_mem; + bool kernel_irqchip_allowed; + bool kernel_irqchip_required; + OnOffAuto kernel_irqchip_split; + bool sync_mmu; + bool guest_state_protected; + uint64_t manual_dirty_log_protect; + /* + * Older POSIX says that ioctl numbers are signed int, but in + * practice they are not. (Newer POSIX doesn't specify ioctl + * at all.) Linux, glibc and *BSD all treat ioctl numbers as + * unsigned, and real-world ioctl values like NVMM_GET_XSAVE have + * bit 31 set, which means that passing them via an 'int' will + * result in sign-extension when they get converted back to the + * 'unsigned long' which the ioctl() prototype uses. Luckily Linux + * always treats the argument as an unsigned 32-bit int, so any + * possible sign-extension is deliberately ignored, but for + * consistency we keep to the same type that glibc is using. + */ + unsigned long irq_set_ioctl; + unsigned int sigmask_len; + GHashTable *gsimap; +#ifdef NVMM_CAP_IRQ_ROUTING + struct kvm_irq_routing *irq_routes; + int nr_allocated_irq_routes; + unsigned long *used_gsi_bitmap; + unsigned int gsi_count; +#endif + NVMMMemoryListener memory_listener; + QLIST_HEAD(, NVMMParkedVcpu) kvm_parked_vcpus; + + /* For "info mtree -f" to tell if an MR is registered in NVMM */ + int nr_as; + struct NVMMAs { + NVMMMemoryListener *ml; + AddressSpace *as; + } *as; + uint64_t kvm_dirty_ring_bytes; /* Size of the per-vcpu dirty ring */ + uint32_t kvm_dirty_ring_size; /* Number of dirty GFNs per ring */ + bool kvm_dirty_ring_with_bitmap; + uint64_t kvm_eager_split_size; /* Eager Page Splitting chunk size */ + struct NVMMDirtyRingReaper reaper; + struct NVMMMsrEnergy msr_energy; + NotifyVmexitOption notify_vmexit; + uint32_t notify_window; + uint32_t xen_version; + uint32_t xen_caps; + uint16_t xen_gnttab_max_frames; + uint16_t xen_evtchn_max_pirq; + char *device; +#endif + + + +}; + + +#if 0 +void kvm_memory_listener_register(NVMMState *s, NVMMMemoryListener *kml, + AddressSpace *as, int as_id, const char *name); + +void kvm_set_max_memslot_size(hwaddr max_slot_size); + +/** + * kvm_hwpoison_page_add: + * + * Parameters: + * @ram_addr: the address in the RAM for the poisoned page + * + * Add a poisoned page to the list + * + * Return: None. + */ +void kvm_hwpoison_page_add(ram_addr_t ram_addr); +#endif + + + + +#endif diff --git a/meson.build b/meson.build index fada7cee03..f4f38efa48 100644 --- a/meson.build +++ b/meson.build @@ -3626,6 +3626,7 @@ if have_system 'accel/hvf', 'accel/kvm', 'accel/mshv', + 'accel/nvmm', 'audio', 'backends', 'backends/tpm', @@ -3698,6 +3699,7 @@ if have_system or have_user 'hw/core', 'target/arm', 'target/arm/hvf', + 'target/arm/nvmm', 'target/hppa', 'target/i386', 'target/i386/kvm', diff --git a/target/arm/nvmm/nvmm.c b/target/arm/nvmm/nvmm.c index 9228fe114f..6f1a451543 100644 --- a/target/arm/nvmm/nvmm.c +++ b/target/arm/nvmm/nvmm.c @@ -12,6 +12,7 @@ #include "system/address-spaces.h" #include "system/ioport.h" #include "qemu/accel.h" +#include "system/hw_accel.h" #include "system/nvmm.h" #include "system/cpus.h" #include "system/runstate.h" @@ -24,12 +25,17 @@ #include "nvmm_arm.h" #include "cpregs.h" #include "strings.h" +#include "trace.h" +#include "arm-powerctl.h" +#include "target/arm/trace.h" #include #include + struct AccelCPUState { struct nvmm_vcpu vcpu; + sigset_t unblock_ipi_mask; bool stop; }; @@ -456,16 +462,18 @@ nvmm_get_registers(CPUState *cpu) } - - static uint64_t nvmm_vtimer_val_raw(void) { /* * mach_absolute_time() returns the vtimer value without the VM * offset that we define. Add our own offset on top. */ - return 0; +#if 1 + asm volatile ("isb" ::: "memory"); + return reg_cntvct_el0_read(); +#else // return mach_absolute_time() - nvmm_state->vtimer_offset; +#endif } static uint64_t nvmm_vtimer_val(void) @@ -479,8 +487,6 @@ static uint64_t nvmm_vtimer_val(void) return nvmm_vtimer_val_raw(); } - - /* * Called before the VCPU is run. We inject events generated by the I/O * thread. @@ -499,29 +505,19 @@ nvmm_vcpu_pre_run(CPUState *cpu) bool has_event = false; int ret; - // XXXNH needed? I don't think so as I've deleted the pic call - bql_lock(); - -#if 0 - /* - * Force the VCPU out of its inner loop to process any INIT requests. XXXNH - */ - if (cpu->interrupt_request & CPU_INTERRUPT_INIT) { - cpu->exit_request = 1; - } - -#endif - if (!has_event && (cpu->interrupt_request & CPU_INTERRUPT_FIQ)) { - cpu->interrupt_request &= ~CPU_INTERRUPT_FIQ; + if (!has_event && cpu_test_interrupt(cpu, CPU_INTERRUPT_FIQ)) { + trace_nvmm_inject_fiq(); event->type = NVMM_VCPU_EVENT_FIQ; has_event = true; + cpu_reset_interrupt(cpu, CPU_INTERRUPT_FIQ); } - if (!has_event && (cpu->interrupt_request & CPU_INTERRUPT_HARD)) { - cpu->interrupt_request &= ~CPU_INTERRUPT_HARD; + if (!has_event && cpu_test_interrupt(cpu, CPU_INTERRUPT_HARD)) { + trace_nvmm_inject_irq(); event->type = NVMM_VCPU_EVENT_IRQ; has_event = true; + cpu_reset_interrupt(cpu, CPU_INTERRUPT_HARD); } if (has_event) { @@ -531,8 +527,6 @@ nvmm_vcpu_pre_run(CPUState *cpu) " error=%d", errno); } } - - bql_unlock(); } /* @@ -563,7 +557,6 @@ nvmm_vcpu_post_run(CPUState *cpu, struct nvmm_vcpu_exit *exit) } /* -------------------------------------------------------------------------- */ - static int nvmm_handle_halted(struct nvmm_machine *mach, CPUState *cpu, struct nvmm_vcpu_exit *exit) @@ -571,8 +564,7 @@ nvmm_handle_halted(struct nvmm_machine *mach, CPUState *cpu, int ret = 0; bql_lock(); - if (!(cpu->interrupt_request & CPU_INTERRUPT_HARD) && - !(cpu->interrupt_request & CPU_INTERRUPT_FIQ)) { + if (!cpu_test_interrupt(cpu, CPU_INTERRUPT_HARD | CPU_INTERRUPT_FIQ)) { cpu->exception_index = EXCP_HLT; cpu->halted = true; ret = 1; @@ -583,42 +575,58 @@ nvmm_handle_halted(struct nvmm_machine *mach, CPUState *cpu, } +static void +nvmm_wait_for_ipi(CPUState *cpu, struct timespec *ts) +{ + /* + * Use pselect to sleep so that other threads can IPI us while we're + * sleeping. + */ + trace_nvmm_wait_for_ipi(true); + qatomic_set_mb(&cpu->thread_kicked, false); + bql_unlock(); + pselect(0, 0, 0, 0, ts, &cpu->accel->unblock_ipi_mask); + bql_lock(); + trace_nvmm_wait_for_ipi(false); +} + +#define CNTCTL_ISTATUS __BIT(2) +#define CNTCTL_IMASK __BIT(1) +#define CNTCTL_ENABLE __BIT(0) + + static void nvmm_wfi(CPUState *cpu) { - ARMCPU *arm_cpu = ARM_CPU(cpu); AccelCPUState *qcpu = cpu->accel; + struct nvmm_machine *mach = get_nvmm_mach(); + ARMCPU *arm_cpu = ARM_CPU(cpu); struct nvmm_vcpu *vcpu = &qcpu->vcpu; struct nvmm_aarch64_state *state = vcpu->state; -// struct timespec ts; -// hv_return_t r; -// uint64_t ctl; -// uint64_t cval; int64_t ticks_to_sleep; + struct timespec ts; uint64_t seconds; uint64_t nanos; uint32_t cntfrq; - -// ARMCPU *arm_cpu = ARM_CPU(cpu); -// CPUArchState *env = cpu_env(cpu); // &cpu->env; -// struct nvmm_machine *mach = get_nvmm_mach(); - - - if (cpu->interrupt_request & (CPU_INTERRUPT_HARD | CPU_INTERRUPT_FIQ)) { + if (cpu_test_interrupt(cpu, CPU_INTERRUPT_HARD | CPU_INTERRUPT_FIQ)) { /* Interrupt pending, no need to wait */ return; } - uint32_t ctl = state->sprs[NVMM_AARCH64_SPR_CNTV_CTL_EL0]; - if (!(ctl & 1) || (ctl & 2)) { + trace_nvmm_wfi(); + /* ret = */ nvmm_vcpu_getstate(mach, vcpu, NVMM_AARCH64_STATE_SPRS); +#if 0 + uint64_t ctl = state->sprs[NVMM_AARCH64_SPR_CNTV_CTL_EL0]; + if (!(ctl & CNTCTL_ENABLE) || (ctl & CNTCTL_IMASK)) { /* Timer disabled or masked, just wait for an IPI. */ -// nvmm_wait_for_ipi(cpu, NULL); + nvmm_wait_for_ipi(cpu, NULL); + state->sprs[NVMM_AARCH64_SPR_PC] += 4; + cpu->vcpu_dirty = true; return; } - - uint64_t cval = state->sprs[NVMM_AARCH64_SPR_CNTV_CTL_EL0]; - +#endif + uint64_t cval = state->sprs[NVMM_AARCH64_SPR_CNTV_CVAL_EL0]; ticks_to_sleep = cval - nvmm_vtimer_val(); if (ticks_to_sleep < 0) { return; @@ -629,7 +637,6 @@ nvmm_wfi(CPUState *cpu) ticks_to_sleep -= muldiv64(seconds, NANOSECONDS_PER_SECOND, cntfrq); nanos = ticks_to_sleep * cntfrq; - // XXXNH update /* * Don't sleep for less than the time a context switch would take, * so that we can satisfy fast timer requests on the same CPU. @@ -639,34 +646,181 @@ nvmm_wfi(CPUState *cpu) return; } -// ts = (struct timespec) { seconds, nanos }; -// nvmm_wait_for_ipi(cpu, &ts); + trace_nvmm_wait_for_ipi_sleep(seconds, nanos); + ts = (struct timespec) { seconds, nanos }; + nvmm_wait_for_ipi(cpu, &ts); + + state->sprs[NVMM_AARCH64_SPR_PC] += 4; + cpu->vcpu_dirty = true; +} + + + + +static void nvmm_psci_cpu_off(ARMCPU *arm_cpu) +{ + int32_t ret = arm_set_cpu_off(arm_cpu_mp_affinity(arm_cpu)); + assert(ret == QEMU_ARM_POWERCTL_RET_SUCCESS); +} + +/* + * Handle a PSCI call. + * + * Returns 0 on success + * -1 when the PSCI call is unknown, + */ +static bool nvmm_handle_psci_call(CPUState *cpu) +{ + ARMCPU *arm_cpu = ARM_CPU(cpu); + CPUARMState *env = &arm_cpu->env; + uint64_t param[4] = { + env->xregs[0], + env->xregs[1], + env->xregs[2], + env->xregs[3] + }; + uint64_t context_id, mpidr; + bool target_aarch64 = true; + CPUState *target_cpu_state; + ARMCPU *target_cpu; + target_ulong entry; + int target_el = 1; + int32_t ret = 0; + + trace_arm_psci_call(param[0], param[1], param[2], param[3], + arm_cpu_mp_affinity(arm_cpu)); + + switch (param[0]) { + case QEMU_PSCI_0_2_FN_PSCI_VERSION: + ret = QEMU_PSCI_VERSION_1_1; + break; + case QEMU_PSCI_0_2_FN_MIGRATE_INFO_TYPE: + ret = QEMU_PSCI_0_2_RET_TOS_MIGRATION_NOT_REQUIRED; /* No trusted OS */ + break; + case QEMU_PSCI_0_2_FN_AFFINITY_INFO: + case QEMU_PSCI_0_2_FN64_AFFINITY_INFO: + mpidr = param[1]; + + switch (param[2]) { + case 0: + target_cpu_state = arm_get_cpu_by_id(mpidr); + if (!target_cpu_state) { + ret = QEMU_PSCI_RET_INVALID_PARAMS; + break; + } + target_cpu = ARM_CPU(target_cpu_state); + + ret = target_cpu->power_state; + break; + default: + /* Everything above affinity level 0 is always on. */ + ret = 0; + } + break; + case QEMU_PSCI_0_2_FN_SYSTEM_RESET: + qemu_system_reset_request(SHUTDOWN_CAUSE_GUEST_RESET); + /* + * QEMU reset and shutdown are async requests, but PSCI + * mandates that we never return from the reset/shutdown + * call, so power the CPU off now so it doesn't execute + * anything further. + */ + nvmm_psci_cpu_off(arm_cpu); + break; + case QEMU_PSCI_0_2_FN_SYSTEM_OFF: + qemu_system_shutdown_request(SHUTDOWN_CAUSE_GUEST_SHUTDOWN); + nvmm_psci_cpu_off(arm_cpu); + break; + case QEMU_PSCI_0_1_FN_CPU_ON: + case QEMU_PSCI_0_2_FN_CPU_ON: + case QEMU_PSCI_0_2_FN64_CPU_ON: + mpidr = param[1]; + entry = param[2]; + context_id = param[3]; + ret = arm_set_cpu_on(mpidr, entry, context_id, + target_el, target_aarch64); + break; + case QEMU_PSCI_0_1_FN_CPU_OFF: + case QEMU_PSCI_0_2_FN_CPU_OFF: + nvmm_psci_cpu_off(arm_cpu); + break; + case QEMU_PSCI_0_1_FN_CPU_SUSPEND: + case QEMU_PSCI_0_2_FN_CPU_SUSPEND: + case QEMU_PSCI_0_2_FN64_CPU_SUSPEND: + /* Affinity levels are not supported in QEMU */ + if (param[1] & 0xfffe0000) { + ret = QEMU_PSCI_RET_INVALID_PARAMS; + break; + } + /* Powerdown is not supported, we always go into WFI */ + env->xregs[0] = 0; + nvmm_wfi(cpu); + break; + case QEMU_PSCI_0_1_FN_MIGRATE: + case QEMU_PSCI_0_2_FN_MIGRATE: + ret = QEMU_PSCI_RET_NOT_SUPPORTED; + break; + case QEMU_PSCI_1_0_FN_PSCI_FEATURES: + switch (param[1]) { + case QEMU_PSCI_0_2_FN_PSCI_VERSION: + case QEMU_PSCI_0_2_FN_MIGRATE_INFO_TYPE: + case QEMU_PSCI_0_2_FN_AFFINITY_INFO: + case QEMU_PSCI_0_2_FN64_AFFINITY_INFO: + case QEMU_PSCI_0_2_FN_SYSTEM_RESET: + case QEMU_PSCI_0_2_FN_SYSTEM_OFF: + case QEMU_PSCI_0_1_FN_CPU_ON: + case QEMU_PSCI_0_2_FN_CPU_ON: + case QEMU_PSCI_0_2_FN64_CPU_ON: + case QEMU_PSCI_0_1_FN_CPU_OFF: + case QEMU_PSCI_0_2_FN_CPU_OFF: + case QEMU_PSCI_0_1_FN_CPU_SUSPEND: + case QEMU_PSCI_0_2_FN_CPU_SUSPEND: + case QEMU_PSCI_0_2_FN64_CPU_SUSPEND: + case QEMU_PSCI_1_0_FN_PSCI_FEATURES: + ret = 0; + break; + case QEMU_PSCI_0_1_FN_MIGRATE: + case QEMU_PSCI_0_2_FN_MIGRATE: + default: + ret = QEMU_PSCI_RET_NOT_SUPPORTED; + } + break; + default: + return false; + } + + env->xregs[0] = ret; + + return true; } int nvmm_vcpu_loop(CPUState *cpu) { + CPUARMState *env = cpu_env(cpu); + ARMCPU *arm_cpu = env_archcpu(env); struct nvmm_machine *mach = get_nvmm_mach(); AccelCPUState *qcpu = cpu->accel; struct nvmm_vcpu *vcpu = &qcpu->vcpu; struct nvmm_vcpu_exit *exit = vcpu->exit; int ret; - +#if 0 //XXXXXXXXXX: nvmm cannot send multiple events at the same time? - if (cpu->interrupt_request & CPU_INTERRUPT_FIQ) { + if (cpu_test_interrupt(cpu, CPU_INTERRUPT_FIQ)) { + trace_nvmm_inject_fiq(); vcpu->event->type = NVMM_VCPU_EVENT_FIQ; nvmm_vcpu_inject(mach, vcpu); + cpu_reset_interrupt(cpu, CPU_INTERRUPT_FIQ); cpu->halted = false; -fprintf(stderr, "%s:%d fiq\n", __func__, __LINE__); - } - if (cpu->interrupt_request & CPU_INTERRUPT_HARD) { + } else if (cpu_test_interrupt(cpu, CPU_INTERRUPT_HARD)) { + trace_nvmm_inject_irq(); vcpu->event->type = NVMM_VCPU_EVENT_IRQ; nvmm_vcpu_inject(mach, vcpu); + cpu_reset_interrupt(cpu, CPU_INTERRUPT_HARD); cpu->halted = false; -fprintf(stderr, "%s:%d irq\n", __func__, __LINE__); } - +#endif if (cpu->halted) { cpu->exception_index = EXCP_HLT; qatomic_set(&cpu->exit_request, false); @@ -696,7 +850,8 @@ fprintf(stderr, "%s:%d irq\n", __func__, __LINE__); nvmm_vcpu_pre_run(cpu); - if (qatomic_read(&cpu->exit_request)) { + /* Read exit_request before the kernel reads the immediate exit flag */ + if (qatomic_load_acquire(&cpu->exit_request)) { #if NVMM_USER_VERSION >= 2 nvmm_vcpu_stop(vcpu); #else @@ -704,8 +859,6 @@ fprintf(stderr, "%s:%d irq\n", __func__, __LINE__); #endif } - /* Read exit_request before the kernel reads the immediate exit flag */ - smp_rmb(); ret = nvmm_vcpu_run(mach, vcpu); if (ret == -1) { error_report("NVMM: Failed to exec a virtual processor," @@ -722,14 +875,11 @@ fprintf(stderr, "%s:%d irq\n", __func__, __LINE__); case NVMM_VCPU_EXIT_IRQ: // Set vtimer / other in virtual gic //switched_level = cpu->device_irq_level ^ run->s.regs.device_irq_level; - + trace_nvmm_exit_irq(); bql_lock(); if (exit->exitstate.vtimer) { - ARMCPU *arm_cpu = ARM_CPU(cpu); - + // Gets sets in interrupt_request qemu_set_irq(arm_cpu->gt_timer_outputs[GTIMER_VIRT], 1); - // inject - cpu->interrupt_request |= CPU_INTERRUPT_HARD; } bql_unlock(); @@ -746,26 +896,42 @@ fprintf(stderr, "%s:%d irq\n", __func__, __LINE__); ret = nvmm_handle_mem(mach, vcpu); break; case NVMM_VCPU_EXIT_MRS: - error_report("XXXXXX: NOTYET: NVMM_VCPU_EXIT_MRS\n"); + error_report("XXXXXX: NOTYET: NVMM_VCPU_EXIT_MRS"); abort(); break; case NVMM_VCPU_EXIT_MSR: - error_report("XXXXXX: NOTYET: NVMM_VCPU_EXIT_MSR\n"); + error_report("XXXXXX: NOTYET: NVMM_VCPU_EXIT_MSR"); abort(); break; case NVMM_VCPU_EXIT_HVC: - error_report("XXXXXX: NOTYET: NVMM_VCPU_EXIT_HVC\n"); - abort(); + cpu_synchronize_state(cpu); + if (arm_cpu->psci_conduit == QEMU_PSCI_CONDUIT_HVC) { + /* Do NOT advance $pc for HVC */ + if (!nvmm_handle_psci_call(cpu)) { + trace_nvmm_unknown_hvc(env->pc, env->xregs[0]); + /* SMCCC 1.3 section 5.2 says every unknown SMCCC call returns -1 */ + env->xregs[0] = -1; + } else { + error_report("XXX handled HVC"); + } + cpu->vcpu_dirty = true; + } else { + error_report("XXX unknown conduit %d", arm_cpu->psci_conduit); +// trace_nvmm_unknown_hvc(env->pc, env->xregs[0]); +// nvmm_raise_exception(cpu, EXCP_UDEF, syn_uncategorized(), 1); + } break; case NVMM_VCPU_EXIT_SMC: - error_report("XXXXXX: NOTYET: NVMM_VCPU_EXIT_SMC\n"); + error_report("XXXXXX: NOTYET: NVMM_VCPU_EXIT_SMC"); abort(); break; case NVMM_VCPU_EXIT_WFI: + bql_lock(); nvmm_wfi(cpu); + bql_unlock(); break; case NVMM_VCPU_EXIT_WFE: - error_report("XXXXXX: NOTYET: NVMM_VCPU_EXIT_WFE\n"); + error_report("XXXXXX: NOTYET: NVMM_VCPU_EXIT_WFE"); break; case NVMM_VCPU_EXIT_HALTED: ret = nvmm_handle_halted(mach, cpu, exit); @@ -796,8 +962,6 @@ fprintf(stderr, "%s:%d irq\n", __func__, __LINE__); cpu_exec_end(cpu); bql_lock(); - qatomic_set(&cpu->exit_request, false); - return ret < 0; } @@ -871,7 +1035,6 @@ nvmm_sreg_init(CPUState *cpu) ri = get_arm_cp_reginfo(arm_cpu->cp_regs, key); if (ri) { -// printf("group=%d, reg=%d\n", nvmm_sreg_match[i].group, nvmm_sreg_match[i].reg); assert(!(ri->type & ARM_CP_NO_RAW)); nvmm_sreg_match[i].cp_idx = sregs_cnt; arm_cpu->cpreg_indexes[sregs_cnt++] = cpreg_to_kvm_id(key); @@ -1019,6 +1182,10 @@ nvmm_init_vcpu(CPUState *cpu) cpu->vcpu_dirty = true; cpu->accel = qcpu; + // XXXNH maybe not, cf. nvmm_init_cpu_signals above + pthread_sigmask(SIG_BLOCK, NULL, &cpu->accel->unblock_ipi_mask); + sigdelset(&cpu->accel->unblock_ipi_mask, SIG_IPI); + nvmm_sreg_init(cpu); /* Set CP_NO_RAW system registers on init */ @@ -1029,7 +1196,7 @@ nvmm_init_vcpu(CPUState *cpu) state->sprs[NVMM_AARCH64_SPR_MIDR_EL1] = arm_cpu->midr; state->sprs[NVMM_AARCH64_SPR_MPIDR_EL1] = (1U << 31) | arm_cpu->mp_affinity; - CPUArchState *env = &arm_cpu->env; +// CPUArchState *env = &arm_cpu->env; #if 0 uint64_t pfr = state->tids[NVMM_AARCH64_TID_ID_AA64PFR0_EL1]; diff --git a/target/arm/nvmm/trace-events b/target/arm/nvmm/trace-events new file mode 100644 index 0000000000..729d60bc4d --- /dev/null +++ b/target/arm/nvmm/trace-events @@ -0,0 +1,22 @@ +#nvmm_unhandled_sysreg_read(uint64_t pc, uint32_t reg, uint32_t op0, uint32_t op1, uint32_t crn, uint32_t crm, uint32_t op2) "unhandled sysreg read at pc=0x%"PRIx64": 0x%08x (op0=%d op1=%d crn=%d crm=%d op2=%d)" +#nvmm_unhandled_sysreg_write(uint64_t pc, uint32_t reg, uint32_t op0, uint32_t op1, uint32_t crn, uint32_t crm, uint32_t op2) "unhandled sysreg write at pc=0x%"PRIx64": 0x%08x (op0=%d op1=%d crn=%d crm=%d op2=%d)" +nvmm_inject_fiq(void) "injecting FIQ" +nvmm_inject_irq(void) "injecting IRQ" +nvmm_exit_irq(void) "irq exit" +nvmm_wfi(void) "wfi" +nvmm_wait_for_ipi(int sleep) "IPI sleep %d" +nvmm_wait_for_ipi_sleep(uint64_t seconds, uint64_t nanos) "sleep %"PRIu64" seconds %"PRIu64 + + +#nvmm_data_abort(uint64_t va, uint64_t pa, bool isv, bool iswrite, bool s1ptw, uint32_t len, uint32_t srt) "data abort: [va=0x%016"PRIx64" pa=0x%016"PRIx64" isv=%d iswrite=%d s1ptw=%d len=%d srt=%d]" +#nvmm_insn_abort(uint64_t pc, uint32_t set, bool fnv, bool ea, bool s1ptw, uint32_t ifsc) "insn abort: [pc=0x%"PRIx64" set=%d fnv=%d ea=%d s1ptw=%d ifsc=%d]" +#nvmm_sysreg_read(uint32_t reg, uint32_t op0, uint32_t op1, uint32_t crn, uint32_t crm, uint32_t op2, uint64_t val) "sysreg read 0x%08x (op0=%d op1=%d crn=%d crm=%d op2=%d) = 0x%016"PRIx64 +#nvmm_sysreg_write(uint32_t reg, uint32_t op0, uint32_t op1, uint32_t crn, uint32_t crm, uint32_t op2, uint64_t val) "sysreg write 0x%08x (op0=%d op1=%d crn=%d crm=%d op2=%d, val=0x%016"PRIx64")" +nvmm_unknown_hvc(uint64_t pc, uint64_t x0) "pc=0x%"PRIx64" unknown HVC! 0x%016"PRIx64 +#nvmm_unknown_smc(uint64_t x0) "unknown SMC! 0x%016"PRIx64 +#nvmm_exit(uint64_t syndrome, uint32_t ec, uint64_t pc) "exit: 0x%"PRIx64" [ec=0x%x pc=0x%"PRIx64"]" +nvmm_psci_call(uint64_t x0, uint64_t x1, uint64_t x2, uint64_t x3, uint32_t cpuid) "PSCI Call x0=0x%016"PRIx64" x1=0x%016"PRIx64" x2=0x%016"PRIx64" x3=0x%016"PRIx64" cpuid=0x%x" +#nvmm_emu_reginfo_write(const char *cpname, const char *regname, uint64_t val) "[%s] write to %s [val=0x%016"PRIx64"]" +#nvmm_emu_reginfo_read(const char *cpname, const char *regname, uint64_t val) "[%s] read from %s [val=0x%016"PRIx64"]" +#nvmm_illegal_guest_state(void) "HV_ILLEGAL_GUEST_STATE" +#nvmm_kick_vcpu_thread(unsigned cpuidx, bool stop) "cpu:%u stop:%u" diff --git a/target/arm/nvmm/trace.h b/target/arm/nvmm/trace.h new file mode 100644 index 0000000000..bed5fbbac7 --- /dev/null +++ b/target/arm/nvmm/trace.h @@ -0,0 +1 @@ +#include "trace/trace-target_arm_nvmm.h" -- 2.54.0 (Apple Git-157)