2003-05-25 20:46:15 +04:00
|
|
|
/*
|
|
|
|
* internal execution defines for qemu
|
2007-09-17 01:08:06 +04:00
|
|
|
*
|
2003-05-25 20:46:15 +04:00
|
|
|
* Copyright (c) 2003 Fabrice Bellard
|
|
|
|
*
|
|
|
|
* This library is free software; you can redistribute it and/or
|
|
|
|
* modify it under the terms of the GNU Lesser General Public
|
|
|
|
* License as published by the Free Software Foundation; either
|
2020-10-23 15:33:53 +03:00
|
|
|
* version 2.1 of the License, or (at your option) any later version.
|
2003-05-25 20:46:15 +04:00
|
|
|
*
|
|
|
|
* This library is distributed in the hope that it will be useful,
|
|
|
|
* but WITHOUT ANY WARRANTY; without even the implied warranty of
|
|
|
|
* MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the GNU
|
|
|
|
* Lesser General Public License for more details.
|
|
|
|
*
|
|
|
|
* You should have received a copy of the GNU Lesser General Public
|
2009-07-17 00:47:01 +04:00
|
|
|
* License along with this library; if not, see <http://www.gnu.org/licenses/>.
|
2003-05-25 20:46:15 +04:00
|
|
|
*/
|
|
|
|
|
2016-06-29 14:47:03 +03:00
|
|
|
#ifndef EXEC_ALL_H
|
|
|
|
#define EXEC_ALL_H
|
2009-01-14 22:00:36 +03:00
|
|
|
|
2019-08-12 08:23:31 +03:00
|
|
|
#include "cpu.h"
|
2019-05-17 14:36:10 +03:00
|
|
|
#ifdef CONFIG_TCG
|
2018-11-04 00:40:22 +03:00
|
|
|
#include "exec/cpu_ldst.h"
|
2019-05-17 14:36:10 +03:00
|
|
|
#endif
|
2022-10-01 23:36:33 +03:00
|
|
|
#include "qemu/interval-tree.h"
|
2023-01-17 16:52:02 +03:00
|
|
|
#include "qemu/clang-tsa.h"
|
2009-01-14 22:00:36 +03:00
|
|
|
|
2010-03-12 19:54:58 +03:00
|
|
|
/* Page tracking code uses ram addresses in system mode, and virtual
|
|
|
|
addresses in userspace mode. Define tb_page_addr_t to be an appropriate
|
|
|
|
type. */
|
|
|
|
#if defined(CONFIG_USER_ONLY)
|
2010-03-13 02:23:29 +03:00
|
|
|
typedef abi_ulong tb_page_addr_t;
|
2017-07-14 00:18:15 +03:00
|
|
|
#define TB_PAGE_ADDR_FMT TARGET_ABI_FMT_lx
|
2010-03-12 19:54:58 +03:00
|
|
|
#else
|
|
|
|
typedef ram_addr_t tb_page_addr_t;
|
2017-07-14 00:18:15 +03:00
|
|
|
#define TB_PAGE_ADDR_FMT RAM_ADDR_FMT
|
2010-03-12 19:54:58 +03:00
|
|
|
#endif
|
|
|
|
|
2022-10-24 15:15:04 +03:00
|
|
|
/**
|
|
|
|
* cpu_unwind_state_data:
|
|
|
|
* @cpu: the cpu context
|
|
|
|
* @host_pc: the host pc within the translation
|
|
|
|
* @data: output data
|
|
|
|
*
|
|
|
|
* Attempt to load the the unwind state for a host pc occurring in
|
|
|
|
* translated code. If @host_pc is not in translated code, the
|
|
|
|
* function returns false; otherwise @data is loaded.
|
|
|
|
* This is the same unwind info as given to restore_state_to_opc.
|
|
|
|
*/
|
|
|
|
bool cpu_unwind_state_data(CPUState *cpu, uintptr_t host_pc, uint64_t *data);
|
|
|
|
|
2017-11-13 16:55:27 +03:00
|
|
|
/**
|
|
|
|
* cpu_restore_state:
|
2022-10-24 15:15:04 +03:00
|
|
|
* @cpu: the cpu context
|
|
|
|
* @host_pc: the host pc within the translation
|
2017-11-13 16:55:27 +03:00
|
|
|
* @return: true if state was restored, false otherwise
|
|
|
|
*
|
|
|
|
* Attempt to restore the state for a fault occurring in translated
|
2022-10-24 15:15:04 +03:00
|
|
|
* code. If @host_pc is not in translated code no state is
|
2017-11-13 16:55:27 +03:00
|
|
|
* restored and the function returns false.
|
|
|
|
*/
|
2022-10-24 16:09:57 +03:00
|
|
|
bool cpu_restore_state(CPUState *cpu, uintptr_t host_pc);
|
2012-12-05 00:16:07 +04:00
|
|
|
|
2022-04-20 16:26:02 +03:00
|
|
|
G_NORETURN void cpu_loop_exit_noexc(CPUState *cpu);
|
|
|
|
G_NORETURN void cpu_loop_exit(CPUState *cpu);
|
|
|
|
G_NORETURN void cpu_loop_exit_restore(CPUState *cpu, uintptr_t pc);
|
|
|
|
G_NORETURN void cpu_loop_exit_atomic(CPUState *cpu, uintptr_t pc);
|
2015-04-22 15:15:48 +03:00
|
|
|
|
s390x/tcg: MVCL: Exit to main loop if requested
MVCL is interruptible and we should check for interrupts and process
them after writing back the variables to the registers. Let's check
for any exit requests and exit to the main loop. Introduce a new helper
function for that: cpu_loop_exit_requested().
When booting Fedora 30, I can see a handful of these exits and it seems
to work reliable. Also, Richard explained why this works correctly even
when MVCL is called via EXECUTE:
(1) TB with EXECUTE runs, at address Ae
- env->psw_addr stored with Ae.
- helper_ex() runs, memory address Am computed
from D2a(X2a,B2a) or from psw.addr+RI2.
- env->ex_value stored with memory value modified by R1a
(2) TB of executee runs,
- env->ex_value stored with 0.
- helper_mvcl() runs, using and updating R1b, R1b+1, R2b, R2b+1.
(3a) helper_mvcl() completes,
- TB of executee continues, psw.addr += ilen.
- Next instruction is the one following EXECUTE.
(3b) helper_mvcl() exits to main loop,
- cpu_loop_exit_restore() unwinds psw.addr = Ae.
- Next instruction is the EXECUTE itself...
- goto 1.
As the PoP mentiones that an interruptible instruction called via EXECUTE
should avoid modifying storage/registers that are used by EXECUTE itself,
it is fine to retrigger EXECUTE.
Cc: Alex Bennée <alex.bennee@linaro.org>
Cc: Peter Maydell <peter.maydell@linaro.org>
Cc: Paolo Bonzini <pbonzini@redhat.com>
Suggested-by: Richard Henderson <richard.henderson@linaro.org>
Reviewed-by: Richard Henderson <richard.henderson@linaro.org>
Reviewed-by: Alex Bennée <alex.bennee@linaro.org>
Signed-off-by: David Hildenbrand <david@redhat.com>
2019-10-01 21:03:54 +03:00
|
|
|
/**
|
|
|
|
* cpu_loop_exit_requested:
|
|
|
|
* @cpu: The CPU state to be tested
|
|
|
|
*
|
|
|
|
* Indicate if somebody asked for a return of the CPU to the main loop
|
|
|
|
* (e.g., via cpu_exit() or cpu_interrupt()).
|
|
|
|
*
|
|
|
|
* This is helpful for architectures that support interruptible
|
|
|
|
* instructions. After writing back all state to registers/memory, this
|
|
|
|
* call can be used to check if it makes sense to return to the main loop
|
|
|
|
* or to continue executing the interruptible instruction.
|
|
|
|
*/
|
|
|
|
static inline bool cpu_loop_exit_requested(CPUState *cpu)
|
|
|
|
{
|
2020-09-23 13:56:46 +03:00
|
|
|
return (int32_t)qatomic_read(&cpu_neg(cpu)->icount_decr.u32) < 0;
|
s390x/tcg: MVCL: Exit to main loop if requested
MVCL is interruptible and we should check for interrupts and process
them after writing back the variables to the registers. Let's check
for any exit requests and exit to the main loop. Introduce a new helper
function for that: cpu_loop_exit_requested().
When booting Fedora 30, I can see a handful of these exits and it seems
to work reliable. Also, Richard explained why this works correctly even
when MVCL is called via EXECUTE:
(1) TB with EXECUTE runs, at address Ae
- env->psw_addr stored with Ae.
- helper_ex() runs, memory address Am computed
from D2a(X2a,B2a) or from psw.addr+RI2.
- env->ex_value stored with memory value modified by R1a
(2) TB of executee runs,
- env->ex_value stored with 0.
- helper_mvcl() runs, using and updating R1b, R1b+1, R2b, R2b+1.
(3a) helper_mvcl() completes,
- TB of executee continues, psw.addr += ilen.
- Next instruction is the one following EXECUTE.
(3b) helper_mvcl() exits to main loop,
- cpu_loop_exit_restore() unwinds psw.addr = Ae.
- Next instruction is the EXECUTE itself...
- goto 1.
As the PoP mentiones that an interruptible instruction called via EXECUTE
should avoid modifying storage/registers that are used by EXECUTE itself,
it is fine to retrigger EXECUTE.
Cc: Alex Bennée <alex.bennee@linaro.org>
Cc: Peter Maydell <peter.maydell@linaro.org>
Cc: Paolo Bonzini <pbonzini@redhat.com>
Suggested-by: Richard Henderson <richard.henderson@linaro.org>
Reviewed-by: Richard Henderson <richard.henderson@linaro.org>
Reviewed-by: Alex Bennée <alex.bennee@linaro.org>
Signed-off-by: David Hildenbrand <david@redhat.com>
2019-10-01 21:03:54 +03:00
|
|
|
}
|
|
|
|
|
2017-07-03 13:12:21 +03:00
|
|
|
#if !defined(CONFIG_USER_ONLY) && defined(CONFIG_TCG)
|
2012-04-09 20:50:52 +04:00
|
|
|
/* cputlb.c */
|
2018-10-09 20:45:54 +03:00
|
|
|
/**
|
|
|
|
* tlb_init - initialize a CPU's TLB
|
|
|
|
* @cpu: CPU whose TLB should be initialized
|
|
|
|
*/
|
|
|
|
void tlb_init(CPUState *cpu);
|
2020-06-12 22:02:26 +03:00
|
|
|
/**
|
|
|
|
* tlb_destroy - destroy a CPU's TLB
|
|
|
|
* @cpu: CPU whose TLB should be destroyed
|
|
|
|
*/
|
|
|
|
void tlb_destroy(CPUState *cpu);
|
2015-08-25 17:45:09 +03:00
|
|
|
/**
|
|
|
|
* tlb_flush_page:
|
|
|
|
* @cpu: CPU whose TLB should be flushed
|
|
|
|
* @addr: virtual address of page to be flushed
|
|
|
|
*
|
|
|
|
* Flush one page from the TLB of the specified CPU, for all
|
|
|
|
* MMU indexes.
|
|
|
|
*/
|
2013-09-04 03:29:02 +04:00
|
|
|
void tlb_flush_page(CPUState *cpu, target_ulong addr);
|
2017-02-23 21:29:22 +03:00
|
|
|
/**
|
|
|
|
* tlb_flush_page_all_cpus:
|
|
|
|
* @cpu: src CPU of the flush
|
|
|
|
* @addr: virtual address of page to be flushed
|
|
|
|
*
|
|
|
|
* Flush one page from the TLB of the specified CPU, for all
|
|
|
|
* MMU indexes.
|
|
|
|
*/
|
|
|
|
void tlb_flush_page_all_cpus(CPUState *src, target_ulong addr);
|
|
|
|
/**
|
|
|
|
* tlb_flush_page_all_cpus_synced:
|
|
|
|
* @cpu: src CPU of the flush
|
|
|
|
* @addr: virtual address of page to be flushed
|
|
|
|
*
|
|
|
|
* Flush one page from the TLB of the specified CPU, for all MMU
|
|
|
|
* indexes like tlb_flush_page_all_cpus except the source vCPUs work
|
|
|
|
* is scheduled as safe work meaning all flushes will be complete once
|
|
|
|
* the source vCPUs safe work is complete. This will depend on when
|
|
|
|
* the guests translation ends the TB.
|
|
|
|
*/
|
|
|
|
void tlb_flush_page_all_cpus_synced(CPUState *src, target_ulong addr);
|
2015-08-25 17:45:09 +03:00
|
|
|
/**
|
|
|
|
* tlb_flush:
|
|
|
|
* @cpu: CPU whose TLB should be flushed
|
|
|
|
*
|
2016-11-14 17:17:28 +03:00
|
|
|
* Flush the entire TLB for the specified CPU. Most CPU architectures
|
|
|
|
* allow the implementation to drop entries from the TLB at any time
|
|
|
|
* so this is generally safe. If more selective flushing is required
|
|
|
|
* use one of the other functions for efficiency.
|
2015-08-25 17:45:09 +03:00
|
|
|
*/
|
2016-11-14 17:17:28 +03:00
|
|
|
void tlb_flush(CPUState *cpu);
|
2017-02-23 21:29:22 +03:00
|
|
|
/**
|
|
|
|
* tlb_flush_all_cpus:
|
|
|
|
* @cpu: src CPU of the flush
|
|
|
|
*/
|
|
|
|
void tlb_flush_all_cpus(CPUState *src_cpu);
|
|
|
|
/**
|
|
|
|
* tlb_flush_all_cpus_synced:
|
|
|
|
* @cpu: src CPU of the flush
|
|
|
|
*
|
|
|
|
* Like tlb_flush_all_cpus except this except the source vCPUs work is
|
|
|
|
* scheduled as safe work meaning all flushes will be complete once
|
|
|
|
* the source vCPUs safe work is complete. This will depend on when
|
|
|
|
* the guests translation ends the TB.
|
|
|
|
*/
|
|
|
|
void tlb_flush_all_cpus_synced(CPUState *src_cpu);
|
2015-08-25 17:45:09 +03:00
|
|
|
/**
|
|
|
|
* tlb_flush_page_by_mmuidx:
|
|
|
|
* @cpu: CPU whose TLB should be flushed
|
|
|
|
* @addr: virtual address of page to be flushed
|
2017-02-23 21:29:19 +03:00
|
|
|
* @idxmap: bitmap of MMU indexes to flush
|
2015-08-25 17:45:09 +03:00
|
|
|
*
|
|
|
|
* Flush one page from the TLB of the specified CPU, for the specified
|
|
|
|
* MMU indexes.
|
|
|
|
*/
|
2017-02-23 21:29:19 +03:00
|
|
|
void tlb_flush_page_by_mmuidx(CPUState *cpu, target_ulong addr,
|
|
|
|
uint16_t idxmap);
|
2017-02-23 21:29:22 +03:00
|
|
|
/**
|
|
|
|
* tlb_flush_page_by_mmuidx_all_cpus:
|
|
|
|
* @cpu: Originating CPU of the flush
|
|
|
|
* @addr: virtual address of page to be flushed
|
|
|
|
* @idxmap: bitmap of MMU indexes to flush
|
|
|
|
*
|
|
|
|
* Flush one page from the TLB of all CPUs, for the specified
|
|
|
|
* MMU indexes.
|
|
|
|
*/
|
|
|
|
void tlb_flush_page_by_mmuidx_all_cpus(CPUState *cpu, target_ulong addr,
|
|
|
|
uint16_t idxmap);
|
|
|
|
/**
|
|
|
|
* tlb_flush_page_by_mmuidx_all_cpus_synced:
|
|
|
|
* @cpu: Originating CPU of the flush
|
|
|
|
* @addr: virtual address of page to be flushed
|
|
|
|
* @idxmap: bitmap of MMU indexes to flush
|
|
|
|
*
|
|
|
|
* Flush one page from the TLB of all CPUs, for the specified MMU
|
|
|
|
* indexes like tlb_flush_page_by_mmuidx_all_cpus except the source
|
|
|
|
* vCPUs work is scheduled as safe work meaning all flushes will be
|
|
|
|
* complete once the source vCPUs safe work is complete. This will
|
|
|
|
* depend on when the guests translation ends the TB.
|
|
|
|
*/
|
|
|
|
void tlb_flush_page_by_mmuidx_all_cpus_synced(CPUState *cpu, target_ulong addr,
|
|
|
|
uint16_t idxmap);
|
2015-08-25 17:45:09 +03:00
|
|
|
/**
|
|
|
|
* tlb_flush_by_mmuidx:
|
|
|
|
* @cpu: CPU whose TLB should be flushed
|
2017-02-23 21:29:22 +03:00
|
|
|
* @wait: If true ensure synchronisation by exiting the cpu_loop
|
2017-02-23 21:29:19 +03:00
|
|
|
* @idxmap: bitmap of MMU indexes to flush
|
2015-08-25 17:45:09 +03:00
|
|
|
*
|
|
|
|
* Flush all entries from the TLB of the specified CPU, for the specified
|
|
|
|
* MMU indexes.
|
|
|
|
*/
|
2017-02-23 21:29:19 +03:00
|
|
|
void tlb_flush_by_mmuidx(CPUState *cpu, uint16_t idxmap);
|
2017-02-23 21:29:22 +03:00
|
|
|
/**
|
|
|
|
* tlb_flush_by_mmuidx_all_cpus:
|
|
|
|
* @cpu: Originating CPU of the flush
|
|
|
|
* @idxmap: bitmap of MMU indexes to flush
|
|
|
|
*
|
|
|
|
* Flush all entries from all TLBs of all CPUs, for the specified
|
|
|
|
* MMU indexes.
|
|
|
|
*/
|
|
|
|
void tlb_flush_by_mmuidx_all_cpus(CPUState *cpu, uint16_t idxmap);
|
|
|
|
/**
|
|
|
|
* tlb_flush_by_mmuidx_all_cpus_synced:
|
|
|
|
* @cpu: Originating CPU of the flush
|
|
|
|
* @idxmap: bitmap of MMU indexes to flush
|
|
|
|
*
|
|
|
|
* Flush all entries from all TLBs of all CPUs, for the specified
|
|
|
|
* MMU indexes like tlb_flush_by_mmuidx_all_cpus except except the source
|
|
|
|
* vCPUs work is scheduled as safe work meaning all flushes will be
|
|
|
|
* complete once the source vCPUs safe work is complete. This will
|
|
|
|
* depend on when the guests translation ends the TB.
|
|
|
|
*/
|
|
|
|
void tlb_flush_by_mmuidx_all_cpus_synced(CPUState *cpu, uint16_t idxmap);
|
2020-10-17 00:07:53 +03:00
|
|
|
|
|
|
|
/**
|
|
|
|
* tlb_flush_page_bits_by_mmuidx
|
|
|
|
* @cpu: CPU whose TLB should be flushed
|
|
|
|
* @addr: virtual address of page to be flushed
|
|
|
|
* @idxmap: bitmap of mmu indexes to flush
|
|
|
|
* @bits: number of significant bits in address
|
|
|
|
*
|
|
|
|
* Similar to tlb_flush_page_mask, but with a bitmap of indexes.
|
|
|
|
*/
|
|
|
|
void tlb_flush_page_bits_by_mmuidx(CPUState *cpu, target_ulong addr,
|
|
|
|
uint16_t idxmap, unsigned bits);
|
|
|
|
|
|
|
|
/* Similarly, with broadcast and syncing. */
|
|
|
|
void tlb_flush_page_bits_by_mmuidx_all_cpus(CPUState *cpu, target_ulong addr,
|
|
|
|
uint16_t idxmap, unsigned bits);
|
|
|
|
void tlb_flush_page_bits_by_mmuidx_all_cpus_synced
|
|
|
|
(CPUState *cpu, target_ulong addr, uint16_t idxmap, unsigned bits);
|
|
|
|
|
2021-05-09 18:16:13 +03:00
|
|
|
/**
|
|
|
|
* tlb_flush_range_by_mmuidx
|
|
|
|
* @cpu: CPU whose TLB should be flushed
|
|
|
|
* @addr: virtual address of the start of the range to be flushed
|
|
|
|
* @len: length of range to be flushed
|
|
|
|
* @idxmap: bitmap of mmu indexes to flush
|
|
|
|
* @bits: number of significant bits in address
|
|
|
|
*
|
|
|
|
* For each mmuidx in @idxmap, flush all pages within [@addr,@addr+@len),
|
|
|
|
* comparing only the low @bits worth of each virtual page.
|
|
|
|
*/
|
|
|
|
void tlb_flush_range_by_mmuidx(CPUState *cpu, target_ulong addr,
|
|
|
|
target_ulong len, uint16_t idxmap,
|
|
|
|
unsigned bits);
|
2021-05-09 18:16:14 +03:00
|
|
|
|
|
|
|
/* Similarly, with broadcast and syncing. */
|
|
|
|
void tlb_flush_range_by_mmuidx_all_cpus(CPUState *cpu, target_ulong addr,
|
|
|
|
target_ulong len, uint16_t idxmap,
|
|
|
|
unsigned bits);
|
2021-05-09 18:16:15 +03:00
|
|
|
void tlb_flush_range_by_mmuidx_all_cpus_synced(CPUState *cpu,
|
|
|
|
target_ulong addr,
|
|
|
|
target_ulong len,
|
|
|
|
uint16_t idxmap,
|
|
|
|
unsigned bits);
|
2021-05-09 18:16:14 +03:00
|
|
|
|
2022-08-20 02:33:23 +03:00
|
|
|
/**
|
|
|
|
* tlb_set_page_full:
|
|
|
|
* @cpu: CPU context
|
|
|
|
* @mmu_idx: mmu index of the tlb to modify
|
|
|
|
* @vaddr: virtual address of the entry to add
|
|
|
|
* @full: the details of the tlb entry
|
|
|
|
*
|
|
|
|
* Add an entry to @cpu tlb index @mmu_idx. All of the fields of
|
|
|
|
* @full must be filled, except for xlat_section, and constitute
|
|
|
|
* the complete description of the translated page.
|
|
|
|
*
|
|
|
|
* This is generally called by the target tlb_fill function after
|
|
|
|
* having performed a successful page table walk to find the physical
|
|
|
|
* address and attributes for the translation.
|
|
|
|
*
|
|
|
|
* At most one entry for a given virtual address is permitted. Only a
|
|
|
|
* single TARGET_PAGE_SIZE region is mapped; @full->lg_page_size is only
|
|
|
|
* used by tlb_flush_page.
|
|
|
|
*/
|
|
|
|
void tlb_set_page_full(CPUState *cpu, int mmu_idx, target_ulong vaddr,
|
|
|
|
CPUTLBEntryFull *full);
|
|
|
|
|
2016-01-21 17:15:04 +03:00
|
|
|
/**
|
|
|
|
* tlb_set_page_with_attrs:
|
|
|
|
* @cpu: CPU to add this TLB entry for
|
|
|
|
* @vaddr: virtual address of page to add entry for
|
|
|
|
* @paddr: physical address of the page
|
|
|
|
* @attrs: memory transaction attributes
|
|
|
|
* @prot: access permissions (PAGE_READ/PAGE_WRITE/PAGE_EXEC bits)
|
|
|
|
* @mmu_idx: MMU index to insert TLB entry for
|
|
|
|
* @size: size of the page in bytes
|
|
|
|
*
|
|
|
|
* Add an entry to this CPU's TLB (a mapping from virtual address
|
|
|
|
* @vaddr to physical address @paddr) with the specified memory
|
|
|
|
* transaction attributes. This is generally called by the target CPU
|
|
|
|
* specific code after it has been called through the tlb_fill()
|
|
|
|
* entry point and performed a successful page table walk to find
|
|
|
|
* the physical address and attributes for the virtual address
|
|
|
|
* which provoked the TLB miss.
|
|
|
|
*
|
|
|
|
* At most one entry for a given virtual address is permitted. Only a
|
|
|
|
* single TARGET_PAGE_SIZE region is mapped; the supplied @size is only
|
|
|
|
* used by tlb_flush_page.
|
|
|
|
*/
|
2015-04-26 18:49:24 +03:00
|
|
|
void tlb_set_page_with_attrs(CPUState *cpu, target_ulong vaddr,
|
|
|
|
hwaddr paddr, MemTxAttrs attrs,
|
|
|
|
int prot, int mmu_idx, target_ulong size);
|
2016-01-21 17:15:04 +03:00
|
|
|
/* tlb_set_page:
|
|
|
|
*
|
|
|
|
* This function is equivalent to calling tlb_set_page_with_attrs()
|
|
|
|
* with an @attrs argument of MEMTXATTRS_UNSPECIFIED. It's provided
|
|
|
|
* as a convenience for CPUs which don't use memory transaction attributes.
|
|
|
|
*/
|
|
|
|
void tlb_set_page(CPUState *cpu, target_ulong vaddr,
|
|
|
|
hwaddr paddr, int prot,
|
|
|
|
int mmu_idx, target_ulong size);
|
2012-04-09 20:50:52 +04:00
|
|
|
#else
|
2018-10-09 20:45:54 +03:00
|
|
|
static inline void tlb_init(CPUState *cpu)
|
|
|
|
{
|
|
|
|
}
|
2020-06-12 22:02:26 +03:00
|
|
|
static inline void tlb_destroy(CPUState *cpu)
|
|
|
|
{
|
|
|
|
}
|
2013-09-04 03:29:02 +04:00
|
|
|
static inline void tlb_flush_page(CPUState *cpu, target_ulong addr)
|
2012-04-09 20:50:52 +04:00
|
|
|
{
|
|
|
|
}
|
2017-02-23 21:29:22 +03:00
|
|
|
static inline void tlb_flush_page_all_cpus(CPUState *src, target_ulong addr)
|
|
|
|
{
|
|
|
|
}
|
|
|
|
static inline void tlb_flush_page_all_cpus_synced(CPUState *src,
|
|
|
|
target_ulong addr)
|
|
|
|
{
|
|
|
|
}
|
2016-11-14 17:17:28 +03:00
|
|
|
static inline void tlb_flush(CPUState *cpu)
|
2012-04-09 20:50:52 +04:00
|
|
|
{
|
|
|
|
}
|
2017-02-23 21:29:22 +03:00
|
|
|
static inline void tlb_flush_all_cpus(CPUState *src_cpu)
|
|
|
|
{
|
|
|
|
}
|
|
|
|
static inline void tlb_flush_all_cpus_synced(CPUState *src_cpu)
|
|
|
|
{
|
|
|
|
}
|
2015-08-25 17:45:09 +03:00
|
|
|
static inline void tlb_flush_page_by_mmuidx(CPUState *cpu,
|
2017-02-23 21:29:19 +03:00
|
|
|
target_ulong addr, uint16_t idxmap)
|
2015-08-25 17:45:09 +03:00
|
|
|
{
|
|
|
|
}
|
|
|
|
|
2017-02-23 21:29:19 +03:00
|
|
|
static inline void tlb_flush_by_mmuidx(CPUState *cpu, uint16_t idxmap)
|
2015-08-25 17:45:09 +03:00
|
|
|
{
|
|
|
|
}
|
2017-02-23 21:29:22 +03:00
|
|
|
static inline void tlb_flush_page_by_mmuidx_all_cpus(CPUState *cpu,
|
|
|
|
target_ulong addr,
|
|
|
|
uint16_t idxmap)
|
|
|
|
{
|
|
|
|
}
|
|
|
|
static inline void tlb_flush_page_by_mmuidx_all_cpus_synced(CPUState *cpu,
|
|
|
|
target_ulong addr,
|
|
|
|
uint16_t idxmap)
|
|
|
|
{
|
|
|
|
}
|
|
|
|
static inline void tlb_flush_by_mmuidx_all_cpus(CPUState *cpu, uint16_t idxmap)
|
|
|
|
{
|
|
|
|
}
|
2018-05-30 12:58:36 +03:00
|
|
|
|
2017-02-23 21:29:22 +03:00
|
|
|
static inline void tlb_flush_by_mmuidx_all_cpus_synced(CPUState *cpu,
|
|
|
|
uint16_t idxmap)
|
|
|
|
{
|
|
|
|
}
|
2020-10-17 00:07:53 +03:00
|
|
|
static inline void tlb_flush_page_bits_by_mmuidx(CPUState *cpu,
|
|
|
|
target_ulong addr,
|
|
|
|
uint16_t idxmap,
|
|
|
|
unsigned bits)
|
|
|
|
{
|
|
|
|
}
|
|
|
|
static inline void tlb_flush_page_bits_by_mmuidx_all_cpus(CPUState *cpu,
|
|
|
|
target_ulong addr,
|
|
|
|
uint16_t idxmap,
|
|
|
|
unsigned bits)
|
|
|
|
{
|
|
|
|
}
|
|
|
|
static inline void
|
|
|
|
tlb_flush_page_bits_by_mmuidx_all_cpus_synced(CPUState *cpu, target_ulong addr,
|
|
|
|
uint16_t idxmap, unsigned bits)
|
|
|
|
{
|
|
|
|
}
|
2021-05-09 18:16:13 +03:00
|
|
|
static inline void tlb_flush_range_by_mmuidx(CPUState *cpu, target_ulong addr,
|
|
|
|
target_ulong len, uint16_t idxmap,
|
|
|
|
unsigned bits)
|
|
|
|
{
|
|
|
|
}
|
2021-05-09 18:16:14 +03:00
|
|
|
static inline void tlb_flush_range_by_mmuidx_all_cpus(CPUState *cpu,
|
|
|
|
target_ulong addr,
|
|
|
|
target_ulong len,
|
|
|
|
uint16_t idxmap,
|
|
|
|
unsigned bits)
|
|
|
|
{
|
|
|
|
}
|
2021-05-09 18:16:15 +03:00
|
|
|
static inline void tlb_flush_range_by_mmuidx_all_cpus_synced(CPUState *cpu,
|
|
|
|
target_ulong addr,
|
|
|
|
target_long len,
|
|
|
|
uint16_t idxmap,
|
|
|
|
unsigned bits)
|
|
|
|
{
|
|
|
|
}
|
2010-03-01 06:31:14 +03:00
|
|
|
#endif
|
2020-05-08 18:43:43 +03:00
|
|
|
/**
|
|
|
|
* probe_access:
|
|
|
|
* @env: CPUArchState
|
|
|
|
* @addr: guest virtual address to look up
|
|
|
|
* @size: size of the access
|
|
|
|
* @access_type: read, write or execute permission
|
|
|
|
* @mmu_idx: MMU index to use for lookup
|
|
|
|
* @retaddr: return address for unwinding
|
|
|
|
*
|
|
|
|
* Look up the guest virtual address @addr. Raise an exception if the
|
|
|
|
* page does not satisfy @access_type. Raise an exception if the
|
|
|
|
* access (@addr, @size) hits a watchpoint. For writes, mark a clean
|
|
|
|
* page as dirty.
|
|
|
|
*
|
|
|
|
* Finally, return the host address for a page that is backed by RAM,
|
|
|
|
* or NULL if the page requires I/O.
|
|
|
|
*/
|
2019-08-30 13:09:59 +03:00
|
|
|
void *probe_access(CPUArchState *env, target_ulong addr, int size,
|
|
|
|
MMUAccessType access_type, int mmu_idx, uintptr_t retaddr);
|
|
|
|
|
|
|
|
static inline void *probe_write(CPUArchState *env, target_ulong addr, int size,
|
|
|
|
int mmu_idx, uintptr_t retaddr)
|
|
|
|
{
|
|
|
|
return probe_access(env, addr, size, MMU_DATA_STORE, mmu_idx, retaddr);
|
|
|
|
}
|
2003-05-25 20:46:15 +04:00
|
|
|
|
2019-11-21 03:08:40 +03:00
|
|
|
static inline void *probe_read(CPUArchState *env, target_ulong addr, int size,
|
|
|
|
int mmu_idx, uintptr_t retaddr)
|
|
|
|
{
|
|
|
|
return probe_access(env, addr, size, MMU_DATA_LOAD, mmu_idx, retaddr);
|
|
|
|
}
|
|
|
|
|
2020-05-08 18:43:45 +03:00
|
|
|
/**
|
|
|
|
* probe_access_flags:
|
|
|
|
* @env: CPUArchState
|
|
|
|
* @addr: guest virtual address to look up
|
2023-02-24 02:44:24 +03:00
|
|
|
* @size: size of the access
|
2020-05-08 18:43:45 +03:00
|
|
|
* @access_type: read, write or execute permission
|
|
|
|
* @mmu_idx: MMU index to use for lookup
|
|
|
|
* @nonfault: suppress the fault
|
|
|
|
* @phost: return value for host address
|
|
|
|
* @retaddr: return address for unwinding
|
|
|
|
*
|
|
|
|
* Similar to probe_access, loosely returning the TLB_FLAGS_MASK for
|
|
|
|
* the page, and storing the host address for RAM in @phost.
|
|
|
|
*
|
|
|
|
* If @nonfault is set, do not raise an exception but return TLB_INVALID_MASK.
|
|
|
|
* Do not handle watchpoints, but include TLB_WATCHPOINT in the returned flags.
|
|
|
|
* Do handle clean pages, so exclude TLB_NOTDIRY from the returned flags.
|
|
|
|
* For simplicity, all "mmio-like" flags are folded to TLB_MMIO.
|
|
|
|
*/
|
2023-02-24 02:44:24 +03:00
|
|
|
int probe_access_flags(CPUArchState *env, target_ulong addr, int size,
|
2020-05-08 18:43:45 +03:00
|
|
|
MMUAccessType access_type, int mmu_idx,
|
|
|
|
bool nonfault, void **phost, uintptr_t retaddr);
|
|
|
|
|
2022-08-20 01:49:41 +03:00
|
|
|
#ifndef CONFIG_USER_ONLY
|
|
|
|
/**
|
|
|
|
* probe_access_full:
|
|
|
|
* Like probe_access_flags, except also return into @pfull.
|
|
|
|
*
|
|
|
|
* The CPUTLBEntryFull structure returned via @pfull is transient
|
|
|
|
* and must be consumed or copied immediately, before any further
|
|
|
|
* access or changes to TLB @mmu_idx.
|
|
|
|
*/
|
2023-02-24 03:44:14 +03:00
|
|
|
int probe_access_full(CPUArchState *env, target_ulong addr, int size,
|
2022-08-20 01:49:41 +03:00
|
|
|
MMUAccessType access_type, int mmu_idx,
|
|
|
|
bool nonfault, void **phost,
|
|
|
|
CPUTLBEntryFull **pfull, uintptr_t retaddr);
|
|
|
|
#endif
|
|
|
|
|
2003-05-25 20:46:15 +04:00
|
|
|
#define CODE_GEN_ALIGN 16 /* must be >= of the size of a icache line */
|
|
|
|
|
2015-09-26 19:23:42 +03:00
|
|
|
/* Estimated block size for TB allocation. */
|
|
|
|
/* ??? The following is based on a 2015 survey of x86_64 host output.
|
|
|
|
Better would seem to be some sort of dynamically sized TB array,
|
|
|
|
adapting to the block sizes actually being produced. */
|
2004-01-04 21:03:10 +03:00
|
|
|
#if defined(CONFIG_SOFTMMU)
|
2015-09-26 19:23:42 +03:00
|
|
|
#define CODE_GEN_AVG_BLOCK_SIZE 400
|
2004-01-04 21:03:10 +03:00
|
|
|
#else
|
2015-09-26 19:23:42 +03:00
|
|
|
#define CODE_GEN_AVG_BLOCK_SIZE 150
|
2004-01-04 21:03:10 +03:00
|
|
|
#endif
|
|
|
|
|
2017-07-12 07:08:21 +03:00
|
|
|
/*
|
|
|
|
* Translation Cache-related fields of a TB.
|
translate-all: use a binary search tree to track TBs in TBContext
This is a prerequisite for supporting multiple TCG contexts, since
we will have threads generating code in separate regions of
code_gen_buffer.
For this we need a new field (.size) in struct tb_tc to keep
track of the size of the translated code. This field uses a size_t
to avoid adding a hole to the struct, although really an unsigned
int would have been enough.
The comparison function we use is optimized for the common case:
insertions. Profiling shows that upon booting debian-arm, 98%
of comparisons are between existing tb's (i.e. a->size and b->size
are both !0), which happens during insertions (and removals, but
those are rare). The remaining cases are lookups. From reading the glib
sources we see that the first key is always the lookup key. However,
the code does not assume this to always be the case because this
behaviour is not guaranteed in the glib docs. However, we embed
this knowledge in the code as a branch hint for the compiler.
Note that tb_free does not free space in the code_gen_buffer anymore,
since we cannot easily know whether the tb is the last one inserted
in code_gen_buffer. The next patch in this series renames tb_free
to tb_remove to reflect this.
Performance-wise, lookups in tb_find_pc are the same as before:
O(log n). However, insertions are O(log n) instead of O(1), which
results in a small slowdown when booting debian-arm:
Performance counter stats for 'build/arm-softmmu/qemu-system-arm \
-machine type=virt -nographic -smp 1 -m 4096 \
-netdev user,id=unet,hostfwd=tcp::2222-:22 \
-device virtio-net-device,netdev=unet \
-drive file=img/arm/jessie-arm32.qcow2,id=myblock,index=0,if=none \
-device virtio-blk-device,drive=myblock \
-kernel img/arm/aarch32-current-linux-kernel-only.img \
-append console=ttyAMA0 root=/dev/vda1 \
-name arm,debug-threads=on -smp 1' (10 runs):
- Before:
8048.598422 task-clock (msec) # 0.931 CPUs utilized ( +- 0.28% )
16,974 context-switches # 0.002 M/sec ( +- 0.12% )
0 cpu-migrations # 0.000 K/sec
10,125 page-faults # 0.001 M/sec ( +- 1.23% )
35,144,901,879 cycles # 4.367 GHz ( +- 0.14% )
<not supported> stalled-cycles-frontend
<not supported> stalled-cycles-backend
65,758,252,643 instructions # 1.87 insns per cycle ( +- 0.33% )
10,871,298,668 branches # 1350.707 M/sec ( +- 0.41% )
192,322,212 branch-misses # 1.77% of all branches ( +- 0.32% )
8.640869419 seconds time elapsed ( +- 0.57% )
- After:
8146.242027 task-clock (msec) # 0.923 CPUs utilized ( +- 1.23% )
17,016 context-switches # 0.002 M/sec ( +- 0.40% )
0 cpu-migrations # 0.000 K/sec
18,769 page-faults # 0.002 M/sec ( +- 0.45% )
35,660,956,120 cycles # 4.378 GHz ( +- 1.22% )
<not supported> stalled-cycles-frontend
<not supported> stalled-cycles-backend
65,095,366,607 instructions # 1.83 insns per cycle ( +- 1.73% )
10,803,480,261 branches # 1326.192 M/sec ( +- 1.95% )
195,601,289 branch-misses # 1.81% of all branches ( +- 0.39% )
8.828660235 seconds time elapsed ( +- 0.38% )
Reviewed-by: Richard Henderson <rth@twiddle.net>
Signed-off-by: Emilio G. Cota <cota@braap.org>
Signed-off-by: Richard Henderson <richard.henderson@linaro.org>
2017-06-24 02:00:11 +03:00
|
|
|
* This struct exists just for convenience; we keep track of TB's in a binary
|
|
|
|
* search tree, and the only fields needed to compare TB's in the tree are
|
|
|
|
* @ptr and @size.
|
|
|
|
* Note: the address of search data can be obtained by adding @size to @ptr.
|
2017-07-12 07:08:21 +03:00
|
|
|
*/
|
|
|
|
struct tb_tc {
|
2020-10-28 22:05:44 +03:00
|
|
|
const void *ptr; /* pointer to the translated code */
|
translate-all: use a binary search tree to track TBs in TBContext
This is a prerequisite for supporting multiple TCG contexts, since
we will have threads generating code in separate regions of
code_gen_buffer.
For this we need a new field (.size) in struct tb_tc to keep
track of the size of the translated code. This field uses a size_t
to avoid adding a hole to the struct, although really an unsigned
int would have been enough.
The comparison function we use is optimized for the common case:
insertions. Profiling shows that upon booting debian-arm, 98%
of comparisons are between existing tb's (i.e. a->size and b->size
are both !0), which happens during insertions (and removals, but
those are rare). The remaining cases are lookups. From reading the glib
sources we see that the first key is always the lookup key. However,
the code does not assume this to always be the case because this
behaviour is not guaranteed in the glib docs. However, we embed
this knowledge in the code as a branch hint for the compiler.
Note that tb_free does not free space in the code_gen_buffer anymore,
since we cannot easily know whether the tb is the last one inserted
in code_gen_buffer. The next patch in this series renames tb_free
to tb_remove to reflect this.
Performance-wise, lookups in tb_find_pc are the same as before:
O(log n). However, insertions are O(log n) instead of O(1), which
results in a small slowdown when booting debian-arm:
Performance counter stats for 'build/arm-softmmu/qemu-system-arm \
-machine type=virt -nographic -smp 1 -m 4096 \
-netdev user,id=unet,hostfwd=tcp::2222-:22 \
-device virtio-net-device,netdev=unet \
-drive file=img/arm/jessie-arm32.qcow2,id=myblock,index=0,if=none \
-device virtio-blk-device,drive=myblock \
-kernel img/arm/aarch32-current-linux-kernel-only.img \
-append console=ttyAMA0 root=/dev/vda1 \
-name arm,debug-threads=on -smp 1' (10 runs):
- Before:
8048.598422 task-clock (msec) # 0.931 CPUs utilized ( +- 0.28% )
16,974 context-switches # 0.002 M/sec ( +- 0.12% )
0 cpu-migrations # 0.000 K/sec
10,125 page-faults # 0.001 M/sec ( +- 1.23% )
35,144,901,879 cycles # 4.367 GHz ( +- 0.14% )
<not supported> stalled-cycles-frontend
<not supported> stalled-cycles-backend
65,758,252,643 instructions # 1.87 insns per cycle ( +- 0.33% )
10,871,298,668 branches # 1350.707 M/sec ( +- 0.41% )
192,322,212 branch-misses # 1.77% of all branches ( +- 0.32% )
8.640869419 seconds time elapsed ( +- 0.57% )
- After:
8146.242027 task-clock (msec) # 0.923 CPUs utilized ( +- 1.23% )
17,016 context-switches # 0.002 M/sec ( +- 0.40% )
0 cpu-migrations # 0.000 K/sec
18,769 page-faults # 0.002 M/sec ( +- 0.45% )
35,660,956,120 cycles # 4.378 GHz ( +- 1.22% )
<not supported> stalled-cycles-frontend
<not supported> stalled-cycles-backend
65,095,366,607 instructions # 1.83 insns per cycle ( +- 1.73% )
10,803,480,261 branches # 1326.192 M/sec ( +- 1.95% )
195,601,289 branch-misses # 1.81% of all branches ( +- 0.39% )
8.828660235 seconds time elapsed ( +- 0.38% )
Reviewed-by: Richard Henderson <rth@twiddle.net>
Signed-off-by: Emilio G. Cota <cota@braap.org>
Signed-off-by: Richard Henderson <richard.henderson@linaro.org>
2017-06-24 02:00:11 +03:00
|
|
|
size_t size;
|
2017-07-12 07:08:21 +03:00
|
|
|
};
|
|
|
|
|
2008-06-29 05:03:05 +04:00
|
|
|
struct TranslationBlock {
|
2022-08-12 19:53:53 +03:00
|
|
|
/*
|
|
|
|
* Guest PC corresponding to this block. This must be the true
|
|
|
|
* virtual address. Therefore e.g. x86 stores EIP + CS_BASE, and
|
|
|
|
* targets like Arm, MIPS, HP-PA, which reuse low bits for ISA or
|
|
|
|
* privilege, must store those bits elsewhere.
|
|
|
|
*
|
2023-02-27 16:51:40 +03:00
|
|
|
* If CF_PCREL, the opcodes for the TranslationBlock are written
|
|
|
|
* such that the TB is associated only with the physical page and
|
|
|
|
* may be run in any virtual address context. In this case, PC
|
|
|
|
* must always be taken from ENV in a target-specific manner.
|
2022-08-12 19:53:53 +03:00
|
|
|
* Unwind information is taken as offsets from the page, to be
|
|
|
|
* deposited into the "current" PC.
|
|
|
|
*/
|
|
|
|
target_ulong pc;
|
|
|
|
|
|
|
|
/*
|
|
|
|
* Target-specific data associated with the TranslationBlock, e.g.:
|
|
|
|
* x86: the original user, the Code Segment virtual base,
|
|
|
|
* arm: an extension of tb->flags,
|
|
|
|
* s390x: instruction data for EXECUTE,
|
|
|
|
* sparc: the next pc of the instruction queue (for delay slots).
|
|
|
|
*/
|
|
|
|
target_ulong cs_base;
|
|
|
|
|
2016-04-07 20:19:22 +03:00
|
|
|
uint32_t flags; /* flags defining in which context the code was generated */
|
2014-11-26 13:39:53 +03:00
|
|
|
uint32_t cflags; /* compile flags */
|
2021-07-18 01:18:39 +03:00
|
|
|
|
|
|
|
/* Note that TCG_MAX_INSNS is 512; we validate this match elsewhere. */
|
2021-07-18 01:18:41 +03:00
|
|
|
#define CF_COUNT_MASK 0x000001ff
|
|
|
|
#define CF_NO_GOTO_TB 0x00000200 /* Do not chain with goto_tb */
|
|
|
|
#define CF_NO_GOTO_PTR 0x00000400 /* Do not chain with goto_ptr */
|
2021-07-19 23:43:46 +03:00
|
|
|
#define CF_SINGLE_STEP 0x00000800 /* gdbstub single-step in effect */
|
2021-07-18 01:18:41 +03:00
|
|
|
#define CF_LAST_IO 0x00008000 /* Last insn may be an IO access. */
|
|
|
|
#define CF_MEMI_ONLY 0x00010000 /* Only instrument memory ops */
|
|
|
|
#define CF_USE_ICOUNT 0x00020000
|
|
|
|
#define CF_INVALID 0x00040000 /* TB is stale. Set with @jmp_lock held */
|
|
|
|
#define CF_PARALLEL 0x00080000 /* Generate code for a parallel context */
|
2021-11-29 17:09:25 +03:00
|
|
|
#define CF_NOIRQ 0x00100000 /* Generate an uninterruptible TB */
|
2023-02-27 16:51:36 +03:00
|
|
|
#define CF_PCREL 0x00200000 /* Opcodes in TB are PC-relative */
|
2021-07-18 01:18:41 +03:00
|
|
|
#define CF_CLUSTER_MASK 0xff000000 /* Top 8 bits are cluster ID */
|
2019-01-29 14:46:06 +03:00
|
|
|
#define CF_CLUSTER_SHIFT 24
|
2004-02-17 01:11:32 +03:00
|
|
|
|
2021-02-24 19:58:10 +03:00
|
|
|
/*
|
|
|
|
* Above fields used for comparing
|
|
|
|
*/
|
|
|
|
|
|
|
|
/* size of target code for this block (1 <= size <= TARGET_PAGE_SIZE) */
|
|
|
|
uint16_t size;
|
|
|
|
uint16_t icount;
|
|
|
|
|
2017-07-12 07:08:21 +03:00
|
|
|
struct tb_tc tc;
|
|
|
|
|
2022-10-01 23:36:33 +03:00
|
|
|
/*
|
|
|
|
* Track tb_page_addr_t intervals that intersect this TB.
|
|
|
|
* For user-only, the virtual addresses are always contiguous,
|
|
|
|
* and we use a unified interval tree. For system, we use a
|
|
|
|
* linked list headed in each PageDesc. Within the list, the lsb
|
|
|
|
* of the previous pointer tells the index of page_next[], and the
|
|
|
|
* list is protected by the PageDesc lock(s).
|
|
|
|
*/
|
|
|
|
#ifdef CONFIG_USER_ONLY
|
|
|
|
IntervalTreeNode itree;
|
|
|
|
#else
|
2017-08-04 01:37:15 +03:00
|
|
|
uintptr_t page_next[2];
|
2010-03-12 19:54:58 +03:00
|
|
|
tb_page_addr_t page_addr[2];
|
2022-10-01 23:36:33 +03:00
|
|
|
#endif
|
2004-01-04 21:03:10 +03:00
|
|
|
|
translate-all: protect TB jumps with a per-destination-TB lock
This applies to both user-mode and !user-mode emulation.
Instead of relying on a global lock, protect the list of incoming
jumps with tb->jmp_lock. This lock also protects tb->cflags,
so update all tb->cflags readers outside tb->jmp_lock to use
atomic reads via tb_cflags().
In order to find the destination TB (and therefore its jmp_lock)
from the origin TB, we introduce tb->jmp_dest[].
I considered not using a linked list of jumps, which simplifies
code and makes the struct smaller. However, it unnecessarily increases
memory usage, which results in a performance decrease. See for
instance these numbers booting+shutting down debian-arm:
Time (s) Rel. err (%) Abs. err (s) Rel. slowdown (%)
------------------------------------------------------------------------------
before 20.88 0.74 0.154512 0.
after 20.81 0.38 0.079078 -0.33524904
GTree 21.02 0.28 0.058856 0.67049808
GHashTable + xxhash 21.63 1.08 0.233604 3.5919540
Using a hash table or a binary tree to keep track of the jumps
doesn't really pay off, not only due to the increased memory usage,
but also because most TBs have only 0 or 1 jumps to them. The maximum
number of jumps when booting debian-arm that I measured is 35, but
as we can see in the histogram below a TB with that many incoming jumps
is extremely rare; the average TB has 0.80 incoming jumps.
n_jumps: 379208; avg jumps/tb: 0.801099
dist: [0.0,1.0)|▄█▁▁▁▁▁▁▁▁▁▁▁ ▁▁▁▁▁▁ ▁▁▁ ▁▁▁ ▁|[34.0,35.0]
Reviewed-by: Richard Henderson <richard.henderson@linaro.org>
Signed-off-by: Emilio G. Cota <cota@braap.org>
Signed-off-by: Richard Henderson <richard.henderson@linaro.org>
2017-08-03 03:34:06 +03:00
|
|
|
/* jmp_lock placed here to fill a 4-byte hole. Its documentation is below */
|
|
|
|
QemuSpin jmp_lock;
|
|
|
|
|
2016-04-10 23:35:45 +03:00
|
|
|
/* The following data are used to directly call another TB from
|
|
|
|
* the code of this one. This can be done either by emitting direct or
|
|
|
|
* indirect native jump instructions. These jumps are reset so that the TB
|
2017-06-24 02:43:01 +03:00
|
|
|
* just continues its execution. The TB can be linked to another one by
|
2016-04-10 23:35:45 +03:00
|
|
|
* setting one of the jump targets (or patching the jump instruction). Only
|
|
|
|
* two of such jumps are supported.
|
|
|
|
*/
|
2022-11-27 05:20:57 +03:00
|
|
|
#define TB_JMP_OFFSET_INVALID 0xffff /* indicates no jump generated */
|
2016-04-10 23:35:45 +03:00
|
|
|
uint16_t jmp_reset_offset[2]; /* offset of original jump target */
|
2022-11-27 05:54:23 +03:00
|
|
|
uint16_t jmp_insn_offset[2]; /* offset of direct jump insn */
|
|
|
|
uintptr_t jmp_target_addr[2]; /* target address */
|
2017-08-01 08:02:31 +03:00
|
|
|
|
translate-all: protect TB jumps with a per-destination-TB lock
This applies to both user-mode and !user-mode emulation.
Instead of relying on a global lock, protect the list of incoming
jumps with tb->jmp_lock. This lock also protects tb->cflags,
so update all tb->cflags readers outside tb->jmp_lock to use
atomic reads via tb_cflags().
In order to find the destination TB (and therefore its jmp_lock)
from the origin TB, we introduce tb->jmp_dest[].
I considered not using a linked list of jumps, which simplifies
code and makes the struct smaller. However, it unnecessarily increases
memory usage, which results in a performance decrease. See for
instance these numbers booting+shutting down debian-arm:
Time (s) Rel. err (%) Abs. err (s) Rel. slowdown (%)
------------------------------------------------------------------------------
before 20.88 0.74 0.154512 0.
after 20.81 0.38 0.079078 -0.33524904
GTree 21.02 0.28 0.058856 0.67049808
GHashTable + xxhash 21.63 1.08 0.233604 3.5919540
Using a hash table or a binary tree to keep track of the jumps
doesn't really pay off, not only due to the increased memory usage,
but also because most TBs have only 0 or 1 jumps to them. The maximum
number of jumps when booting debian-arm that I measured is 35, but
as we can see in the histogram below a TB with that many incoming jumps
is extremely rare; the average TB has 0.80 incoming jumps.
n_jumps: 379208; avg jumps/tb: 0.801099
dist: [0.0,1.0)|▄█▁▁▁▁▁▁▁▁▁▁▁ ▁▁▁▁▁▁ ▁▁▁ ▁▁▁ ▁|[34.0,35.0]
Reviewed-by: Richard Henderson <richard.henderson@linaro.org>
Signed-off-by: Emilio G. Cota <cota@braap.org>
Signed-off-by: Richard Henderson <richard.henderson@linaro.org>
2017-08-03 03:34:06 +03:00
|
|
|
/*
|
|
|
|
* Each TB has a NULL-terminated list (jmp_list_head) of incoming jumps.
|
|
|
|
* Each TB can have two outgoing jumps, and therefore can participate
|
|
|
|
* in two lists. The list entries are kept in jmp_list_next[2]. The least
|
|
|
|
* significant bit (LSB) of the pointers in these lists is used to encode
|
|
|
|
* which of the two list entries is to be used in the pointed TB.
|
|
|
|
*
|
|
|
|
* List traversals are protected by jmp_lock. The destination TB of each
|
|
|
|
* outgoing jump is kept in jmp_dest[] so that the appropriate jmp_lock
|
|
|
|
* can be acquired from any origin TB.
|
|
|
|
*
|
|
|
|
* jmp_dest[] are tagged pointers as well. The LSB is set when the TB is
|
|
|
|
* being invalidated, so that no further outgoing jumps from it can be set.
|
|
|
|
*
|
|
|
|
* jmp_lock also protects the CF_INVALID cflag; a jump must not be chained
|
|
|
|
* to a destination TB that has CF_INVALID set.
|
2016-04-10 23:35:45 +03:00
|
|
|
*/
|
translate-all: protect TB jumps with a per-destination-TB lock
This applies to both user-mode and !user-mode emulation.
Instead of relying on a global lock, protect the list of incoming
jumps with tb->jmp_lock. This lock also protects tb->cflags,
so update all tb->cflags readers outside tb->jmp_lock to use
atomic reads via tb_cflags().
In order to find the destination TB (and therefore its jmp_lock)
from the origin TB, we introduce tb->jmp_dest[].
I considered not using a linked list of jumps, which simplifies
code and makes the struct smaller. However, it unnecessarily increases
memory usage, which results in a performance decrease. See for
instance these numbers booting+shutting down debian-arm:
Time (s) Rel. err (%) Abs. err (s) Rel. slowdown (%)
------------------------------------------------------------------------------
before 20.88 0.74 0.154512 0.
after 20.81 0.38 0.079078 -0.33524904
GTree 21.02 0.28 0.058856 0.67049808
GHashTable + xxhash 21.63 1.08 0.233604 3.5919540
Using a hash table or a binary tree to keep track of the jumps
doesn't really pay off, not only due to the increased memory usage,
but also because most TBs have only 0 or 1 jumps to them. The maximum
number of jumps when booting debian-arm that I measured is 35, but
as we can see in the histogram below a TB with that many incoming jumps
is extremely rare; the average TB has 0.80 incoming jumps.
n_jumps: 379208; avg jumps/tb: 0.801099
dist: [0.0,1.0)|▄█▁▁▁▁▁▁▁▁▁▁▁ ▁▁▁▁▁▁ ▁▁▁ ▁▁▁ ▁|[34.0,35.0]
Reviewed-by: Richard Henderson <richard.henderson@linaro.org>
Signed-off-by: Emilio G. Cota <cota@braap.org>
Signed-off-by: Richard Henderson <richard.henderson@linaro.org>
2017-08-03 03:34:06 +03:00
|
|
|
uintptr_t jmp_list_head;
|
2016-03-21 23:11:00 +03:00
|
|
|
uintptr_t jmp_list_next[2];
|
translate-all: protect TB jumps with a per-destination-TB lock
This applies to both user-mode and !user-mode emulation.
Instead of relying on a global lock, protect the list of incoming
jumps with tb->jmp_lock. This lock also protects tb->cflags,
so update all tb->cflags readers outside tb->jmp_lock to use
atomic reads via tb_cflags().
In order to find the destination TB (and therefore its jmp_lock)
from the origin TB, we introduce tb->jmp_dest[].
I considered not using a linked list of jumps, which simplifies
code and makes the struct smaller. However, it unnecessarily increases
memory usage, which results in a performance decrease. See for
instance these numbers booting+shutting down debian-arm:
Time (s) Rel. err (%) Abs. err (s) Rel. slowdown (%)
------------------------------------------------------------------------------
before 20.88 0.74 0.154512 0.
after 20.81 0.38 0.079078 -0.33524904
GTree 21.02 0.28 0.058856 0.67049808
GHashTable + xxhash 21.63 1.08 0.233604 3.5919540
Using a hash table or a binary tree to keep track of the jumps
doesn't really pay off, not only due to the increased memory usage,
but also because most TBs have only 0 or 1 jumps to them. The maximum
number of jumps when booting debian-arm that I measured is 35, but
as we can see in the histogram below a TB with that many incoming jumps
is extremely rare; the average TB has 0.80 incoming jumps.
n_jumps: 379208; avg jumps/tb: 0.801099
dist: [0.0,1.0)|▄█▁▁▁▁▁▁▁▁▁▁▁ ▁▁▁▁▁▁ ▁▁▁ ▁▁▁ ▁|[34.0,35.0]
Reviewed-by: Richard Henderson <richard.henderson@linaro.org>
Signed-off-by: Emilio G. Cota <cota@braap.org>
Signed-off-by: Richard Henderson <richard.henderson@linaro.org>
2017-08-03 03:34:06 +03:00
|
|
|
uintptr_t jmp_dest[2];
|
2008-06-29 05:03:05 +04:00
|
|
|
};
|
2003-05-25 20:46:15 +04:00
|
|
|
|
2020-09-23 13:56:46 +03:00
|
|
|
/* Hide the qatomic_read to make code a little easier on the eyes */
|
2017-07-11 21:29:37 +03:00
|
|
|
static inline uint32_t tb_cflags(const TranslationBlock *tb)
|
|
|
|
{
|
2020-09-23 13:56:46 +03:00
|
|
|
return qatomic_read(&tb->cflags);
|
2017-07-11 21:29:37 +03:00
|
|
|
}
|
|
|
|
|
2022-09-20 14:21:40 +03:00
|
|
|
static inline tb_page_addr_t tb_page_addr0(const TranslationBlock *tb)
|
|
|
|
{
|
2022-10-01 23:36:33 +03:00
|
|
|
#ifdef CONFIG_USER_ONLY
|
|
|
|
return tb->itree.start;
|
|
|
|
#else
|
2022-09-20 14:21:40 +03:00
|
|
|
return tb->page_addr[0];
|
2022-10-01 23:36:33 +03:00
|
|
|
#endif
|
2022-09-20 14:21:40 +03:00
|
|
|
}
|
|
|
|
|
|
|
|
static inline tb_page_addr_t tb_page_addr1(const TranslationBlock *tb)
|
|
|
|
{
|
2022-10-01 23:36:33 +03:00
|
|
|
#ifdef CONFIG_USER_ONLY
|
|
|
|
tb_page_addr_t next = tb->itree.last & TARGET_PAGE_MASK;
|
|
|
|
return next == (tb->itree.start & TARGET_PAGE_MASK) ? -1 : next;
|
|
|
|
#else
|
2022-09-20 14:21:40 +03:00
|
|
|
return tb->page_addr[1];
|
2022-10-01 23:36:33 +03:00
|
|
|
#endif
|
2022-09-20 14:21:40 +03:00
|
|
|
}
|
|
|
|
|
|
|
|
static inline void tb_set_page_addr0(TranslationBlock *tb,
|
|
|
|
tb_page_addr_t addr)
|
|
|
|
{
|
2022-10-01 23:36:33 +03:00
|
|
|
#ifdef CONFIG_USER_ONLY
|
|
|
|
tb->itree.start = addr;
|
|
|
|
/*
|
|
|
|
* To begin, we record an interval of one byte. When the translation
|
|
|
|
* loop encounters a second page, the interval will be extended to
|
|
|
|
* include the first byte of the second page, which is sufficient to
|
|
|
|
* allow tb_page_addr1() above to work properly. The final corrected
|
|
|
|
* interval will be set by tb_page_add() from tb->size before the
|
|
|
|
* node is added to the interval tree.
|
|
|
|
*/
|
|
|
|
tb->itree.last = addr;
|
|
|
|
#else
|
2022-09-20 14:21:40 +03:00
|
|
|
tb->page_addr[0] = addr;
|
2022-10-01 23:36:33 +03:00
|
|
|
#endif
|
2022-09-20 14:21:40 +03:00
|
|
|
}
|
|
|
|
|
|
|
|
static inline void tb_set_page_addr1(TranslationBlock *tb,
|
|
|
|
tb_page_addr_t addr)
|
|
|
|
{
|
2022-10-01 23:36:33 +03:00
|
|
|
#ifdef CONFIG_USER_ONLY
|
|
|
|
/* Extend the interval to the first byte of the second page. See above. */
|
|
|
|
tb->itree.last = addr;
|
|
|
|
#else
|
2022-09-20 14:21:40 +03:00
|
|
|
tb->page_addr[1] = addr;
|
2022-10-01 23:36:33 +03:00
|
|
|
#endif
|
2022-09-20 14:21:40 +03:00
|
|
|
}
|
|
|
|
|
2017-07-11 21:29:37 +03:00
|
|
|
/* current cflags for hashing/comparison */
|
2021-07-18 01:18:40 +03:00
|
|
|
uint32_t curr_cflags(CPUState *cpu);
|
2017-07-11 21:29:37 +03:00
|
|
|
|
2018-06-29 23:07:10 +03:00
|
|
|
/* TranslationBlock invalidate API */
|
|
|
|
#if defined(CONFIG_USER_ONLY)
|
2018-07-02 15:45:25 +03:00
|
|
|
void tb_invalidate_phys_addr(target_ulong addr);
|
|
|
|
#else
|
|
|
|
void tb_invalidate_phys_addr(AddressSpace *as, hwaddr addr, MemTxAttrs attrs);
|
2018-06-29 23:07:10 +03:00
|
|
|
#endif
|
2010-03-12 19:54:58 +03:00
|
|
|
void tb_phys_invalidate(TranslationBlock *tb, tb_page_addr_t page_addr);
|
2023-03-06 04:30:11 +03:00
|
|
|
void tb_invalidate_phys_range(tb_page_addr_t start, tb_page_addr_t last);
|
2017-08-01 08:02:31 +03:00
|
|
|
void tb_set_jmp_target(TranslationBlock *tb, int n, uintptr_t addr);
|
2003-05-25 20:46:15 +04:00
|
|
|
|
2016-07-26 03:39:16 +03:00
|
|
|
/* GETPC is the true target of the return instruction that we'll execute. */
|
2011-10-05 22:03:02 +04:00
|
|
|
#if defined(CONFIG_TCG_INTERPRETER)
|
2021-01-24 23:57:01 +03:00
|
|
|
extern __thread uintptr_t tci_tb_ptr;
|
2016-07-26 03:39:16 +03:00
|
|
|
# define GETPC() tci_tb_ptr
|
2013-08-27 21:22:54 +04:00
|
|
|
#else
|
2016-07-26 03:39:16 +03:00
|
|
|
# define GETPC() \
|
2013-08-27 21:22:54 +04:00
|
|
|
((uintptr_t)__builtin_extract_return_addr(__builtin_return_address(0)))
|
|
|
|
#endif
|
|
|
|
|
|
|
|
/* The true return address will often point to a host insn that is part of
|
|
|
|
the next translated guest insn. Adjust the address backward to point to
|
|
|
|
the middle of the call insn. Subtracting one would do the job except for
|
|
|
|
several compressed mode architectures (arm, mips) which set the low bit
|
|
|
|
to indicate the compressed mode; subtracting two works around that. It
|
|
|
|
is also the case that there are no host isas that contain a call insn
|
|
|
|
smaller than 4 bytes, so we don't worry about special-casing this. */
|
2015-08-18 06:28:18 +03:00
|
|
|
#define GETPC_ADJ 2
|
2011-09-21 22:13:16 +04:00
|
|
|
|
2004-10-01 02:22:08 +04:00
|
|
|
#if !defined(CONFIG_USER_ONLY)
|
2003-10-28 00:24:54 +03:00
|
|
|
|
2018-06-15 16:57:14 +03:00
|
|
|
/**
|
|
|
|
* iotlb_to_section:
|
|
|
|
* @cpu: CPU performing the access
|
|
|
|
* @index: TCG CPU IOTLB entry
|
|
|
|
*
|
|
|
|
* Given a TCG CPU IOTLB entry, return the MemoryRegionSection that
|
|
|
|
* it refers to. @index will have been initially created and returned
|
|
|
|
* by memory_region_section_get_iotlb().
|
|
|
|
*/
|
|
|
|
struct MemoryRegionSection *iotlb_to_section(CPUState *cpu,
|
|
|
|
hwaddr index, MemTxAttrs attrs);
|
2003-10-28 00:24:54 +03:00
|
|
|
#endif
|
2004-01-04 21:03:10 +03:00
|
|
|
|
2019-02-07 01:11:04 +03:00
|
|
|
/**
|
2022-08-10 23:52:50 +03:00
|
|
|
* get_page_addr_code_hostp()
|
2019-02-07 01:11:04 +03:00
|
|
|
* @env: CPUArchState
|
|
|
|
* @addr: guest virtual address of guest code
|
|
|
|
*
|
2022-08-10 23:52:50 +03:00
|
|
|
* See get_page_addr_code() (full-system version) for documentation on the
|
|
|
|
* return value.
|
|
|
|
*
|
|
|
|
* Sets *@hostp (when @hostp is non-NULL) as follows.
|
|
|
|
* If the return value is -1, sets *@hostp to NULL. Otherwise, sets *@hostp
|
|
|
|
* to the host address where @addr's content is kept.
|
|
|
|
*
|
|
|
|
* Note: this function can trigger an exception.
|
2019-02-07 01:11:04 +03:00
|
|
|
*/
|
2022-08-10 23:52:50 +03:00
|
|
|
tb_page_addr_t get_page_addr_code_hostp(CPUArchState *env, target_ulong addr,
|
|
|
|
void **hostp);
|
2018-11-04 00:40:22 +03:00
|
|
|
|
|
|
|
/**
|
2022-08-10 23:52:50 +03:00
|
|
|
* get_page_addr_code()
|
2018-11-04 00:40:22 +03:00
|
|
|
* @env: CPUArchState
|
|
|
|
* @addr: guest virtual address of guest code
|
|
|
|
*
|
2022-08-10 23:52:50 +03:00
|
|
|
* If we cannot translate and execute from the entire RAM page, or if
|
|
|
|
* the region is not backed by RAM, returns -1. Otherwise, returns the
|
|
|
|
* ram_addr_t corresponding to the guest code at @addr.
|
2018-11-04 00:40:22 +03:00
|
|
|
*
|
2022-08-10 23:52:50 +03:00
|
|
|
* Note: this function can trigger an exception.
|
2018-11-04 00:40:22 +03:00
|
|
|
*/
|
2022-08-10 23:52:50 +03:00
|
|
|
static inline tb_page_addr_t get_page_addr_code(CPUArchState *env,
|
|
|
|
target_ulong addr)
|
2018-11-04 00:40:22 +03:00
|
|
|
{
|
2022-08-10 23:52:50 +03:00
|
|
|
return get_page_addr_code_hostp(env, addr, NULL);
|
2018-11-04 00:40:22 +03:00
|
|
|
}
|
2021-08-03 18:31:43 +03:00
|
|
|
|
2022-08-10 23:52:50 +03:00
|
|
|
#if defined(CONFIG_USER_ONLY)
|
2023-01-17 16:52:02 +03:00
|
|
|
void TSA_NO_TSA mmap_lock(void);
|
|
|
|
void TSA_NO_TSA mmap_unlock(void);
|
2022-08-10 23:52:50 +03:00
|
|
|
bool have_mmap_lock(void);
|
|
|
|
|
2021-09-13 05:25:22 +03:00
|
|
|
/**
|
|
|
|
* adjust_signal_pc:
|
|
|
|
* @pc: raw pc from the host signal ucontext_t.
|
|
|
|
* @is_write: host memory operation was write, or read-modify-write.
|
|
|
|
*
|
|
|
|
* Alter @pc as required for unwinding. Return the type of the
|
|
|
|
* guest memory access -- host reads may be for guest execution.
|
|
|
|
*/
|
|
|
|
MMUAccessType adjust_signal_pc(uintptr_t *pc, bool is_write);
|
|
|
|
|
2021-09-13 05:47:29 +03:00
|
|
|
/**
|
|
|
|
* handle_sigsegv_accerr_write:
|
|
|
|
* @cpu: the cpu context
|
|
|
|
* @old_set: the sigset_t from the signal ucontext_t
|
|
|
|
* @host_pc: the host pc, adjusted for the signal
|
|
|
|
* @host_addr: the host address of the fault
|
|
|
|
*
|
|
|
|
* Return true if the write fault has been handled, and should be re-tried.
|
|
|
|
*/
|
|
|
|
bool handle_sigsegv_accerr_write(CPUState *cpu, sigset_t *old_set,
|
|
|
|
uintptr_t host_pc, abi_ptr guest_addr);
|
|
|
|
|
2021-09-18 03:32:56 +03:00
|
|
|
/**
|
|
|
|
* cpu_loop_exit_sigsegv:
|
|
|
|
* @cpu: the cpu context
|
|
|
|
* @addr: the guest address of the fault
|
|
|
|
* @access_type: access was read/write/execute
|
|
|
|
* @maperr: true for invalid page, false for permission fault
|
|
|
|
* @ra: host pc for unwinding
|
|
|
|
*
|
|
|
|
* Use the TCGCPUOps hook to record cpu state, do guest operating system
|
|
|
|
* specific things to raise SIGSEGV, and jump to the main cpu loop.
|
|
|
|
*/
|
2022-04-20 16:26:02 +03:00
|
|
|
G_NORETURN void cpu_loop_exit_sigsegv(CPUState *cpu, target_ulong addr,
|
|
|
|
MMUAccessType access_type,
|
|
|
|
bool maperr, uintptr_t ra);
|
2021-09-18 03:32:56 +03:00
|
|
|
|
2021-10-04 20:06:10 +03:00
|
|
|
/**
|
|
|
|
* cpu_loop_exit_sigbus:
|
|
|
|
* @cpu: the cpu context
|
|
|
|
* @addr: the guest address of the alignment fault
|
|
|
|
* @access_type: access was read/write/execute
|
|
|
|
* @ra: host pc for unwinding
|
|
|
|
*
|
|
|
|
* Use the TCGCPUOps hook to record cpu state, do guest operating system
|
|
|
|
* specific things to raise SIGBUS, and jump to the main cpu loop.
|
|
|
|
*/
|
2022-04-20 16:26:02 +03:00
|
|
|
G_NORETURN void cpu_loop_exit_sigbus(CPUState *cpu, target_ulong addr,
|
|
|
|
MMUAccessType access_type,
|
|
|
|
uintptr_t ra);
|
2021-10-04 20:06:10 +03:00
|
|
|
|
2004-01-04 21:03:10 +03:00
|
|
|
#else
|
2015-08-11 11:57:52 +03:00
|
|
|
static inline void mmap_lock(void) {}
|
|
|
|
static inline void mmap_unlock(void) {}
|
|
|
|
|
2015-09-11 08:39:43 +03:00
|
|
|
void tlb_reset_dirty(CPUState *cpu, ram_addr_t start1, ram_addr_t length);
|
|
|
|
void tlb_set_dirty(CPUState *cpu, target_ulong vaddr);
|
|
|
|
|
|
|
|
MemoryRegionSection *
|
2016-01-21 17:15:05 +03:00
|
|
|
address_space_translate_for_iotlb(CPUState *cpu, int asidx, hwaddr addr,
|
2018-06-15 16:57:16 +03:00
|
|
|
hwaddr *xlat, hwaddr *plen,
|
|
|
|
MemTxAttrs attrs, int *prot);
|
2015-09-11 08:39:43 +03:00
|
|
|
hwaddr memory_region_section_get_iotlb(CPUState *cpu,
|
2019-09-20 07:09:58 +03:00
|
|
|
MemoryRegionSection *section);
|
2004-01-04 21:03:10 +03:00
|
|
|
#endif
|
2005-02-11 01:05:51 +03:00
|
|
|
|
2008-10-23 17:52:00 +04:00
|
|
|
#endif
|