diff --git a/.gitignore b/.gitignore index 6b74118..767ea3b 100644 --- a/.gitignore +++ b/.gitignore @@ -6,5 +6,4 @@ lwip-1.3.0.tar.gz pciutils-2.2.9.tar.bz2 zlib-1.2.3.tar.gz polarssl-1.1.4-gpl.tgz -/mini-os-4.21.0.tar.xz -/xen-4.21.1.tar.xz +/xen-4.11.2.tar.gz diff --git a/CVE-2014-0150.patch b/CVE-2014-0150.patch new file mode 100644 index 0000000..adcbcc7 --- /dev/null +++ b/CVE-2014-0150.patch @@ -0,0 +1,11 @@ +--- xen-4.4.1/tools/qemu-xen-traditional/hw/virtio-net.c.orig 2014-07-02 15:54:37.000000000 +0100 ++++ xen-4.4.1/tools/qemu-xen-traditional/hw/virtio-net.c 2014-11-18 20:50:13.593122915 +0000 +@@ -192,7 +192,7 @@ + return VIRTIO_NET_ERR; + + if (mac_data.entries) { +- if (n->mac_table.in_use + mac_data.entries <= MAC_TABLE_ENTRIES) { ++ if (n->mac_table.in_use <= MAC_TABLE_ENTRIES - mac_data.entries) { + memcpy(n->mac_table.macs + (n->mac_table.in_use * ETH_ALEN), + elem->out_sg[2].iov_base + sizeof(mac_data), + mac_data.entries * ETH_ALEN); diff --git a/qemu.trad.CVE-2015-5278.patch b/qemu.trad.CVE-2015-5278.patch new file mode 100644 index 0000000..950817a --- /dev/null +++ b/qemu.trad.CVE-2015-5278.patch @@ -0,0 +1,11 @@ +--- xen-4.5.1/tools/qemu-xen-traditional/hw/ne2000.c.orig 2015-09-26 17:27:49.494334726 +0100 ++++ xen-4.5.1/tools/qemu-xen-traditional/hw/ne2000.c 2015-09-26 17:31:53.107474932 +0100 +@@ -331,7 +331,7 @@ + if (index <= s->stop) + avail = s->stop - index; + else +- avail = 0; ++ break; + len = size; + if (len > avail) + len = avail; diff --git a/qemu.trad.CVE-2015-5279.patch b/qemu.trad.CVE-2015-5279.patch new file mode 100644 index 0000000..ea08067 --- /dev/null +++ b/qemu.trad.CVE-2015-5279.patch @@ -0,0 +1,48 @@ +--- xen-4.5.1/tools/qemu-xen-traditional/hw/ne2000.c.orig 2015-06-09 16:32:24.000000000 +0100 ++++ xen-4.5.1/tools/qemu-xen-traditional/hw/ne2000.c 2015-09-26 17:27:49.494334726 +0100 +@@ -304,6 +304,9 @@ + } + + index = s->curpag << 8; ++ if (index >= NE2000_PMEM_END) { ++ index = s->start; ++ } + /* 4 bytes for header */ + total_len = size + 4; + /* address for next packet (4 bytes for CRC) */ +@@ -387,15 +390,21 @@ + offset = addr | (page << 4); + switch(offset) { + case EN0_STARTPG: +- s->start = val << 8; ++ if (val << 8 <= NE2000_PMEM_END) { ++ s->start = val << 8; ++ } + s->tainted = 1; + break; + case EN0_STOPPG: +- s->stop = val << 8; ++ if (val << 8 <= NE2000_PMEM_END) { ++ s->stop = val << 8; ++ } + s->tainted = 1; + break; + case EN0_BOUNDARY: +- s->boundary = val; ++ if (val << 8 < NE2000_PMEM_END) { ++ s->boundary = val; ++ } + break; + case EN0_IMR: + s->imr = val; +@@ -436,7 +445,9 @@ + s->phys[offset - EN1_PHYS] = val; + break; + case EN1_CURPAG: +- s->curpag = val; ++ if (val << 8 < NE2000_PMEM_END) { ++ s->curpag = val; ++ } + s->tainted = 1; + break; + case EN1_MULT ... EN1_MULT + 7: diff --git a/qemu.trad.CVE-2015-6815.patch b/qemu.trad.CVE-2015-6815.patch new file mode 100644 index 0000000..7386d6c --- /dev/null +++ b/qemu.trad.CVE-2015-6815.patch @@ -0,0 +1,12 @@ +--- xen-4.5.1/tools/qemu-xen-traditional/hw/e1000.c.orig 2015-06-09 16:32:24.000000000 +0100 ++++ xen-4.5.1/tools/qemu-xen-traditional/hw/e1000.c 2015-09-26 17:16:36.406544380 +0100 +@@ -461,7 +461,8 @@ + memmove(tp->data, tp->header, hdr); + tp->size = hdr; + } +- } while (split_size -= bytes); ++ split_size -= bytes; ++ } while (bytes && split_size); + } else if (!tp->tse && tp->cptse) { + // context descriptor TSE is not set, while data descriptor TSE is set + DBGOUT(TXERR, "TCP segmentaion Error\n"); diff --git a/qemu.trad.CVE-2015-7295.patch b/qemu.trad.CVE-2015-7295.patch new file mode 100644 index 0000000..1c74270 --- /dev/null +++ b/qemu.trad.CVE-2015-7295.patch @@ -0,0 +1,63 @@ +--- xen-4.5.1/tools/qemu-xen-traditional/hw/virtio.c.orig 2015-06-09 16:32:24.000000000 +0100 ++++ xen-4.5.1/tools/qemu-xen-traditional/hw/virtio.c 2015-10-10 16:57:01.806370020 +0100 +@@ -268,8 +268,8 @@ + return vring_avail_idx(vq) == vq->last_avail_idx; + } + +-void virtqueue_fill(VirtQueue *vq, const VirtQueueElement *elem, +- unsigned int len, unsigned int idx) ++static void virtqueue_unmap_sg(VirtQueue *vq, const VirtQueueElement *elem, ++ unsigned int len) + { + unsigned int offset; + int i; +@@ -302,7 +302,19 @@ + + offset += size; + } ++} + ++void virtqueue_discard(VirtQueue *vq, const VirtQueueElement *elem, ++ unsigned int len) ++{ ++ vq->last_avail_idx--; ++ virtqueue_unmap_sg(vq, elem, len); ++} ++ ++void virtqueue_fill(VirtQueue *vq, const VirtQueueElement *elem, ++ unsigned int len, unsigned int idx) ++{ ++ virtqueue_unmap_sg(vq, elem, len); + idx = (idx + vring_used_idx(vq)) % vq->vring.num; + + /* Get a pointer to the next entry in the used ring. */ +--- xen-4.5.1/tools/qemu-xen-traditional/hw/virtio.h.orig 2015-06-09 16:32:24.000000000 +0100 ++++ xen-4.5.1/tools/qemu-xen-traditional/hw/virtio.h 2015-10-10 16:57:53.146216039 +0100 +@@ -105,6 +105,8 @@ + void virtqueue_push(VirtQueue *vq, const VirtQueueElement *elem, + unsigned int len); + void virtqueue_flush(VirtQueue *vq, unsigned int count); ++void virtqueue_discard(VirtQueue *vq, const VirtQueueElement *elem, ++ unsigned int len); + void virtqueue_fill(VirtQueue *vq, const VirtQueueElement *elem, + unsigned int len, unsigned int idx); + +--- xen-4.5.1/tools/qemu-xen-traditional/hw/virtio-net.c.orig 2015-10-10 16:10:05.071786348 +0100 ++++ xen-4.5.1/tools/qemu-xen-traditional/hw/virtio-net.c 2015-10-10 19:05:34.510029916 +0100 +@@ -424,11 +424,15 @@ + len = iov_fill(sg, elem.in_num, + buf + offset, size - offset); + total += len; ++ offset += len; ++ if (!n->mergeable_rx_bufs && offset < size) { ++ virtqueue_discard(n->rx_vq, &elem, total); ++ return; ++ } + + /* signal other side */ + virtqueue_fill(n->rx_vq, &elem, total, i++); + +- offset += len; + } + + if (mhdr) diff --git a/qemu.trad.CVE-2015-7512.patch b/qemu.trad.CVE-2015-7512.patch new file mode 100644 index 0000000..6a1f33f --- /dev/null +++ b/qemu.trad.CVE-2015-7512.patch @@ -0,0 +1,37 @@ +From 8b98a2f07175d46c3f7217639bd5e03f2ec56343 Mon Sep 17 00:00:00 2001 +From: Jason Wang +Date: Mon, 30 Nov 2015 15:00:06 +0800 +Subject: [PATCH] pcnet: fix rx buffer overflow(CVE-2015-7512) + +Backends could provide a packet whose length is greater than buffer +size. Check for this and truncate the packet to avoid rx buffer +overflow in this case. + +Cc: Prasad J Pandit +Cc: qemu-stable@nongnu.org +Reviewed-by: Michael S. Tsirkin +Signed-off-by: Jason Wang +--- + tools/qemu-xen-traditional/hw/pcnet.c | 6 ++++++ + 1 files changed, 6 insertions(+), 0 deletions(-) + +diff --git a/tools/qemu-xen-traditional/hw/pcnet.c b/tools/qemu-xen-traditional/hw/pcnet.c +index 309c40b..1f4a3db 100644 +--- a/tools/qemu-xen-traditional/hw/pcnet.c ++++ b/tools/qemu-xen-traditional/hw/pcnet.c +@@ -1064,6 +1064,12 @@ ssize_t pcnet_receive(NetClientState *nc, const uint8_t *buf, size_t size_) + int pktcount = 0; + + if (!s->looptest) { ++ if (size > 4092) { ++#ifdef PCNET_DEBUG_RMD ++ fprintf(stderr, "pcnet: truncates rx packet.\n"); ++#endif ++ size = 4092; ++ } + memcpy(src, buf, size); + /* no need to compute the CRC */ + src[size] = 0; +-- +1.7.0.4 + diff --git a/qemu.trad.CVE-2015-8345.patch b/qemu.trad.CVE-2015-8345.patch new file mode 100644 index 0000000..73215ca --- /dev/null +++ b/qemu.trad.CVE-2015-8345.patch @@ -0,0 +1,38 @@ +From 00837731d254908a841d69298a4f9f077babaf24 Mon Sep 17 00:00:00 2001 +From: Stefan Weil +Date: Fri, 20 Nov 2015 08:42:33 +0100 +Subject: [PATCH] eepro100: Prevent two endless loops + +http://lists.nongnu.org/archive/html/qemu-devel/2015-11/msg04592.html +shows an example how an endless loop in function action_command can +be achieved. + +During my code review, I noticed a 2nd case which can result in an +endless loop. + +Reported-by: Qinghao Tang +Signed-off-by: Stefan Weil +Signed-off-by: Jason Wang +--- + tools/qemu-xen-traditional/hw/eepro100.c | 16 ++++++++++++++++ + 1 files changed, 16 insertions(+), 0 deletions(-) + +diff --git a/tools/qemu-xen-traditional/hw/eepro100.c b/tools/qemu-xen-traditional/hw/eepro100.c +index 60333b7..685a478 100644 +--- a/tools/qemu-xen-traditional/hw/eepro100.c ++++ b/tools/qemu-xen-traditional/hw/eepro100.c +@@ -774,6 +774,11 @@ static void tx_command(EEPRO100State *s) + uint32_t tx_buffer_address = ldl_phys(tbd_address); + uint16_t tx_buffer_size = lduw_phys(tbd_address + 4); + //~ uint16_t tx_buffer_el = lduw_phys(tbd_address + 6); ++ if (tx_buffer_size == 0) { ++ /* Prevent an endless loop. */ ++ logout("loop in %s:%u\n", __FILE__, __LINE__); ++ break; ++ } + tbd_address += 8; + logout + ("TBD (simplified mode): buffer address 0x%08x, size 0x%04x\n", +-- +1.7.0.4 + diff --git a/qemu.trad.CVE-2015-8504.patch b/qemu.trad.CVE-2015-8504.patch new file mode 100644 index 0000000..3620d40 --- /dev/null +++ b/qemu.trad.CVE-2015-8504.patch @@ -0,0 +1,44 @@ +From 4c65fed8bdf96780735dbdb92a8bd0d6b6526cc3 Mon Sep 17 00:00:00 2001 +From: Prasad J Pandit +Date: Thu, 3 Dec 2015 18:54:17 +0530 +Subject: [PATCH] ui: vnc: avoid floating point exception + +While sending 'SetPixelFormat' messages to a VNC server, +the client could set the 'red-max', 'green-max' and 'blue-max' +values to be zero. This leads to a floating point exception in +write_png_palette while doing frame buffer updates. + +Reported-by: Lian Yihan +Signed-off-by: Prasad J Pandit +Reviewed-by: Gerd Hoffmann +Signed-off-by: Peter Maydell +--- + tools/qemu-xen-traditional/vnc.c | 6 +++--- + 1 files changed, 3 insertions(+), 3 deletions(-) + +diff --git a/tools/qemu-xen-traditional/vnc.c b/tools/qemu-xen-traditional/vnc.c +index 7538405..cbe4d33 100644 +--- a/tools/qemu-xen-traditional/vnc.c ++++ b/tools/qemu-xen-traditional/vnc.c +@@ -2198,15 +2198,15 @@ static void set_pixel_format(VncState *vs, + } + + vs->clientds = vs->serverds; +- vs->clientds.pf.rmax = red_max; ++ vs->clientds.pf.rmax = red_max ? red_max : 0xFF; + count_bits(vs->clientds.pf.rbits, red_max); + vs->clientds.pf.rshift = red_shift; + vs->clientds.pf.rmask = red_max << red_shift; +- vs->clientds.pf.gmax = green_max; ++ vs->clientds.pf.gmax = green_max ? green_max : 0xFF; + count_bits(vs->clientds.pf.gbits, green_max); + vs->clientds.pf.gshift = green_shift; + vs->clientds.pf.gmask = green_max << green_shift; +- vs->clientds.pf.bmax = blue_max; ++ vs->clientds.pf.bmax = blue_max ? blue_max : 0xFF; + count_bits(vs->clientds.pf.bbits, blue_max); + vs->clientds.pf.bshift = blue_shift; + vs->clientds.pf.bmask = blue_max << blue_shift; +-- +1.7.0.4 + diff --git a/qemu.trad.CVE-2016-1714.patch b/qemu.trad.CVE-2016-1714.patch new file mode 100644 index 0000000..59b840b --- /dev/null +++ b/qemu.trad.CVE-2016-1714.patch @@ -0,0 +1,30 @@ +--- xen-4.6.1/tools/qemu-xen-traditional/hw/fw_cfg.c.orig 2016-01-04 15:35:42.000000000 +0000 ++++ xen-4.6.1/tools/qemu-xen-traditional/hw/fw_cfg.c 2016-03-06 16:42:33.464296362 +0000 +@@ -54,11 +54,15 @@ + static void fw_cfg_write(FWCfgState *s, uint8_t value) + { + int arch = !!(s->cur_entry & FW_CFG_ARCH_LOCAL); +- FWCfgEntry *e = &s->entries[arch][s->cur_entry & FW_CFG_ENTRY_MASK]; ++ FWCfgEntry *e = (s->cur_entry == FW_CFG_INVALID) ? NULL : ++ &s->entries[arch][s->cur_entry & FW_CFG_ENTRY_MASK]; + + FW_CFG_DPRINTF("write %d\n", value); + +- if (s->cur_entry & FW_CFG_WRITE_CHANNEL && s->cur_offset < e->len) { ++ if (s->cur_entry & FW_CFG_WRITE_CHANNEL ++ && e != NULL ++ && e->callback ++ && s->cur_offset < e->len) { + e->data[s->cur_offset++] = value; + if (s->cur_offset == e->len) { + e->callback(e->callback_opaque, e->data); +@@ -88,7 +92,8 @@ + static uint8_t fw_cfg_read(FWCfgState *s) + { + int arch = !!(s->cur_entry & FW_CFG_ARCH_LOCAL); +- FWCfgEntry *e = &s->entries[arch][s->cur_entry & FW_CFG_ENTRY_MASK]; ++ FWCfgEntry *e = (s->cur_entry == FW_CFG_INVALID) ? NULL : ++ &s->entries[arch][s->cur_entry & FW_CFG_ENTRY_MASK]; + uint8_t ret; + + if (s->cur_entry == FW_CFG_INVALID || !e->data || s->cur_offset >= e->len) diff --git a/qemu.trad.CVE-2016-1981.patch b/qemu.trad.CVE-2016-1981.patch new file mode 100644 index 0000000..cd2a8c1 --- /dev/null +++ b/qemu.trad.CVE-2016-1981.patch @@ -0,0 +1,104 @@ +------------------------------------------------------------------------ +*From*: Laszlo Ersek +*Subject*: [Qemu-devel] [PATCH] e1000: eliminate infinite loops on +out-of-bounds transfer start +*Date*: Tue, 19 Jan 2016 14:17:20 +0100 + +------------------------------------------------------------------------ + +The start_xmit() and e1000_receive_iov() functions implement DMA transfers +iterating over a set of descriptors that the guest's e1000 driver +prepares: + +- the TDLEN and RDLEN registers store the total size of the descriptor + area, + +- while the TDH and RDH registers store the offset (in whole tx / rx + descriptors) into the area where the transfer is supposed to start. + +Each time a descriptor is processed, the TDH and RDH register is bumped +(as appropriate for the transfer direction). + +QEMU already contains logic to deal with bogus transfers submitted by the +guest: + +- Normally, the transmit case wants to increase TDH from its initial value + to TDT. (TDT is allowed to be numerically smaller than the initial TDH + value; wrapping at or above TDLEN bytes to zero is normal.) The failsafe + that QEMU currently has here is a check against reaching the original + TDH value again -- a complete wraparound, which should never happen. + +- In the receive case RDH is increased from its initial value until + "total_size" bytes have been received; preferably in a single step, or + in "s->rxbuf_size" byte steps, if the latter is smaller. However, null + RX descriptors are skipped without receiving data, while RDH is + incremented just the same. QEMU tries to prevent an infinite loop + (processing only null RX descriptors) by detecting whether RDH assumes + its original value during the loop. (Again, wrapping from RDLEN to 0 is + normal.) + +What both directions miss is that the guest could program TDLEN and RDLEN +so low, and the initial TDH and RDH so high, that these registers will +immediately be truncated to zero, and then never reassume their initial +values in the loop -- a full wraparound will never occur. + +The condition that expresses this is: + + xdh_start >= s->mac_reg[XDLEN] / sizeof(desc) + +i.e., TDH or RDH start out after the last whole rx or tx descriptor that +fits into the TDLEN or RDLEN sized area. + +This condition could be checked before we enter the loops, but +pci_dma_read() / pci_dma_write() knows how to fill in buffers safely for +bogus DMA addresses, so we just extend the existing failsafes with the +above condition. + +Cc: "Michael S. Tsirkin" +Cc: Petr Matousek +Cc: Stefano Stabellini +Cc: Prasad Pandit +Cc: Michael Roth +Cc: Jason Wang +RHBZ: https://bugzilla.redhat.com/show_bug.cgi?id=1296044 +Signed-off-by: Laszlo Ersek +Reviewed-by: Jason Wang +--- + +Notes: + Regarding the public posting: we made an honest effort to vet this + vulnerability, and the impact seems low -- no host side reads/writes, + "just" a DoS (infinite loop). We decided the patch could be posted + publicly, for the usual review process. Jason and Prasad checked the + patch in the internal discussion already, but comments, improvements + etc. are clearly welcome. The CVE request is underway. Thanks. + + hw/net/e1000.c | 6 ++++-- + 1 file changed, 4 insertions(+), 2 deletions(-) + +diff --git a/hw/net/e1000.c b/hw/net/e1000.c +index bec06e9..34d0823 100644 +--- a/tools/qemu-xen-traditional/hw/e1000.c ++++ b/tools/qemu-xen-traditional/hw/e1000.c +@@ -908,7 +908,8 @@ start_xmit(E1000State *s) + * bogus values to TDT/TDLEN. + * there's nothing too intelligent we could do about this. + */ +- if (s->mac_reg[TDH] == tdh_start) { ++ if (s->mac_reg[TDH] == tdh_start || ++ tdh_start >= s->mac_reg[TDLEN] / sizeof(desc)) { + DBGOUT(TXERR, "TDH wraparound @%x, TDT %x, TDLEN %x\n", + tdh_start, s->mac_reg[TDT], s->mac_reg[TDLEN]); + break; +@@ -1165,7 +1166,8 @@ e1000_receive_iov(NetClientState *nc, const struct iovec *iov, int iovcnt) + s->mac_reg[RDH] = 0; + s->check_rxov = 1; + /* see comment in start_xmit; same here */ +- if (s->mac_reg[RDH] == rdh_start) { ++ if (s->mac_reg[RDH] == rdh_start || ++ rdh_start >= s->mac_reg[RDLEN] / sizeof(desc)) { + DBGOUT(RXERR, "RDH wraparound @%x, RDT %x, RDLEN %x\n", + rdh_start, s->mac_reg[RDT], s->mac_reg[RDLEN]); + set_ics(s, 0, E1000_ICS_RXO); +-- +1.8.3.1 diff --git a/qemu.trad.CVE-2016-2538.patch b/qemu.trad.CVE-2016-2538.patch new file mode 100644 index 0000000..be05dd7 --- /dev/null +++ b/qemu.trad.CVE-2016-2538.patch @@ -0,0 +1,56 @@ +From: Prasad J Pandit + +When processing remote NDIS control message packets, +the USB Net device emulator uses a fixed length(4096) data buffer. +The incoming informationBufferOffset & Length combination could +overflow and cross that range. Check control message buffer +offsets and length to avoid it. + +Reported-by: Qinghao Tang +Signed-off-by: Prasad J Pandit +--- + hw/usb/dev-network.c | 9 ++++++--- + 1 file changed, 6 insertions(+), 3 deletions(-) + +Update as per review + -> https://lists.gnu.org/archive/html/qemu-devel/2016-02/msg03475.html + +diff --git a/hw/usb/dev-network.c b/hw/usb/dev-network.c +index 8a4ff49..180adce 100644 +--- a/tools/qemu-xen-traditional/hw/usb-net.c ++++ b/tools/qemu-xen-traditional/hw/usb-net.c +@@ -915,8 +915,9 @@ static int rndis_query_response(USBNetState *s, + + bufoffs = le32_to_cpu(buf->InformationBufferOffset) + 8; + buflen = le32_to_cpu(buf->InformationBufferLength); +- if (bufoffs + buflen > length) ++ if (buflen > length || bufoffs >= length || bufoffs + buflen > length) { + return USB_RET_STALL; ++ } + + infobuflen = ndis_query(s, le32_to_cpu(buf->OID), + bufoffs + (uint8_t *) buf, buflen, infobuf, +@@ -961,8 +962,9 @@ static int rndis_set_response(USBNetState *s, + + bufoffs = le32_to_cpu(buf->InformationBufferOffset) + 8; + buflen = le32_to_cpu(buf->InformationBufferLength); +- if (bufoffs + buflen > length) ++ if (buflen > length || bufoffs >= length || bufoffs + buflen > length) { + return USB_RET_STALL; ++ } + + ret = ndis_set(s, le32_to_cpu(buf->OID), + bufoffs + (uint8_t *) buf, buflen); +@@ -1212,8 +1214,9 @@ static void usb_net_handle_dataout(USBNetState *s, USBPacket *p) + if (le32_to_cpu(msg->MessageType) == RNDIS_PACKET_MSG) { + uint32_t offs = 8 + le32_to_cpu(msg->DataOffset); + uint32_t size = le32_to_cpu(msg->DataLength); +- if (offs + size <= len) ++ if (offs < len && size < len && offs + size <= len) { + qemu_send_packet(s->vc, s->out_buf + offs, size); ++ } + } + s->out_ptr -= len; + memmove(s->out_buf, &s->out_buf[len], s->out_ptr); +-- +2.5.0 diff --git a/qemu.trad.CVE-2016-2841.patch b/qemu.trad.CVE-2016-2841.patch new file mode 100644 index 0000000..6979fbc --- /dev/null +++ b/qemu.trad.CVE-2016-2841.patch @@ -0,0 +1,34 @@ +From: Prasad J Pandit + +Ne2000 NIC uses ring buffer of NE2000_MEM_SIZE(49152) +bytes to process network packets. Registers PSTART & PSTOP +define ring buffer size & location. Setting these registers +to invalid values could lead to infinite loop or OOB r/w +access issues. Add check to avoid it. + +Reported-by: Yang Hongke +Signed-off-by: Prasad J Pandit +--- + hw/net/ne2000.c | 4 ++++ + 1 file changed, 4 insertions(+) + +Update per review: + -> https://lists.gnu.org/archive/html/qemu-devel/2016-02/msg05522.html + +diff --git a/hw/net/ne2000.c b/hw/net/ne2000.c +index b032212..ced4666 100644 +--- a/tools/qemu-xen-traditional/hw/ne2000.c ++++ b/tools/qemu-xen-traditional/hw/ne2000.c +@@ -154,6 +154,10 @@ static int ne2000_buffer_full(NE2000State *s) + { + int avail, index, boundary; + ++ if (s->stop <= s->start) { ++ return 1; ++ } ++ + index = s->curpag << 8; + boundary = s->boundary << 8; + if (index < boundary) +-- +2.5.0 diff --git a/qemu.trad.CVE-2016-2857.patch b/qemu.trad.CVE-2016-2857.patch new file mode 100644 index 0000000..5bef1a7 --- /dev/null +++ b/qemu.trad.CVE-2016-2857.patch @@ -0,0 +1,45 @@ +From: Prasad J Pandit + +While computing IP checksum, 'net_checksum_calculate' reads +payload length from the packet. It could exceed the given 'data' +buffer size. Add a check to avoid it. + +Reported-by: Liu Ling +Signed-off-by: Prasad J Pandit +--- + net/checksum.c | 10 ++++++++-- + 1 file changed, 8 insertions(+), 2 deletions(-) + +Update as per review: + -> https://lists.gnu.org/archive/html/qemu-devel/2016-02/msg06121.html + +diff --git a/net/checksum.c b/net/checksum.c +index 14c0855..0942437 100644 +--- a/tools/qemu-xen-traditional/net-checksum.c ++++ b/tools/qemu-xen-traditional/net-checksum.c +@@ -59,6 +59,11 @@ void net_checksum_calculate(uint8_t *data, int length) + int hlen, plen, proto, csum_offset; + uint16_t csum; + ++ /* Ensure data has complete L2 & L3 headers. */ ++ if (length < 14 + 20) { ++ return; ++ } ++ + if ((data[14] & 0xf0) != 0x40) + return; /* not IPv4 */ + hlen = (data[14] & 0x0f) * 4; +@@ -76,8 +81,9 @@ void net_checksum_calculate(uint8_t *data, int length) + return; + } + +- if (plen < csum_offset+2) +- return; ++ if (plen < csum_offset + 2 || 14 + hlen + plen > length) { ++ return; ++ } + + data[14+hlen+csum_offset] = 0; + data[14+hlen+csum_offset+1] = 0; +-- +2.5.0 diff --git a/qemu.trad.CVE-2016-4001.patch b/qemu.trad.CVE-2016-4001.patch new file mode 100644 index 0000000..9ca362f --- /dev/null +++ b/qemu.trad.CVE-2016-4001.patch @@ -0,0 +1,46 @@ +From 3a15cc0e1ee7168db0782133d2607a6bfa422d66 Mon Sep 17 00:00:00 2001 +From: Prasad J Pandit +Date: Fri, 8 Apr 2016 11:33:48 +0530 +Subject: [PATCH] net: stellaris_enet: check packet length against receive buffer + +When receiving packets over Stellaris ethernet controller, it +uses receive buffer of size 2048 bytes. In case the controller +accepts large(MTU) packets, it could lead to memory corruption. +Add check to avoid it. + +Reported-by: Oleksandr Bazhaniuk +Signed-off-by: Prasad J Pandit +Message-id: 1460095428-22698-1-git-send-email-ppandit@redhat.com +Reviewed-by: Peter Maydell +Signed-off-by: Peter Maydell +--- + tools/qemu-xen-traditional/hw/stellaris_enet.c | 12 +++++++++++- + 1 files changed, 11 insertions(+), 1 deletions(-) + +diff --git a/tools/qemu-xen-traditional/hw/stellaris_enet.c b/tools/qemu-xen-traditional/hw/stellaris_enet.c +index 84cf60b..6880894 100644 +--- a/tools/qemu-xen-traditional/hw/stellaris_enet.c ++++ b/tools/qemu-xen-traditional/hw/stellaris_enet.c +@@ -236,8 +236,18 @@ static ssize_t stellaris_enet_receive(NetClientState *nc, const uint8_t *buf, si + n = s->next_packet + s->np; + if (n >= 31) + n -= 31; +- s->np++; + ++ if (size >= sizeof(s->rx[n].data) - 6) { ++ /* If the packet won't fit into the ++ * emulated 2K RAM, this is reported ++ * as a FIFO overrun error. ++ */ ++ s->ris |= SE_INT_FOV; ++ stellaris_enet_update(s); ++ return -1; ++ } ++ ++ s->np++; + s->rx[n].len = size + 6; + p = s->rx[n].data; + *(p++) = (size + 6); +-- +1.7.0.4 + diff --git a/qemu.trad.CVE-2016-4002.patch b/qemu.trad.CVE-2016-4002.patch new file mode 100644 index 0000000..e122297 --- /dev/null +++ b/qemu.trad.CVE-2016-4002.patch @@ -0,0 +1,31 @@ +From: Prasad J Pandit + +When receiving packets over MIPSnet network device, it uses + receive buffer of size 1514 bytes. In case the controller +accepts large(MTU) packets, it could lead to memory corruption. +Add check to avoid it. + +Reported by: Oleksandr Bazhaniuk + +Signed-off-by: Prasad J Pandit +--- + tools/qemu-xen-traditional/hw/mipsnet.c | 3 +++ + 1 file changed, 3 insertions(+) + +diff --git a/tools/qemu-xen-traditional/hw/mipsnet.c b/tools/qemu-xen-traditional/hw/mipsnet.c +index f261011..e134b31 100644 +--- a/tools/qemu-xen-traditional/hw/mipsnet.c ++++ b/tools/qemu-xen-traditional/hw/mipsnet.c +@@ -82,6 +82,9 @@ static ssize_t mipsnet_receive(NetClientState *nc, const uint8_t *buf, size_t si + if (!mipsnet_can_receive(opaque)) + return; + ++ if (size >= sizeof(s->rx_buffer)) { ++ return; ++ } + s->busy = 1; + + /* Just accept everything. */ +-- +2.5.5 + diff --git a/qemu.trad.CVE-2016-4439.patch b/qemu.trad.CVE-2016-4439.patch new file mode 100644 index 0000000..6816695 --- /dev/null +++ b/qemu.trad.CVE-2016-4439.patch @@ -0,0 +1,44 @@ +------------------------------------------------------------------------ +*From*: P J P +*Subject*: [Qemu-devel] [PATCH 1/2] scsi: check command buffer length +before write(CVE-2016-4439) +*Date*: Thu, 19 May 2016 16:09:30 +0530 + +------------------------------------------------------------------------ + +From: Prasad J Pandit + +The 53C9X Fast SCSI Controller(FSC) comes with an internal 16-byte +FIFO buffer. It is used to handle command and data transfer. While +writing to this command buffer 's->cmdbuf[TI_BUFSZ=16]', a check +was missing to validate input length. Add check to avoid OOB write +access. + +Fixes CVE-2016-4439 +Reported-by: Li Qiang + +Signed-off-by: Prasad J Pandit +--- + hw/scsi/esp.c | 6 +++++- + 1 file changed, 5 insertions(+), 1 deletion(-) + +diff --git a/tools/qemu-xen-traditional/hw/esp.c b/tools/qemu-xen-traditional/hw/esp.c +index 8961be2..01497e6 100644 +--- a/tools/qemu-xen-traditional/hw/esp.c ++++ b/tools/qemu-xen-traditional/hw/esp.c +@@ -448,7 +448,11 @@ void esp_reg_write(ESPState *s, uint32_t saddr, uint64_t val) + break; + case ESP_FIFO: + if (s->do_cmd) { +- s->cmdbuf[s->cmdlen++] = val & 0xff; ++ if (s->cmdlen < TI_BUFSZ) { ++ s->cmdbuf[s->cmdlen++] = val & 0xff; ++ } else { ++ ESP_ERROR("fifo overrun\n"); ++ } + } else if (s->ti_size == TI_BUFSZ - 1) { + ESP_ERROR("fifo overrun\n"); + } else { +-- +2.5.5 + diff --git a/qemu.trad.CVE-2016-4441.patch b/qemu.trad.CVE-2016-4441.patch new file mode 100644 index 0000000..fab6a35 --- /dev/null +++ b/qemu.trad.CVE-2016-4441.patch @@ -0,0 +1,68 @@ +------------------------------------------------------------------------ +*From*: P J P +*Subject*: [Qemu-devel] [PATCH 2/2] scsi: check dma length before +reading scsi command(CVE-2016-4441) +*Date*: Thu, 19 May 2016 16:09:31 +0530 + +------------------------------------------------------------------------ + +From: Prasad J Pandit + +The 53C9X Fast SCSI Controller(FSC) comes with an internal 16-byte +FIFO buffer. It is used to handle command and data transfer. +Routine get_cmd() uses DMA to read scsi commands into this buffer. +Add check to validate DMA length against buffer size to avoid any +overrun. + +Fixes CVE-2016-4441 +Reported-by: Li Qiang + +Signed-off-by: Prasad J Pandit +--- + hw/scsi/esp.c | 11 +++++++---- + 1 file changed, 7 insertions(+), 4 deletions(-) + +diff --git a/tools/qemu-xen-traditional/hw/esp.c b/tools/qemu-xen-traditional/hw/esp.c +index 01497e6..591c817 100644 +--- a/tools/qemu-xen-traditional/hw/esp.c ++++ b/tools/qemu-xen-traditional/hw/esp.c +@@ -82,7 +82,7 @@ void esp_request_cancelled(SCSIRequest *req) + } + } + +-static uint32_t get_cmd(ESPState *s, uint8_t *buf) ++static uint32_t get_cmd(ESPState *s, uint8_t *buf, uint8_t buflen) + { + uint32_t dmalen; + int target; +@@ -92,6 +92,9 @@ static uint32_t get_cmd(ESPState *s, uint8_t *buf) + target = s->wregs[ESP_WBUSID] & BUSID_DID; + if (s->dma) { + dmalen = s->rregs[ESP_TCLO] | (s->rregs[ESP_TCMID] << 8); ++ if (dmalen > buflen) { ++ return 0; ++ } + s->dma_memory_read(s->dma_opaque, buf, dmalen); + } else { + dmalen = s->ti_size; +@@ -166,7 +169,7 @@ static void handle_satn(ESPState *s) + uint8_t buf[32]; + int len; + +- len = get_cmd(s, buf); ++ len = get_cmd(s, buf, sizeof(buf)); + if (len) + do_cmd(s, buf); + } +@@ -192,7 +195,7 @@ static void handle_satn_stop(ESPState *s) + + static void handle_satn_stop(ESPState *s) + { +- s->cmdlen = get_cmd(s, s->cmdbuf); ++ s->cmdlen = get_cmd(s, s->cmdbuf, sizeof(s->cmdbuf)); + if (s->cmdlen) { + DPRINTF("Set ATN & Stop: cmdlen %d\n", s->cmdlen); + s->do_cmd = 1; +-- +2.5.5 + diff --git a/qemu.trad.CVE-2016-5238.patch b/qemu.trad.CVE-2016-5238.patch new file mode 100644 index 0000000..f6767de --- /dev/null +++ b/qemu.trad.CVE-2016-5238.patch @@ -0,0 +1,65 @@ +------------------------------------------------------------------------ +*From*: Paolo Bonzini +*Subject*: Re: [Qemu-devel] [PATCH] scsi: check buffer length before +reading scsi command +*Date*: Wed, 1 Jun 2016 15:10:16 +0200 +*User-agent*: Mozilla/5.0 (X11; Linux x86_64; rv:45.0) Gecko/20100101 +Thunderbird/45.1.0 + +------------------------------------------------------------------------ + + +On 31/05/2016 19:53, P J P wrote: +>/ From: Prasad J Pandit / +>/ / +>/ The 53C9X Fast SCSI Controller(FSC) comes with an internal 16-byte/ +>/ FIFO buffer. It is used to handle command and data transfer./ +>/ Routine get_cmd() in non-DMA mode, uses 'ti_size' to read scsi/ +>/ command into a buffer. Add check to validate command length against/ +>/ buffer size to avoid any overrun./ +>/ / +>/ Reported-by: Li Qiang / +>/ Signed-off-by: Prasad J Pandit / +>/ ---/ +>/ hw/scsi/esp.c | 3 +++/ +>/ 1 file changed, 3 insertions(+)/ +>/ / +>/ diff --git a/tools/qemu-xen-traditional/hw/esp.c b/tools/qemu-xen-traditional/hw/esp.c/ +>/ index 60c1b28..953027a 100644/ +>/ --- a/tools/qemu-xen-traditional/hw/esp.c/ +>/ +++ b/tools/qemu-xen-traditional/hw/esp.c/ +>/ @@ -98,6 +98,9 @@ static uint32_t get_cmd(ESPState *s, uint8_t *buf, uint8_t / +>/ buflen)/ +>/ s->dma_memory_read(s->dma_opaque, buf, dmalen);/ +>/ } else {/ +>/ dmalen = s->ti_size;/ +>/ + if (dmalen > TI_BUFSZ) {/ +>/ + return 0;/ +>/ + }/ +>/ memcpy(buf, s->ti_buf, dmalen);/ +>/ buf[0] = buf[2] >> 5;/ +>/ }/ +>/ / + +In theory this shouldn't happen, but I agree that it is better to be +defensive. I'm queuing this patch. + +At least the following patch is needed to ensure that ti_size always +matches ti_rptr/ti_wptr (Hervé, what do you think about it? should I +resubmit it formally?). Also, things are more complicated than +necessary due to ti_size being used for both DMA and FIFO transfers. + +diff --git a/tools/qemu-xen-traditional/hw/esp.c b/tools/qemu-xen-traditional/hw/esp.c +index c2f6f8f..6407844 100644 +--- a/tools/qemu-xen-traditional/hw/esp.c ++++ b/tools/qemu-xen-traditional/hw/esp.c +@@ -222,7 +222,7 @@ static void write_response(ESPState *s) + } else { + s->ti_size = 2; + s->ti_rptr = 0; +- s->ti_wptr = 0; ++ s->ti_wptr = 2; + s->rregs[ESP_RFLAGS] = 2; + } + esp_raise_irq(s); + diff --git a/qemu.trad.CVE-2016-5338.patch b/qemu.trad.CVE-2016-5338.patch new file mode 100644 index 0000000..be36dca --- /dev/null +++ b/qemu.trad.CVE-2016-5338.patch @@ -0,0 +1,76 @@ +------------------------------------------------------------------------ +*From*: P J P +*Subject*: [Qemu-devel] [PATCH v3] scsi: esp: check TI buffer index +before read/write +*Date*: Mon, 6 Jun 2016 22:04:43 +0530 + +------------------------------------------------------------------------ + +From: Prasad J Pandit + +The 53C9X Fast SCSI Controller(FSC) comes with internal 16-byte +FIFO buffers. One is used to handle commands and other is for +information transfer. Three control variables 'ti_rptr', +'ti_wptr' and 'ti_size' are used to control r/w access to the +information transfer buffer ti_buf[TI_BUFSZ=16]. In that, + +'ti_rptr' is used as read index, where read occurs. +'ti_wptr' is a write index, where write would occur. +'ti_size' indicates total bytes to be read from the buffer. + +While reading/writing to this buffer, index could exceed its +size. Add check to avoid OOB r/w access. + +Reported-by: Huawei PSIRT +Reported-by: Li Qiang +Signed-off-by: Prasad J Pandit +--- + hw/scsi/esp.c | 20 +++++++++----------- + 1 file changed, 9 insertions(+), 11 deletions(-) + +Update as per: + -> https://lists.gnu.org/archive/html/qemu-devel/2016-06/msg01326.html + +diff --git a/tools/qemu-xen-traditional/hw/esp.c b/tools/qemu-xen-traditional/hw/esp.c +index c2f6f8f..4b94bbc 100644 +--- a/tools/qemu-xen-traditional/hw/esp.c ++++ b/tools/qemu-xen-traditional/hw/esp.c +@@ -403,18 +403,17 @@ uint64_t esp_reg_read(ESPState *s, uint32_t saddr) + DPRINTF("read reg[%d]: 0x%2.2x\n", saddr, s->rregs[saddr]); + switch (saddr) { + case ESP_FIFO: +- if (s->ti_size > 0) { ++ if ((s->rregs[ESP_RSTAT] & STAT_PIO_MASK) == 0) { ++ /* Data out. */ ++ ESP_ERROR("PIO data read not implemented\n"); ++ s->rregs[ESP_FIFO] = 0; ++ esp_raise_irq(s); ++ } else if (s->ti_rptr < s->ti_wptr) { + s->ti_size--; +- if ((s->rregs[ESP_RSTAT] & STAT_PIO_MASK) == 0) { +- /* Data out. */ +- ESP_ERROR("PIO data read not implemented\n"); +- s->rregs[ESP_FIFO] = 0; +- } else { +- s->rregs[ESP_FIFO] = s->ti_buf[s->ti_rptr++]; +- } ++ s->rregs[ESP_FIFO] = s->ti_buf[s->ti_rptr++]; + esp_raise_irq(s); + } +- if (s->ti_size == 0) { ++ if (s->ti_rptr == s->ti_wptr) { + s->ti_rptr = 0; + s->ti_wptr = 0; + } +@@ -459,7 +457,7 @@ void esp_reg_write(ESPState *s, uint32_t saddr, uint64_t val) + } else { + ESP_ERROR("fifo overrun\n"); + } +- } else if (s->ti_size == TI_BUFSZ - 1) { ++ } else if (s->ti_wptr == TI_BUFSZ - 1) { + ESP_ERROR("fifo overrun\n"); + } else { + s->ti_size++; +-- +2.5.5 + diff --git a/qemu.trad.CVE-2016-6351.patch b/qemu.trad.CVE-2016-6351.patch new file mode 100644 index 0000000..10f1ab3 --- /dev/null +++ b/qemu.trad.CVE-2016-6351.patch @@ -0,0 +1,81 @@ +From 926cde5f3e4d2504ed161ed0cb771ac7cad6fd11 Mon Sep 17 00:00:00 2001 +From: Prasad J Pandit +Date: Thu, 16 Jun 2016 00:22:35 +0200 +Subject: [PATCH] scsi: esp: make cmdbuf big enough for maximum CDB size + +While doing DMA read into ESP command buffer 's->cmdbuf', it could +write past the 's->cmdbuf' area, if it was transferring more than 16 +bytes. Increase the command buffer size to 32, which is maximum when +'s->do_cmd' is set, and add a check on 'len' to avoid OOB access. + +Reported-by: Li Qiang +Signed-off-by: Prasad J Pandit +Signed-off-by: Paolo Bonzini +--- + hw/esp.c | 6 ++++-- + hw/esp.c | 3 ++- + 2 files changed, 6 insertions(+), 3 deletions(-) + +diff --git a/hw/esp.c b/hw/esp.c +index 64680b3..baa0a2c 100644 +--- a/hw/esp.c ++++ b/hw/esp.c +@@ -25,6 +25,7 @@ + #include "hw.h" + #include "scsi-disk.h" + #include "scsi.h" ++#include + + /* debug ESP card */ + //#define DEBUG_ESP +@@ -248,6 +248,8 @@ static void esp_do_dma(ESPState *s) + len = s->dma_left; + if (s->do_cmd) { + DPRINTF("command len %d + %d\n", s->cmdlen, len); ++ assert (s->cmdlen <= sizeof(s->cmdbuf) && ++ len <= sizeof(s->cmdbuf) - s->cmdlen); + s->dma_memory_read(s->dma_opaque, &s->cmdbuf[s->cmdlen], len); + s->ti_size = 0; + s->cmdlen = 0; +@@ -345,7 +347,7 @@ static void handle_ti(ESPState *s) + s->dma_counter = dmalen; + + if (s->do_cmd) +- minlen = (dmalen < 32) ? dmalen : 32; ++ minlen = (dmalen < ESP_CMDBUF_SZ) ? dmalen : ESP_CMDBUF_SZ; + else if (s->ti_size < 0) + minlen = (dmalen < -s->ti_size) ? dmalen : -s->ti_size; + else +@@ -449,7 +451,7 @@ void esp_reg_write(ESPState *s, uint32_t saddr, uint64_t val) + break; + case ESP_FIFO: + if (s->do_cmd) { +- if (s->cmdlen < TI_BUFSZ) { ++ if (s->cmdlen < ESP_CMDBUF_SZ) { + s->cmdbuf[s->cmdlen++] = val & 0xff; + } else { + ESP_ERROR("fifo overrun\n"); +diff --git a/hw/esp.c b/hw/esp.c +index 6c79527..d2c4886 100644 +--- a/hw/esp.c ++++ b/hw/esp.c +@@ -14,6 +14,7 @@ void esp_init(hwaddr espaddr, int it_shift, + + #define ESP_REGS 16 + #define TI_BUFSZ 16 ++#define ESP_CMDBUF_SZ 32 + + typedef struct ESPState ESPState; + +@@ -31,7 +32,7 @@ struct ESPState { + uint32_t dma; + SCSIDevice *scsi_dev[ESP_MAX_DEVS]; + SCSIDevice *current_dev; +- uint8_t cmdbuf[TI_BUFSZ]; ++ uint8_t cmdbuf[ESP_CMDBUF_SZ]; + uint32_t cmdlen; + uint32_t do_cmd; + +-- +1.7.0.4 + diff --git a/qemu.trad.CVE-2016-8669.patch b/qemu.trad.CVE-2016-8669.patch new file mode 100644 index 0000000..05abe36 --- /dev/null +++ b/qemu.trad.CVE-2016-8669.patch @@ -0,0 +1,37 @@ +From 3592fe0c919cf27a81d8e9f9b4f269553418bb01 Mon Sep 17 00:00:00 2001 +From: Prasad J Pandit +Date: Wed, 12 Oct 2016 11:28:08 +0530 +Subject: [PATCH] char: serial: check divider value against baud base + +16550A UART device uses an oscillator to generate frequencies +(baud base), which decide communication speed. This speed could +be changed by dividing it by a divider. If the divider is +greater than the baud base, speed is set to zero, leading to a +divide by zero error. Add check to avoid it. + +Reported-by: Huawei PSIRT +Signed-off-by: Prasad J Pandit +Message-Id: <1476251888-20238-1-git-send-email-ppandit@redhat.com> +Signed-off-by: Paolo Bonzini +--- + hw/char/serial.c | 3 ++- + 1 files changed, 2 insertions(+), 1 deletions(-) + +diff --git a/hw/serial.c b/hw/serial.c +index 3442f47..eec72b7 100644 +--- a/hw/serial.c ++++ b/hw/serial.c +@@ -153,8 +153,9 @@ static void serial_update_parameters(SerialState *s) + int speed, parity, data_bits, stop_bits, frame_size; + QEMUSerialSetParams ssp; + +- if (s->divider == 0) ++ if (s->divider == 0 || s->divider > s->baudbase) { + return; ++ } + + frame_size = 1; + if (s->lcr & 0x08) { +-- +1.7.0.4 + diff --git a/qemu.trad.CVE-2016-8910.patch b/qemu.trad.CVE-2016-8910.patch new file mode 100644 index 0000000..ddb67b1 --- /dev/null +++ b/qemu.trad.CVE-2016-8910.patch @@ -0,0 +1,29 @@ +From: Prasad J Pandit + +RTL8139 ethernet controller in C+ mode supports multiple +descriptor rings, each with maximum of 64 descriptors. While +processing transmit descriptor ring in 'rtl8139_cplus_transmit', +it does not limit the descriptor count and runs forever. Add +check to avoid it. + +Reported-by: Andrew Henderson +Signed-off-by: Prasad J Pandit +--- + hw/net/rtl8139.c | 2 +- + 1 file changed, 1 insertion(+), 1 deletion(-) + +diff --git a/hw/rtl8139.c b/hw/rtl8139.c +index 3345bc6..f05e59c 100644 +--- a/hw/rtl8139.c ++++ b/hw/rtl8139.c +@@ -2350,7 +2350,7 @@ static void rtl8139_cplus_transmit(RTL8139State *s) + { + int txcount = 0; + +- while (rtl8139_cplus_transmit_one(s)) ++ while (txcount < 64 && rtl8139_cplus_transmit_one(s)) + { + ++txcount; + } +-- +2.7.4 diff --git a/qemu.trad.CVE-2016-9776.patch b/qemu.trad.CVE-2016-9776.patch new file mode 100644 index 0000000..2098ed3 --- /dev/null +++ b/qemu.trad.CVE-2016-9776.patch @@ -0,0 +1,34 @@ +From 77d54985b85a0cb760330ec2bd92505e0a2a97a9 Mon Sep 17 00:00:00 2001 +From: Prasad J Pandit +Date: Tue, 29 Nov 2016 00:38:39 +0530 +Subject: [PATCH] net: mcf: check receive buffer size register value + +ColdFire Fast Ethernet Controller uses a receive buffer size +register(EMRBR) to hold maximum size of all receive buffers. +It is set by a user before any operation. If it was set to be +zero, ColdFire emulator would go into an infinite loop while +receiving data in mcf_fec_receive. Add check to avoid it. + +Reported-by: Wjjzhang +Signed-off-by: Prasad J Pandit +Signed-off-by: Jason Wang +--- + hw/net/mcf_fec.c | 2 +- + 1 files changed, 1 insertions(+), 1 deletions(-) + +diff --git a/hw/mcf_fec.c b/hw/mcf_fec.c +index dc61bac..4025eb3 100644 +--- a/hw/mcf_fec.c ++++ b/hw/mcf_fec.c +@@ -393,7 +393,7 @@ static void mcf_fec_write(void *opaque, hwaddr addr, + s->tx_descriptor = s->etdsr; + break; + case 0x188: +- s->emrbr = value & 0x7f0; ++ s->emrbr = value > 0 ? value & 0x7F0 : 0x7F0; + break; + default: + cpu_abort(cpu_single_env, "mcf_fec_write Bad address 0x%x\n", +-- +1.7.0.4 + diff --git a/qemu.trad.CVE-2017-6505.patch b/qemu.trad.CVE-2017-6505.patch new file mode 100644 index 0000000..b374a3d --- /dev/null +++ b/qemu.trad.CVE-2017-6505.patch @@ -0,0 +1,51 @@ +From 95ed56939eb2eaa4e2f349fe6dcd13ca4edfd8fb Mon Sep 17 00:00:00 2001 +From: Li Qiang +Date: Tue, 7 Feb 2017 02:23:33 -0800 +Subject: [PATCH] usb: ohci: limit the number of link eds + +The guest may builds an infinite loop with link eds. This patch +limit the number of linked ed to avoid this. + +Signed-off-by: Li Qiang +Message-id: 5899a02e.45ca240a.6c373.93c1@mx.google.com +Signed-off-by: Gerd Hoffmann +--- + hw/usb-ohci.c | 9 ++++++++- + 1 file changed, 8 insertions(+), 1 deletion(-) + +diff --git a/hw/usb-ohci.c b/hw/usb-ohci.c +index 2cba3e3..21c93e0 100644 +--- a/hw/usb-ohci.c ++++ b/hw/usb-ohci.c +@@ -42,6 +42,8 @@ + + #define OHCI_MAX_PORTS 15 + ++#define ED_LINK_LIMIT 4 ++ + static int64_t usb_frame_time; + static int64_t usb_bit_time; + +@@ -1184,7 +1186,7 @@ static int ohci_service_ed_list(OHCIState *ohci, uint32_t head, int completion) + uint32_t next_ed; + uint32_t cur; + int active; +- ++ uint32_t link_cnt = 0; + active = 0; + + if (head == 0) +@@ -1199,6 +1201,10 @@ static int ohci_service_ed_list(OHCIState *ohci, uint32_t head, int completion) + + next_ed = ed.next & OHCI_DPTR_MASK; + ++ if (++link_cnt > ED_LINK_LIMIT) { ++ return 0; ++ } ++ + if ((ed.head & OHCI_ED_H) || (ed.flags & OHCI_ED_K)) { + uint32_t addr; + /* Cancel pending packets for ED that have been paused. */ +-- +1.8.3.1 + diff --git a/qemu.trad.CVE-2017-7718.patch b/qemu.trad.CVE-2017-7718.patch new file mode 100644 index 0000000..70382ab --- /dev/null +++ b/qemu.trad.CVE-2017-7718.patch @@ -0,0 +1,51 @@ +From 215902d7b6fb50c6fc216fc74f770858278ed904 Mon Sep 17 00:00:00 2001 +From: hangaohuai +Date: Tue, 14 Mar 2017 14:39:19 +0800 +Subject: [PATCH] fix :cirrus_vga fix OOB read case qemu Segmentation fault + +check the validity of parameters in cirrus_bitblt_rop_fwd_transp_xxx +and cirrus_bitblt_rop_fwd_xxx to avoid the OOB read which causes qemu Segmentation fault. + +After the fix, we will touch the assert in +cirrus_invalidate_region: +assert(off_cur_end >= off_cur); + +Signed-off-by: fangying +Signed-off-by: hangaohuai +Message-id: 20170314063919.16200-1-hangaohuai@huawei.com +Signed-off-by: Gerd Hoffmann +--- + hw/cirrus_vga_rop.h | 10 ++++++++++ + 1 file changed, 10 insertions(+) + +diff --git a/hw/cirrus_vga_rop.h b/hw/cirrus_vga_rop.h +index 0925a00..b7447f8 100644 +--- a/hw/cirrus_vga_rop.h ++++ b/hw/cirrus_vga_rop.h +@@ -97,6 +97,11 @@ glue(glue(cirrus_bitblt_rop_fwd_transp_, ROP_NAME),_8)(CirrusVGAState *s, + src = src_ - src_base; + dstpitch -= bltwidth; + srcpitch -= bltwidth; ++ ++ if (bltheight > 1 && (dstpitch < 0 || srcpitch < 0)) { ++ return; ++ } ++ + for (y = 0; y < bltheight; y++) { + for (x = 0; x < bltwidth; x++) { + p = *(dst_base + m(dst)); +@@ -143,6 +148,11 @@ glue(glue(cirrus_bitblt_rop_fwd_transp_, ROP_NAME),_16)(CirrusVGAState *s, + src = src_ - src_base; + dstpitch -= bltwidth; + srcpitch -= bltwidth; ++ ++ if (bltheight > 1 && (dstpitch < 0 || srcpitch < 0)) { ++ return; ++ } ++ + for (y = 0; y < bltheight; y++) { + for (x = 0; x < bltwidth; x+=2) { + p1 = *(dst_base + m(dst)); +-- +1.8.3.1 + diff --git a/qemu.trad.CVE-2017-8309.patch b/qemu.trad.CVE-2017-8309.patch new file mode 100644 index 0000000..10b5b05 --- /dev/null +++ b/qemu.trad.CVE-2017-8309.patch @@ -0,0 +1,38 @@ +From 3268a845f41253fb55852a8429c32b50f36f349a Mon Sep 17 00:00:00 2001 +From: Gerd Hoffmann +Date: Fri, 28 Apr 2017 09:56:12 +0200 +Subject: [PATCH] audio: release capture buffers + +AUD_add_capture() allocates two buffers which are never released. +Add the missing calls to AUD_del_capture(). + +Impact: Allows vnc clients to exhaust host memory by repeatedly +starting and stopping audio capture. + +Fixes: CVE-2017-8309 +Cc: P J P +Cc: Huawei PSIRT +Reported-by: "Jiangxin (hunter, SCC)" +Signed-off-by: Gerd Hoffmann +Reviewed-by: Prasad J Pandit +Message-id: 20170428075612.9997-1-kraxel@redhat.com +--- + audio/audio.c | 2 ++ + 1 file changed, 2 insertions(+) + +diff --git a/audio/audio.c b/audio/audio.c +index c8898d8..beafed2 100644 +--- a/audio/audio.c ++++ b/audio/audio.c +@@ -2028,6 +2028,8 @@ void AUD_del_capture (CaptureVoiceOut *cap, void *cb_opaque) + sw = sw1; + } + LIST_REMOVE (cap, entries); ++ qemu_free (cap->hw.mix_buf); ++ qemu_free (cap->buf); + qemu_free (cap); + } + return; +-- +1.8.3.1 + diff --git a/qemu.trad.CVE-2017-9330.patch b/qemu.trad.CVE-2017-9330.patch new file mode 100644 index 0000000..046e3e0 --- /dev/null +++ b/qemu.trad.CVE-2017-9330.patch @@ -0,0 +1,31 @@ +From 26f670a244982335cc08943fb1ec099a2c81e42d Mon Sep 17 00:00:00 2001 +From: Li Qiang +Date: Tue, 7 Feb 2017 03:15:03 -0800 +Subject: [PATCH] usb: ohci: fix error return code in servicing iso td + +It should return 1 if an error occurs when reading iso td. +This will avoid an infinite loop issue in ohci_service_ed_list. + +Signed-off-by: Li Qiang +Message-id: 5899ac3e.1033240a.944d5.9a2d@mx.google.com +Signed-off-by: Gerd Hoffmann +--- + hw/usb-ohci.c | 2 +- + 1 file changed, 1 insertion(+), 1 deletion(-) + +diff --git a/hw/usb-ohci.c b/hw/usb-ohci.c +index c82a92f..2cba3e3 100644 +--- a/hw/usb-ohci.c ++++ b/hw/usb-ohci.c +@@ -725,7 +725,7 @@ static int ohci_service_iso_td(OHCIState *ohci, struct ohci_ed *ed, + + if (!ohci_read_iso_td(addr, &iso_td)) { + printf("usb-ohci: ISO_TD read error at %x\n", addr); +- return 0; ++ return 1; + } + + starting_frame = OHCI_BM(iso_td.flags, TD_SF); +-- +1.8.3.1 + diff --git a/qemu.trad.bug1399055.patch b/qemu.trad.bug1399055.patch new file mode 100644 index 0000000..69f8fd6 --- /dev/null +++ b/qemu.trad.bug1399055.patch @@ -0,0 +1,76 @@ +From 4299b90e9ba9ce5ca9024572804ba751aa1a7e70 Mon Sep 17 00:00:00 2001 +From: Prasad J Pandit +Date: Tue, 18 Oct 2016 13:15:17 +0530 +Subject: [PATCH] display: cirrus: check vga bits per pixel(bpp) value + +In Cirrus CLGD 54xx VGA Emulator, if cirrus graphics mode is VGA, +'cirrus_get_bpp' returns zero(0), which could lead to a divide +by zero error in while copying pixel data. The same could occur +via blit pitch values. Add check to avoid it. + +Reported-by: Huawei PSIRT +Signed-off-by: Prasad J Pandit +Message-id: 1476776717-24807-1-git-send-email-ppandit@redhat.com +Signed-off-by: Gerd Hoffmann +--- + hw/cirrus_vga.c | 14 ++++++++++---- + 1 files changed, 10 insertions(+), 4 deletions(-) + +diff --git a/hw/cirrus_vga.c b/hw/cirrus_vga.c +index 3d712d5..bdb092e 100644 +--- a/hw/cirrus_vga.c ++++ b/hw/cirrus_vga.c +@@ -272,6 +272,9 @@ static void cirrus_update_memory_access(CirrusVGAState *s); + static bool blit_region_is_unsafe(struct CirrusVGAState *s, + int32_t pitch, int32_t addr) + { ++ if (!pitch) { ++ return true; ++ } + if (pitch < 0) { + int64_t min = addr + + ((int64_t)s->cirrus_blt_height - 1) * pitch +@@ -715,7 +718,7 @@ static int cirrus_bitblt_videotovideo_patterncopy(CirrusVGAState * s) + s->cirrus_addr_mask)); + } + +-static void cirrus_do_copy(CirrusVGAState *s, int dst, int src, int w, int h) ++static int cirrus_do_copy(CirrusVGAState *s, int dst, int src, int w, int h) + { + int sx = 0, sy = 0; + int dx = 0, dy = 0; +@@ -729,6 +732,9 @@ static void cirrus_do_copy(CirrusVGAState *s, int dst, int src, int w, int h) + int width, height; + + depth = s->get_bpp((VGAState *)s) / 8; ++ if (!depth) { ++ return 0; ++ } + s->get_resolution((VGAState *)s, &width, &height); + + /* extra x, y */ +@@ -783,6 +789,8 @@ static void cirrus_do_copy(CirrusVGAState *s, int dst, int src, int w, int h) + cirrus_invalidate_region(s, s->cirrus_blt_dstaddr, + s->cirrus_blt_dstpitch, s->cirrus_blt_width, + s->cirrus_blt_height); ++ ++ return 1; + } + + static int cirrus_bitblt_videotovideo_copy(CirrusVGAState * s) +@@ -790,11 +798,9 @@ static int cirrus_bitblt_videotovideo_copy(CirrusVGAState * s) + if (blit_is_unsafe(s)) + return 0; + +- cirrus_do_copy(s, s->cirrus_blt_dstaddr - s->start_addr, ++ return cirrus_do_copy(s, s->cirrus_blt_dstaddr - s->start_addr, + s->cirrus_blt_srcaddr - s->start_addr, + s->cirrus_blt_width, s->cirrus_blt_height); +- +- return 1; + } + + /*************************************** +-- +1.7.0.4 + diff --git a/sources b/sources index 4b30fea..ef56fa4 100644 --- a/sources +++ b/sources @@ -4,5 +4,4 @@ SHA512 (newlib-1.16.0.tar.gz) = 40eb96bbc6736a16b6399e0cdb73e853d0d90b685c967e77 SHA512 (zlib-1.2.3.tar.gz) = 021b958fcd0d346c4ba761bcf0cc40f3522de6186cf5a0a6ea34a70504ce9622b1c2626fce40675bc8282cf5f5ade18473656abc38050f72f5d6480507a2106e SHA512 (polarssl-1.1.4-gpl.tgz) = 88da614e4d3f4409c4fd3bb3e44c7587ba051e3fed4e33d526069a67e8180212e1ea22da984656f50e290049f60ddca65383e5983c0f8884f648d71f698303ad SHA512 (pciutils-2.2.9.tar.bz2) = 2b3d98d027e46d8c08037366dde6f0781ca03c610ef2b380984639e4ef39899ed8d8b8e4cd9c9dc54df101279b95879bd66bfd4d04ad07fef41e847ea7ae32b5 -SHA512 (mini-os-4.21.0.tar.xz) = 7543774d15da84476d93d04154990923c82209cb3fa125574c0383652c5a310957200f54b63b34502161f9c3afce4907e0060d9036f3eaf3a7cb6b1b3119b546 -SHA512 (xen-4.21.1.tar.xz) = 8dfe65255e202b3dacf9d0d7265636bc1f97627c11b08babc13a5b8e74c7c65e7e2c6a1513e28b3c713fe512edb6702a73b2bf667e2a8f2ce825b196a2cd5aab +SHA512 (xen-4.11.2.tar.gz) = 48d3d926d35eb56c79c06d0abc6e6be2564fadb43367cc7f46881c669a75016707672179c2cca1c4cfb14af2cefd46e2e7f99470cddf7df2886d8435a2de814e diff --git a/xen-net-disable-iptables-on-bridge.patch b/xen-net-disable-iptables-on-bridge.patch new file mode 100644 index 0000000..bc2de21 --- /dev/null +++ b/xen-net-disable-iptables-on-bridge.patch @@ -0,0 +1,27 @@ +--- xen-4.1.0-orig/tools/hotplug/Linux/vif-bridge 2008-08-22 10:49:07.000000000 +0100 ++++ xen-4.1.0-new/tools/hotplug/Linux/vif-bridge 2008-08-29 11:29:38.000000000 +0100 +@@ -96,8 +96,6 @@ case "$command" in + ;; + esac + +-handle_iptable +- + call_hooks vif post + + log debug "Successful vif-bridge $command for $dev, bridge $bridge." +--- xen-3.3.0-orig/tools/hotplug/Linux/xen-network-common.sh 2008-08-22 10:49:07.000000000 +0100 ++++ xen-3.3.0-new/tools/hotplug/Linux/xen-network-common.sh 2008-08-29 11:29:38.000000000 +0100 +@@ -99,6 +99,13 @@ create_bridge () { + brctl addbr ${bridge} + brctl stp ${bridge} off + brctl setfd ${bridge} 0 ++ # Setting these to zero stops guest<->LAN traffic ++ # traversing the bridge from hitting the *tables ++ # rulesets. guest<->host traffic still gets processed ++ # by the host's iptables rules so this isn't a hole ++ sysctl -q -w "net.bridge.bridge-nf-call-arptables=0" ++ sysctl -q -w "net.bridge.bridge-nf-call-ip6tables=0" ++ sysctl -q -w "net.bridge.bridge-nf-call-iptables=0" + fi + } + diff --git a/xen.canonicalize.patch b/xen.canonicalize.patch index 45fa724..43ccf02 100644 --- a/xen.canonicalize.patch +++ b/xen.canonicalize.patch @@ -1,54 +1,56 @@ ---- xen-4.18.0-rc1/tools/xenstored/watch.c.orig 2023-09-29 09:09:29.000000000 +0100 -+++ xen-4.18.0-rc1/tools/xenstored/watch.c 2023-10-02 16:12:14.971264769 +0100 -@@ -164,7 +164,7 @@ - const char **path, bool *relative) - { - *relative = !strstarts(*path, "/") && !strstarts(*path, "@"); -- *path = canonicalize(conn, ctx, *path, true); -+ *path = xenstore_canonicalize(conn, ctx, *path, true); - - return *path ? 0 : errno; - } -@@ -250,7 +250,7 @@ +--- xen-4.9.0-rc1.2/tools/xenstore/xenstored_watch.c.orig 2017-04-12 16:18:57.000000000 +0100 ++++ xen-4.9.0-rc1.2/tools/xenstore/xenstored_watch.c 2017-04-13 21:17:12.255231094 +0100 +@@ -166,7 +166,7 @@ + /* check if valid event */ + } else { + relative = !strstarts(vec[0], "/"); +- vec[0] = canonicalize(conn, in, vec[0]); ++ vec[0] = xenstore_canonicalize(conn, in, vec[0]); + if (!vec[0]) + return ENOMEM; + if (!is_valid_nodename(vec[0])) +@@ -219,7 +219,7 @@ if (get_strings(in, vec, ARRAY_SIZE(vec)) != ARRAY_SIZE(vec)) return EINVAL; -- node = canonicalize(conn, ctx, vec[0], true); -+ node = xenstore_canonicalize(conn, ctx, vec[0], true); +- node = canonicalize(conn, in, vec[0]); ++ node = xenstore_canonicalize(conn, in, vec[0]); if (!node) - return errno; + return ENOMEM; list_for_each_entry(watch, &conn->watches, list) { ---- xen-4.18.0-rc1/tools/xenstored/core.c.orig 2023-09-29 09:09:29.000000000 +0100 -+++ xen-4.18.0-rc1/tools/xenstored/core.c 2023-10-02 16:12:14.993264626 +0100 -@@ -1249,7 +1249,7 @@ +--- xen-4.9.0-rc1.2/tools/xenstore/xenstored_core.c.orig 2017-04-12 16:18:57.000000000 +0100 ++++ xen-4.9.0-rc1.2/tools/xenstore/xenstored_core.c 2017-04-13 21:19:35.668429881 +0100 +@@ -777,7 +777,7 @@ return strings; } --const char *canonicalize(struct connection *conn, const void *ctx, -+const char *xenstore_canonicalize(struct connection *conn, const void *ctx, - const char *node, bool allow_special) +-char *canonicalize(struct connection *conn, const void *ctx, const char *node) ++char *xenstore_canonicalize(struct connection *conn, const void *ctx, const char *node) { - const char *name; -@@ -1303,7 +1303,7 @@ - { - struct node *node; + const char *prefix; -- *canonical_name = canonicalize(conn, ctx, name, allow_special); -+ *canonical_name = xenstore_canonicalize(conn, ctx, name, allow_special); - if (!*canonical_name) - return NULL; +@@ -799,7 +799,7 @@ -@@ -1320,7 +1320,7 @@ - const char *tmp_name; - const struct node *node; + if (!canonical_name) + canonical_name = &tmp_name; +- *canonical_name = canonicalize(conn, ctx, name); ++ *canonical_name = xenstore_canonicalize(conn, ctx, name); + return get_node(conn, ctx, *canonical_name, perm); + } -- tmp_name = canonicalize(conn, ctx, name, allow_special); -+ tmp_name = xenstore_canonicalize(conn, ctx, name, allow_special); - if (!tmp_name) - return NULL; +--- xen-4.9.0-rc1.2/tools/xenstore/xenstored_core.h.orig 2017-04-12 16:18:57.000000000 +0100 ++++ xen-4.9.0-rc1.2/tools/xenstore/xenstored_core.h 2017-04-13 21:20:29.146368478 +0100 +@@ -148,7 +148,7 @@ + void send_ack(struct connection *conn, enum xsd_sockmsg_type type); ---- xen-4.18.0-rc1/tools/console/testsuite/console-dom0.c.orig 2023-09-29 09:09:29.000000000 +0100 -+++ xen-4.18.0-rc1/tools/console/testsuite/console-dom0.c 2023-10-02 16:12:15.001264574 +0100 + /* Canonicalize this path if possible. */ +-char *canonicalize(struct connection *conn, const void *ctx, const char *node); ++char *xenstore_canonicalize(struct connection *conn, const void *ctx, const char *node); + + /* Write a node to the tdb data base. */ + int write_node_raw(struct connection *conn, TDB_DATA *key, struct node *node); +--- xen-4.8.0/tools/console/testsuite/console-dom0.c.orig 2016-12-05 12:03:27.000000000 +0000 ++++ xen-4.8.0/tools/console/testsuite/console-dom0.c 2017-02-26 21:52:24.554678631 +0000 @@ -18,7 +18,7 @@ } } @@ -85,8 +87,8 @@ fprintf(stderr, "%s", line); } while (strcmp(line, "Okay.\n") != 0); ---- xen-4.18.0-rc1/tools/console/testsuite/console-domU.c.orig 2023-09-29 09:09:29.000000000 +0100 -+++ xen-4.18.0-rc1/tools/console/testsuite/console-domU.c 2023-10-02 16:12:15.008264528 +0100 +--- xen-4.8.0/tools/console/testsuite/console-domU.c.orig 2016-12-05 12:03:27.000000000 +0000 ++++ xen-4.8.0/tools/console/testsuite/console-domU.c 2017-02-26 21:52:50.320622804 +0000 @@ -6,7 +6,7 @@ #include #include @@ -105,14 +107,3 @@ seed = strtoul(line, 0, 0); printf("Seed Okay.\n"); fflush(stdout); ---- xen-4.18.0-rc1/tools/xenstored/core.h.orig 2023-09-29 09:09:29.000000000 +0100 -+++ xen-4.18.0-rc1/tools/xenstored/core.h 2023-10-02 16:12:15.015264482 +0100 -@@ -240,7 +240,7 @@ - void send_ack(struct connection *conn, enum xsd_sockmsg_type type); - - /* Canonicalize this path if possible. */ --const char *canonicalize(struct connection *conn, const void *ctx, -+const char *xenstore_canonicalize(struct connection *conn, const void *ctx, - const char *node, bool allow_special); - - /* Get access permissions. */ diff --git a/xen.drop.brctl.patch b/xen.drop.brctl.patch new file mode 100644 index 0000000..8d51b1e --- /dev/null +++ b/xen.drop.brctl.patch @@ -0,0 +1,100 @@ +--- xen-4.11.0-rc7/tools/hotplug/Linux/colo-proxy-setup.orig 2018-06-28 08:39:45.000000000 +0100 ++++ xen-4.11.0-rc7/tools/hotplug/Linux/colo-proxy-setup 2018-07-03 20:09:26.637017216 +0100 +@@ -76,10 +76,10 @@ + + function setup_secondary() + { +- do_without_error brctl delif $bridge $vifname +- do_without_error brctl addbr $forwardbr +- do_without_error brctl addif $forwardbr $vifname +- do_without_error brctl addif $forwardbr $forwarddev ++ do_without_error ip link set $vifname nomaster ++ do_without_error ip link add name $forwardbr type bridge ++ do_without_error ip link set $vifname master $forwardbr ++ do_without_error ip link set $forwarddev master $forwardbr + do_without_error ip link set dev $forwardbr up + do_without_error modprobe xt_SECCOLO + +@@ -91,10 +91,10 @@ + + function teardown_secondary() + { +- do_without_error brctl delif $forwardbr $forwarddev +- do_without_error brctl delif $forwardbr $vifname +- do_without_error brctl delbr $forwardbr +- do_without_error brctl addif $bridge $vifname ++ do_without_error ip link set $forwarddev nomaster ++ do_without_error ip link set $vifname nomaster ++ do_without_error ip link delete $forwardbr type bridge ++ do_without_error ip link set $vifname master $bridge + + do_without_error iptables -t mangle -D PREROUTING -m physdev --physdev-in \ + $vifname -j SECCOLO --index $index +--- xen-4.11.0-rc7/tools/hotplug/Linux/vif2.orig 2018-06-28 08:39:45.000000000 +0100 ++++ xen-4.11.0-rc7/tools/hotplug/Linux/vif2 2018-07-03 20:11:07.558757301 +0100 +@@ -7,13 +7,12 @@ + bridge=$(xenstore_read_default "$XENBUS_PATH/bridge" "$bridge") + if [ -z "$bridge" ] + then +- nr_bridges=$(($(brctl show | cut -f 1 | grep -v "^$" | wc -l) - 1)) ++ nr_bridges=$(bridge link | wc -l) + if [ "$nr_bridges" != 1 ] + then + fatal "no bridge specified, and don't know which one to use ($nr_bridges found)" + fi +- bridge=$(brctl show | cut -d " +-" -f 2 | cut -f 1) ++ bridge=$(bridge link | cut -d" " -f10) + fi + + command="$1" +--- xen-4.11.0-rc7/tools/hotplug/Linux/vif-bridge.orig 2018-07-03 19:59:18.499474117 +0100 ++++ xen-4.11.0-rc7/tools/hotplug/Linux/vif-bridge 2018-07-03 20:12:31.088852864 +0100 +@@ -33,7 +33,7 @@ + + if [ -z "$bridge" ] + then +- bridge=$(brctl show | awk 'NR==2{print$1}') ++ bridge=$(bridge link | cut -d" " -f10) + + if [ -z "$bridge" ] + then +@@ -82,7 +82,7 @@ + ;; + + offline) +- do_without_error brctl delif "$bridge" "$dev" ++ do_without_error ip link set "$dev" nomaster + do_without_error ifconfig "$dev" down + ;; + +--- xen-4.11.0-rc7/tools/hotplug/Linux/xen-network-common.sh.orig 2018-07-03 19:59:18.500474154 +0100 ++++ xen-4.11.0-rc7/tools/hotplug/Linux/xen-network-common.sh 2018-07-03 20:16:16.466205182 +0100 +@@ -111,9 +111,7 @@ + + # Don't create the bridge if it already exists. + if [ ! -e "/sys/class/net/${bridge}/bridge" ]; then +- brctl addbr ${bridge} +- brctl stp ${bridge} off +- brctl setfd ${bridge} 0 ++ ip link add name ${bridge} type bridge stp_state 0 forward_delay 0 + # Setting these to zero stops guest<->LAN traffic + # traversing the bridge from hitting the *tables + # rulesets. guest<->host traffic still gets processed +@@ -134,7 +132,7 @@ + ip link set dev ${dev} up || true + return + fi +- brctl addif ${bridge} ${dev} ++ ip link set ${dev} master ${bridge} + ip link set dev ${dev} up + } + +--- xen-4.11.0-rc7/tools/qemu-xen-traditional/i386-dm/qemu-ifup-Linux.orig 2017-09-15 19:37:27.000000000 +0100 ++++ xen-4.11.0-rc7/tools/qemu-xen-traditional/i386-dm/qemu-ifup-Linux 2018-07-03 20:17:52.934780235 +0100 +@@ -34,4 +34,4 @@ + fi + + ifconfig $1 0.0.0.0 up +-brctl addif $bridge $1 || true ++ip link set $1 master $bridge || true diff --git a/xen.efi.build.patch b/xen.efi.build.patch deleted file mode 100644 index b5455df..0000000 --- a/xen.efi.build.patch +++ /dev/null @@ -1,13 +0,0 @@ ---- xen-4.20.0-rc4/xen/arch/x86/arch.mk.orig 2025-02-07 11:56:01.000000000 +0000 -+++ xen-4.20.0-rc4/xen/arch/x86/arch.mk 2025-02-09 22:56:05.579507311 +0000 -@@ -95,7 +95,9 @@ - -c $(srctree)/$(efi-check).c -o $(efi-check).o,y) - - # Check if the linker supports PE. --EFI_LDFLAGS := $(patsubst -m%,-mi386pep,$(LDFLAGS)) --subsystem=10 --enable-long-section-names -+#EFI_LDFLAGS := $(patsubst -m%,-mi386pep,$(LDFLAGS)) --subsystem=10 --enable-long-section-names -+# use a reduced set of options from LDFLAGS -+EFI_LDFLAGS = --as-needed --build-id=sha1 -mi386pep --subsystem=10 --enable-long-section-names - LD_PE_check_cmd = $(call ld-option,$(EFI_LDFLAGS) --image-base=0x100000000 -o $(efi-check).efi $(efi-check).o) - XEN_BUILD_PE := $(LD_PE_check_cmd) - diff --git a/xen.fedora.crypt.patch b/xen.fedora.crypt.patch new file mode 100644 index 0000000..7aba2d4 --- /dev/null +++ b/xen.fedora.crypt.patch @@ -0,0 +1,11 @@ +--- xen-4.5.1/tools/qemu-xen-traditional/vnc.c.orig 2015-07-12 21:55:32.875504811 +0100 ++++ xen-4.5.1/tools/qemu-xen-traditional/vnc.c 2015-07-12 22:03:03.860005391 +0100 +@@ -2140,7 +2140,7 @@ + GNUTLS_VERSION_NUMBER >= 0x020200 /* 2.2.0 */ + static int vnc_set_gnutls_priority(gnutls_session_t s, int x509) + { +- const char *priority = x509 ? "NORMAL" : "NORMAL:+ANON-DH"; ++ const char *priority = x509 ? "@SYSTEM" : "@SYSTEM:+ANON-DH"; + int rc; + + rc = gnutls_priority_set_direct(s, priority, NULL); diff --git a/xen.fedora.efi.build.patch b/xen.fedora.efi.build.patch new file mode 100644 index 0000000..36f9608 --- /dev/null +++ b/xen.fedora.efi.build.patch @@ -0,0 +1,10 @@ +--- xen-4.8.0/xen/Makefile.orig 2016-12-05 12:03:27.000000000 +0000 ++++ xen-4.8.0/xen/Makefile 2017-02-28 00:02:54.080529810 +0000 +@@ -20,6 +20,7 @@ + MAKEFLAGS += -rR + + EFI_MOUNTPOINT ?= $(BOOT_DIR)/efi ++EFI_VENDOR=fedora + + ARCH=$(XEN_TARGET_ARCH) + SRCARCH=$(shell echo $(ARCH) | sed -e 's/x86.*/x86/' -e s'/arm\(32\|64\)/arm/g') diff --git a/xen.fedora.systemd.patch b/xen.fedora.systemd.patch index 5b6a7a3..3b75ed0 100644 --- a/xen.fedora.systemd.patch +++ b/xen.fedora.systemd.patch @@ -1,6 +1,7 @@ ---- xen-4.17.0/tools/hotplug/Linux/systemd/Makefile.orig 2022-12-08 18:03:08.000000000 +0000 -+++ xen-4.17.0/tools/hotplug/Linux/systemd/Makefile 2022-12-09 19:47:53.227189371 +0000 -@@ -10,7 +10,8 @@ +diff -uN xen-4.5.0/tools/hotplug/Linux/systemd.orig/Makefile xen-4.5.0/tools/hotplug/Linux/systemd/Makefile +--- xen-4.5.0/tools/hotplug/Linux/systemd.orig/Makefile 2015-01-12 16:53:24.000000000 +0000 ++++ xen-4.5.0/tools/hotplug/Linux/systemd/Makefile 2015-01-25 22:23:26.000000000 +0000 +@@ -14,7 +14,8 @@ XEN_SYSTEMD_SERVICE += xen-qemu-dom0-disk-backend.service XEN_SYSTEMD_SERVICE += xendomains.service XEN_SYSTEMD_SERVICE += xen-watchdog.service @@ -9,7 +10,16 @@ +XEN_SYSTEMD_SERVICE += oxenstored.service XEN_SYSTEMD_SERVICE += xendriverdomain.service - ALL_XEN_SYSTEMD := $(XEN_SYSTEMD_MODULES) \ + ALL_XEN_SYSTEMD = $(XEN_SYSTEMD_MODULES) \ +diff -uN xen-4.5.0/tools/hotplug/Linux/systemd.orig/var-lib-xenstored.mount.in xen-4.5.0/tools/hotplug/Linux/systemd/var-lib-xenstored.mount.in +--- xen-4.5.0/tools/hotplug/Linux/systemd.orig/var-lib-xenstored.mount.in 2015-01-12 16:53:24.000000000 +0000 ++++ xen-4.5.0/tools/hotplug/Linux/systemd/var-lib-xenstored.mount.in 2015-01-25 22:28:59.000000000 +0000 +@@ -9,4 +9,4 @@ + What=xenstore + Where=@XEN_LIB_STORED@ + Type=tmpfs +-Options=mode=755 ++Options=mode=755,context="system_u:object_r:xenstored_var_lib_t:s0" diff -uN xen-4.5.0/tools/hotplug/Linux/systemd.orig/xenconsoled.service.in xen-4.5.0/tools/hotplug/Linux/systemd/xenconsoled.service.in --- xen-4.5.0/tools/hotplug/Linux/systemd.orig/xenconsoled.service.in 2015-01-12 16:53:24.000000000 +0000 +++ xen-4.5.0/tools/hotplug/Linux/systemd/xenconsoled.service.in 2015-01-25 22:30:26.000000000 +0000 @@ -49,26 +59,27 @@ diff -uN xen-4.5.0/tools/hotplug/Linux/systemd.orig/xen-qemu-dom0-disk-backend.s Before=xendomains.service libvirtd.service libvirt-guests.service RefuseManualStop=true ConditionPathExists=/proc/xen/capabilities ---- xen-4.17.0/tools/configure.ac.orig 2022-12-08 18:03:08.000000000 +0000 -+++ xen-4.17.0/tools/configure.ac 2022-12-09 19:50:24.773193862 +0000 -@@ -481,8 +481,8 @@ +--- xen-4.6.0/tools/configure.ac.orig 2015-02-15 16:47:22.000000000 +0000 ++++ xen-4.6.0/tools/configure.ac 2015-03-01 16:18:30.493647587 +0000 +@@ -382,9 +382,9 @@ AS_IF([test "x$systemd" = "xy"], [ AC_CONFIG_FILES([ + hotplug/Linux/systemd/oxenstored.service hotplug/Linux/systemd/proc-xen.mount + hotplug/Linux/systemd/var-lib-xenstored.mount - hotplug/Linux/systemd/xen-init-dom0.service hotplug/Linux/systemd/xen-qemu-dom0-disk-backend.service hotplug/Linux/systemd/xen-watchdog.service hotplug/Linux/systemd/xenconsoled.service ---- xen-4.17.0/tools/configure.orig 2022-12-08 18:03:08.000000000 +0000 -+++ xen-4.17.0/tools/configure 2022-12-09 19:51:43.278708226 +0000 -@@ -10081,7 +10081,7 @@ - if test "x$systemd" = "xy" - then : +--- xen-4.6.0/tools/configure.orig 2015-02-15 16:47:22.000000000 +0000 ++++ xen-4.6.0/tools/configure 2015-03-01 16:20:10.648285840 +0000 +@@ -8995,7 +8995,7 @@ -- ac_config_files="$ac_config_files hotplug/Linux/systemd/proc-xen.mount hotplug/Linux/systemd/xen-init-dom0.service hotplug/Linux/systemd/xen-qemu-dom0-disk-backend.service hotplug/Linux/systemd/xen-watchdog.service hotplug/Linux/systemd/xenconsoled.service hotplug/Linux/systemd/xendomains.service hotplug/Linux/systemd/xendriverdomain.service hotplug/Linux/systemd/xenstored.service" -+ ac_config_files="$ac_config_files hotplug/Linux/systemd/oxenstored.service hotplug/Linux/systemd/proc-xen.mount hotplug/Linux/systemd/xen-qemu-dom0-disk-backend.service hotplug/Linux/systemd/xen-watchdog.service hotplug/Linux/systemd/xenconsoled.service hotplug/Linux/systemd/xendomains.service hotplug/Linux/systemd/xendriverdomain.service hotplug/Linux/systemd/xenstored.service" + if test "x$systemd" = "xy"; then : + +- ac_config_files="$ac_config_files hotplug/Linux/systemd/proc-xen.mount hotplug/Linux/systemd/var-lib-xenstored.mount hotplug/Linux/systemd/xen-init-dom0.service hotplug/Linux/systemd/xen-qemu-dom0-disk-backend.service hotplug/Linux/systemd/xen-watchdog.service hotplug/Linux/systemd/xenconsoled.service hotplug/Linux/systemd/xendomains.service hotplug/Linux/systemd/xendriverdomain.service hotplug/Linux/systemd/xenstored.service" ++ ac_config_files="$ac_config_files hotplug/Linux/systemd/oxenstored.service hotplug/Linux/systemd/proc-xen.mount hotplug/Linux/systemd/var-lib-xenstored.mount hotplug/Linux/systemd/xen-qemu-dom0-disk-backend.service hotplug/Linux/systemd/xen-watchdog.service hotplug/Linux/systemd/xenconsoled.service hotplug/Linux/systemd/xendomains.service hotplug/Linux/systemd/xendriverdomain.service hotplug/Linux/systemd/xenstored.service" fi diff --git a/xen.gcc11.fixes.patch b/xen.gcc11.fixes.patch deleted file mode 100644 index 6971080..0000000 --- a/xen.gcc11.fixes.patch +++ /dev/null @@ -1,24 +0,0 @@ ---- xen-4.14.0/xen/include/crypto/vmac.h.orig 2020-07-23 16:07:51.000000000 +0100 -+++ xen-4.14.0/xen/include/crypto/vmac.h 2020-10-24 15:45:49.246467465 +0100 -@@ -142,7 +142,7 @@ - - #define vmac_update vhash_update - --void vhash_update(unsigned char m[], -+void vhash_update(uint8_t *m, - unsigned int mbytes, - vmac_ctx_t *ctx); - -diff --git a/xen/arch/x86/tboot.c b/xen/arch/x86/tboot.c -index 320e06f..618ae92 100644 ---- a/xen/arch/x86/tboot.c -+++ b/xen/arch/x86/tboot.c -@@ -91,7 +91,7 @@ static void __init tboot_copy_memory(unsigned char *va, uint32_t size, - - void __init tboot_probe(void) - { -- tboot_shared_t *tboot_shared; -+ tboot_shared_t * volatile tboot_shared; - static const uuid_t __initconst tboot_shared_uuid = TBOOT_SHARED_UUID; - - /* Look for valid page-aligned address for shared page. */ diff --git a/xen.gcc12.fixes.patch b/xen.gcc12.fixes.patch deleted file mode 100644 index b35440f..0000000 --- a/xen.gcc12.fixes.patch +++ /dev/null @@ -1,10 +0,0 @@ ---- xen-4.16.0/Config.mk.orig 2021-11-30 11:42:42.000000000 +0000 -+++ xen-4.16.0/Config.mk 2022-01-24 20:25:16.687125822 +0000 -@@ -186,6 +186,7 @@ - - $(call cc-option-add,CFLAGS,CC,-Wno-unused-but-set-variable) - $(call cc-option-add,CFLAGS,CC,-Wno-unused-local-typedefs) -+$(call cc-option-add,CFLAGS,CC,-Wno-error=array-bounds) - - LDFLAGS += $(foreach i, $(EXTRA_LIB), -L$(i)) - CFLAGS += $(foreach i, $(EXTRA_INCLUDES), -I$(i)) diff --git a/xen.gcc7.fix.patch b/xen.gcc7.fix.patch new file mode 100644 index 0000000..b18ba2b --- /dev/null +++ b/xen.gcc7.fix.patch @@ -0,0 +1,12 @@ +--- xen-4.8.0/extras/mini-os/Makefile.orig 2016-09-28 12:09:38.000000000 +0100 ++++ xen-4.8.0/extras/mini-os/Makefile 2017-02-15 21:15:19.340197960 +0000 +@@ -142,6 +142,9 @@ + APP_LDLIBS += -lz + APP_LDLIBS += -lm + LDLIBS += -lc ++ifeq ($(MINIOS_TARGET_ARCH),x86_32) ++LDLIBS += -L$(shell dirname `gcc -m32 -print-libgcc-file-name`) -lgcc ++endif + endif + + ifneq ($(APP_OBJS)-$(lwip),-y) diff --git a/xen.gcc8.temp.fix.patch b/xen.gcc8.temp.fix.patch new file mode 100644 index 0000000..74321aa --- /dev/null +++ b/xen.gcc8.temp.fix.patch @@ -0,0 +1,70 @@ +--- xen-4.10.0/tools/Makefile.orig 2017-12-13 11:37:59.000000000 +0000 ++++ xen-4.10.0/tools/Makefile 2018-02-27 12:04:44.376192357 +0000 +@@ -8,7 +8,7 @@ + SUBDIRS-y += libs + SUBDIRS-y += libxc + SUBDIRS-y += flask +-SUBDIRS-y += fuzz ++#SUBDIRS-y += fuzz + SUBDIRS-y += xenstore + SUBDIRS-y += misc + SUBDIRS-y += examples +--- xen-4.10.0/tools/debugger/kdd/kdd.c.orig 2018-02-22 12:31:57.007039159 +0000 ++++ xen-4.10.0/tools/debugger/kdd/kdd.c 2018-02-22 18:27:37.213653422 +0000 +@@ -687,7 +687,7 @@ + } + } else { + /* 32-bit control-register space starts at 0x[2]cc, for 84 bytes */ +- uint32_t offset = addr; ++/* uint32_t offset = addr; + if (offset > 0x200) + offset -= 0x200; + offset -= 0xcc; +@@ -696,7 +696,9 @@ + len = 0; + } else { + memcpy(buf, ((uint8_t *)&ctrl.c32) + offset, len); +- } ++ } */ ++ /* disable above code due to compile issue for now */ ++ len = 0; + } + + s->txp.cmd.mem.addr = addr; +--- xen-4.10.0/tools/libxl/libxl_arm_acpi.c.orig 2017-12-13 11:37:59.000000000 +0000 ++++ xen-4.10.0/tools/libxl/libxl_arm_acpi.c 2018-02-28 12:37:08.887221211 +0000 +@@ -190,7 +190,7 @@ + struct acpi_table_rsdp *rsdp = (void *)dom->acpi_modules[0].data + offset; + + memcpy(rsdp->signature, "RSD PTR ", sizeof(rsdp->signature)); +- memcpy(rsdp->oem_id, ACPI_OEM_ID, sizeof(rsdp->oem_id)); ++ memcpy(rsdp->oem_id, ACPI_OEM_ID, sizeof(ACPI_OEM_ID)); + rsdp->length = acpitables[RSDP].size; + rsdp->revision = 0x02; + rsdp->xsdt_physical_address = acpitables[XSDT].addr; +@@ -205,11 +205,11 @@ + memcpy(h->signature, sig, 4); + h->length = len; + h->revision = rev; +- memcpy(h->oem_id, ACPI_OEM_ID, sizeof(h->oem_id)); +- memcpy(h->oem_table_id, ACPI_OEM_TABLE_ID, sizeof(h->oem_table_id)); ++ memcpy(h->oem_id, ACPI_OEM_ID, sizeof(ACPI_OEM_ID)); ++ memcpy(h->oem_table_id, ACPI_OEM_TABLE_ID, sizeof(ACPI_OEM_TABLE_ID)); + h->oem_revision = 0; + memcpy(h->asl_compiler_id, ACPI_ASL_COMPILER_ID, +- sizeof(h->asl_compiler_id)); ++ sizeof(ACPI_ASL_COMPILER_ID)); + h->asl_compiler_revision = 0; + h->checksum = 0; + } +--- xen-4.10.0/tools/xenpmd/xenpmd.c.orig 2018-02-28 16:18:50.377726049 +0000 ++++ xen-4.10.0/tools/xenpmd/xenpmd.c 2018-02-28 16:20:31.502426829 +0000 +@@ -352,7 +352,7 @@ + strlen(info->model_number) + + strlen(info->serial_number) + + strlen(info->battery_type) + +- strlen(info->oem_info) + 4)); ++ strlen(info->oem_info) + 4) & 0xff); + write_ulong_lsb_first(val+2, info->present); + write_ulong_lsb_first(val+10, info->design_capacity); + write_ulong_lsb_first(val+18, info->last_full_capacity); diff --git a/xen.gcc9.fixes.patch b/xen.gcc9.fixes.patch index 16c526a..73b8e44 100644 --- a/xen.gcc9.fixes.patch +++ b/xen.gcc9.fixes.patch @@ -13,7 +13,7 @@ +++ xen-4.11.1/xen/arch/x86/cpu/mtrr/generic.c 2019-02-10 19:24:09.378805103 +0000 @@ -171,6 +171,9 @@ printk("%sMTRR variable ranges %sabled:\n", level, - mtrr_state.enabled ? "en" : "dis"); + mtrr_state.enabled & 2 ? "en" : "dis"); width = (paddr_bits - PAGE_SHIFT + 3) / 4; + if ( width > 64 ) { + width=64; @@ -21,3 +21,14 @@ for (i = 0; i < num_var_ranges; ++i) { if (mtrr_state.var_ranges[i].mask & MTRR_PHYSMASK_VALID) +--- xen-4.11.1/tools/firmware/rombios/32bit/rombios_compat.h.orig 2018-11-29 14:04:11.000000000 +0000 ++++ xen-4.11.1/tools/firmware/rombios/32bit/rombios_compat.h 2019-02-14 21:16:28.660456669 +0000 +@@ -52,7 +52,7 @@ + Bit16u filler4; + } r8; + } u; +-} __attribute__((packed)) pushad_regs_t; ++} pushad_regs_t; + + + diff --git a/xen.git-90b20547b756a5cf9b0fec9fb0de5b361e8bf4c3.patch b/xen.git-90b20547b756a5cf9b0fec9fb0de5b361e8bf4c3.patch deleted file mode 100644 index a5e65ba..0000000 --- a/xen.git-90b20547b756a5cf9b0fec9fb0de5b361e8bf4c3.patch +++ /dev/null @@ -1,88 +0,0 @@ -From 90b20547b756a5cf9b0fec9fb0de5b361e8bf4c3 Mon Sep 17 00:00:00 2001 -From: Andrew Cooper -Date: Fri, 10 Apr 2026 21:55:46 +0100 -Subject: [PATCH] x86/amd: Mitigate AMD-SN-7053 / FP-DSS -MIME-Version: 1.0 -Content-Type: text/plain; charset=utf8 -Content-Transfer-Encoding: 8bit - -This is XSA-488 / CVE-2025-54505 - -Signed-off-by: Andrew Cooper -Reviewed-by: Roger Pau Monné -(cherry picked from commit 99912d346009fda1e7fb1510c9501fbab17e92a0) ---- - xen/arch/x86/cpu/amd.c | 37 ++++++++++++++++++++++++++++ - xen/arch/x86/include/asm/msr-index.h | 1 + - 2 files changed, 38 insertions(+) - -diff --git a/xen/arch/x86/cpu/amd.c b/xen/arch/x86/cpu/amd.c -index 8c55d233f3..1bb0766ebf 100644 ---- a/xen/arch/x86/cpu/amd.c -+++ b/xen/arch/x86/cpu/amd.c -@@ -1048,6 +1048,42 @@ void amd_init_de_cfg(const struct cpuinfo_x86 *c) - wrmsrl(MSR_AMD64_DE_CFG, val | new); - } - -+static void amd_init_fp_cfg(const struct cpuinfo_x86 *c) -+{ -+ uint64_t val, new = 0; -+ -+ /* If virtualised, we won't have mutable access even if we can read it. */ -+ if ( cpu_has_hypervisor ) -+ return; -+ -+ /* -+ * On Zen1, mitigate SB-7053 / FP-DSS Floating Point Divider State -+ * Sampling by setting bit 9 as instructed. -+ */ -+ if ( c->family == 0x17 && is_zen1_uarch() ) -+ new |= 1 << 9; -+ -+ /* -+ * Avoid reading FP_CFG if we don't intend to change anything. The -+ * register doesn't exist on all families. -+ */ -+ if ( !new ) -+ return; -+ -+ val = rdmsr(MSR_AMD64_FP_CFG); -+ -+ if ( (val & new) == new ) -+ return; -+ -+ /* -+ * FP_CFG is a Core-scoped MSR, and this write is racy. However, both -+ * threads calculate the new value from state which expected to be -+ * consistent across CPUs and unrelated to the old value, so the result -+ * should be consistent. -+ */ -+ wrmsr(MSR_AMD64_FP_CFG, val | new); -+} -+ - void __init amd_init_lfence_dispatch(void) - { - struct cpuinfo_x86 *c = &boot_cpu_data; -@@ -1120,6 +1156,7 @@ static void cf_check init_amd(struct cpuinfo_x86 *c) - uint64_t value; - - amd_init_de_cfg(c); -+ amd_init_fp_cfg(c); - - if (c == &boot_cpu_data) - amd_init_lfence_dispatch(); /* Needs amd_init_de_cfg() */ -diff --git a/xen/arch/x86/include/asm/msr-index.h b/xen/arch/x86/include/asm/msr-index.h -index df52587c85..6c5b2569e1 100644 ---- a/xen/arch/x86/include/asm/msr-index.h -+++ b/xen/arch/x86/include/asm/msr-index.h -@@ -428,6 +428,7 @@ - #define MSR_AMD64_LS_CFG 0xc0011020U - #define MSR_AMD64_IC_CFG 0xc0011021U - #define MSR_AMD64_DC_CFG 0xc0011022U -+#define MSR_AMD64_FP_CFG 0xc0011028U - #define MSR_AMD64_DE_CFG 0xc0011029U - #define AMD64_DE_CFG_LFENCE_SERIALISE (_AC(1, ULL) << 1) - #define MSR_AMD64_EX_CFG 0xc001102cU --- -2.39.5 - diff --git a/xen.glibcfix.patch b/xen.glibcfix.patch new file mode 100644 index 0000000..7b264d5 --- /dev/null +++ b/xen.glibcfix.patch @@ -0,0 +1,20 @@ +--- xen-4.7.0/tools/blktap2/control/tap-ctl-allocate.c.orig 2016-06-20 11:38:15.000000000 +0100 ++++ xen-4.7.0/tools/blktap2/control/tap-ctl-allocate.c 2016-09-02 10:07:55.964084808 +0100 +@@ -36,6 +36,7 @@ + #include + #include + #include ++#include + #include + + #include "tap-ctl.h" +--- xen-4.7.0/tools/libxl/libxl_internal.h.orig 2016-06-20 11:38:15.000000000 +0100 ++++ xen-4.7.0/tools/libxl/libxl_internal.h 2016-09-02 17:35:24.853783711 +0100 +@@ -47,6 +47,7 @@ + #include + #include + #include ++#include + + #include + #include diff --git a/xen.hypervisor.config b/xen.hypervisor.config index 7f11043..50b237c 100644 --- a/xen.hypervisor.config +++ b/xen.hypervisor.config @@ -1,208 +1,80 @@ # # Automatically generated file; DO NOT EDIT. -# Xen/x86 4.20 Configuration +# Xen/x86 4.10.0 Configuration # -CONFIG_CC_IS_GCC=y -CONFIG_GCC_VERSION=150001 -CONFIG_CLANG_VERSION=0 -CONFIG_LD_IS_GNU=y -CONFIG_CC_HAS_VISIBILITY_ATTRIBUTE=y -CONFIG_CC_SPLIT_SECTIONS=y -CONFIG_FUNCTION_ALIGNMENT_16B=y -CONFIG_FUNCTION_ALIGNMENT=16 CONFIG_X86_64=y CONFIG_X86=y CONFIG_ARCH_DEFCONFIG="arch/x86/configs/x86_64_defconfig" -CONFIG_CC_HAS_INDIRECT_THUNK=y -CONFIG_HAS_AS_CET_SS=y -CONFIG_HAS_CC_CET_IBT=y # # Architecture Features # -CONFIG_AMD=y -CONFIG_INTEL=y -CONFIG_64BIT=y CONFIG_NR_CPUS=256 -CONFIG_NR_NUMA_NODES=64 CONFIG_PV=y -CONFIG_PV32=y CONFIG_PV_LINEAR_PT=y CONFIG_HVM=y -CONFIG_AMD_SVM=y -CONFIG_INTEL_VMX=y -CONFIG_XEN_SHSTK=y -CONFIG_XEN_IBT=y CONFIG_SHADOW_PAGING=y # CONFIG_BIGMEM is not set -CONFIG_HVM_FEP=y -CONFIG_X86_PSR=y -CONFIG_XEN_ALIGN_DEFAULT=y -# CONFIG_XEN_ALIGN_2M is not set -# CONFIG_X2APIC_PHYSICAL is not set -CONFIG_X2APIC_MIXED=y -# CONFIG_XEN_GUEST is not set -# CONFIG_HYPERV_GUEST is not set -# CONFIG_REQUIRE_NX is not set -CONFIG_ALTP2M=y -# end of Architecture Features +# CONFIG_HVM_FEP is not set +CONFIG_TBOOT=y # # Common Features # CONFIG_COMPAT=y CONFIG_CORE_PARKING=y -CONFIG_GRANT_TABLE=y -CONFIG_ALTERNATIVE_CALL=y -CONFIG_ARCH_MAP_DOMAIN_PAGE=y -CONFIG_GENERIC_BUG_FRAME=y CONFIG_HAS_ALTERNATIVE=y -CONFIG_HAS_COMPAT=y -CONFIG_HAS_DIT=y CONFIG_HAS_EX_TABLE=y -CONFIG_HAS_FAST_MULTIPLY=y -CONFIG_HAS_IOPORTS=y +CONFIG_HAS_MEM_ACCESS=y +CONFIG_HAS_MEM_PAGING=y +CONFIG_HAS_MEM_SHARING=y +CONFIG_HAS_PDX=y CONFIG_HAS_KEXEC=y -CONFIG_HAS_PIRQ=y -CONFIG_HAS_SCHED_GRANULARITY=y -CONFIG_HAS_UBSAN=y -CONFIG_HAS_VMAP=y -CONFIG_MEM_ACCESS_ALWAYS_ON=y -CONFIG_MEM_ACCESS=y -CONFIG_NEEDS_LIBELF=y -CONFIG_NUMA=y - -# -# Speculative hardening -# -CONFIG_INDIRECT_THUNK=y -CONFIG_RETURN_THUNK=y -CONFIG_SPECULATIVE_HARDEN_ARRAY=y -CONFIG_SPECULATIVE_HARDEN_BRANCH=y -CONFIG_SPECULATIVE_HARDEN_GUEST_ACCESS=y -CONFIG_SPECULATIVE_HARDEN_LOCK=y -# end of Speculative hardening - -# CONFIG_DIT_DEFAULT is not set -CONFIG_HYPFS=y -CONFIG_HYPFS_CONFIG=y -CONFIG_IOREQ_SERVER=y +CONFIG_HAS_GDBSX=y +CONFIG_HAS_IOPORTS=y CONFIG_KEXEC=y +CONFIG_TMEM=y +CONFIG_XENOPROF=y # CONFIG_XSM is not set CONFIG_SCHED_CREDIT=y CONFIG_SCHED_CREDIT2=y CONFIG_SCHED_RTDS=y CONFIG_SCHED_ARINC653=y CONFIG_SCHED_NULL=y -CONFIG_SCHED_DEFAULT="credit2" -# CONFIG_BOOT_TIME_CPUPOOLS is not set +CONFIG_SCHED_DEFAULT="credit" +CONFIG_CRYPTO=y CONFIG_LIVEPATCH=y CONFIG_FAST_SYMBOL_LOOKUP=y -CONFIG_ENFORCE_UNIQUE_SYMBOLS=y CONFIG_CMDLINE="" -CONFIG_DOM0_MEM="" -CONFIG_DTB_FILE="" -CONFIG_TRACEBUFFER=y -# end of Common Features # # Device Drivers # CONFIG_ACPI=y CONFIG_ACPI_LEGACY_TABLES_LOOKUP=y -CONFIG_ACPI_NUMA=y +CONFIG_NUMA=y CONFIG_HAS_NS16550=y CONFIG_HAS_EHCI=y -CONFIG_SERIAL_TX_BUFSIZE=32768 -# CONFIG_XHCI is not set CONFIG_HAS_CPUFREQ=y CONFIG_HAS_PASSTHROUGH=y -CONFIG_AMD_IOMMU=y -CONFIG_INTEL_IOMMU=y -# CONFIG_IOMMU_QUARANTINE_NONE is not set -CONFIG_IOMMU_QUARANTINE_BASIC=y -# CONFIG_IOMMU_QUARANTINE_SCRATCH_PAGE is not set CONFIG_HAS_PCI=y -CONFIG_HAS_PCI_MSI=y CONFIG_VIDEO=y CONFIG_VGA=y -CONFIG_HAS_VPCI=y -# end of Device Drivers - -# CONFIG_EXPERT is not set -# CONFIG_UNSUPPORTED is not set -CONFIG_ARCH_SUPPORTS_INT128=y -CONFIG_ARCH_VCPU_IOREQ_COMPLETION=y +CONFIG_DEFCONFIG_LIST="$ARCH_DEFCONFIG" +CONFIG_XEN_GUEST=n # # Debugging Options # # CONFIG_DEBUG is not set -CONFIG_GDBSX=y -CONFIG_FRAME_POINTER=y -CONFIG_SELF_TESTS=y -# CONFIG_DEBUG_LOCK_PROFILE is not set -CONFIG_DEBUG_LOCKS=y -# CONFIG_PERF_COUNTERS is not set -CONFIG_VERBOSE_DEBUG=y -CONFIG_SCRUB_DEBUG=y -# CONFIG_UBSAN is not set -# CONFIG_DEBUG_TRACE is not set -CONFIG_XMEM_POOL_POISON=y -CONFIG_DEBUG_INFO=y -# end of Debugging Options -# ARM64 settings -CONFIG_MMU=y -CONFIG_ARM_64=y -CONFIG_ARM=y -CONFIG_ARM_EFI=y -CONFIG_GICV2=y -CONFIG_GICV3=y -CONFIG_VGICV2=y -# CONFIG_NEW_VGIC is not set -CONFIG_SBSA_VUART_CONSOLE=y -CONFIG_HWDOM_VUART=y -CONFIG_ARM_SSBD=y -CONFIG_HARDEN_BRANCH_PREDICTOR=y -CONFIG_STATIC_EVTCHN=y -CONFIG_PARTIAL_EMULATION=y - -# # ARM errata workaround via the alternative framework # CONFIG_ARM64_ERRATUM_827319=y CONFIG_ARM64_ERRATUM_824069=y CONFIG_ARM64_ERRATUM_819472=y -CONFIG_ARM64_ERRATUM_843419=y CONFIG_ARM64_ERRATUM_832075=y CONFIG_ARM64_ERRATUM_834220=y -CONFIG_ARM_ERRATUM_858921=y -CONFIG_ARM64_WORKAROUND_REPEAT_TLBI=y -CONFIG_ARM64_ERRATUM_1286807=y -CONFIG_ARM64_ERRATUM_1508412=y -# end of ARM errata workaround via the alternative framework -CONFIG_ARM64_HARDEN_BRANCH_PREDICTOR=y -CONFIG_ALL_PLAT=y -# CONFIG_QEMU is not set -# CONFIG_RCAR3 is not set -# CONFIG_MPSOC is not set -# CONFIG_NO_PLAT is not set -CONFIG_ALL64_PLAT=y -CONFIG_MPSOC_PLATFORM=y - -# -# Common Features -# -CONFIG_HAS_DEVICE_TREE=y -CONFIG_HAS_CADENCE_UART=y -CONFIG_HAS_LINFLEX=y -CONFIG_HAS_IMX_LPUART=y -CONFIG_HAS_MVEBU=y -CONFIG_HAS_MESON=y -CONFIG_HAS_PL011=y -CONFIG_HAS_OMAP=y -CONFIG_HAS_SCIF=y -CONFIG_ARM_SMMU=y -# CONFIG_IPMMU_VMSA is not set +CONFIG_SBSA_VUART_CONSOLE=y +# CONFIG_NEW_VGIC is not set diff --git a/xen.json.nocpuid.patch b/xen.json.nocpuid.patch deleted file mode 100644 index f701f0d..0000000 --- a/xen.json.nocpuid.patch +++ /dev/null @@ -1,27 +0,0 @@ ---- xen-4.21.0/tools/libs/light/libxl_nocpuid.c.orig 2025-11-18 18:02:13.000000000 +0000 -+++ xen-4.21.0/tools/libs/light/libxl_nocpuid.c 2025-11-20 09:03:56.517804514 +0000 -@@ -40,11 +40,24 @@ - return 0; - } - -+#ifdef HAVE_LIBJSONC -+#ifndef _hidden -+#define _hidden -+#endif -+_hidden int libxl_cpuid_policy_list_gen_jso(json_object **jso_r, -+ libxl_cpuid_policy_list *pcpuid) -+{ -+ return 0; -+} -+#endif -+ -+#if defined(HAVE_LIBYAJL) - yajl_gen_status libxl_cpuid_policy_list_gen_json(yajl_gen hand, - libxl_cpuid_policy_list *pcpuid) - { - return 0; - } -+#endif - - int libxl__cpuid_policy_list_parse_json(libxl__gc *gc, - const libxl__json_object *o, diff --git a/xen.python.env.patch b/xen.python.env.patch new file mode 100644 index 0000000..e33d370 --- /dev/null +++ b/xen.python.env.patch @@ -0,0 +1,51 @@ +--- xen-4.11.0/tools/xenmon/Makefile.orig 2018-07-09 14:47:19.000000000 +0100 ++++ xen-4.11.0/tools/xenmon/Makefile 2018-09-10 21:13:15.200655105 +0100 +@@ -32,7 +32,7 @@ + $(INSTALL_DIR) $(DESTDIR)$(sbindir) + $(INSTALL_PROG) xenbaked $(DESTDIR)$(sbindir)/xenbaked + $(INSTALL_PROG) xentrace_setmask $(DESTDIR)$(sbindir)/xentrace_setmask +- $(INSTALL_PROG) xenmon.py $(DESTDIR)$(sbindir)/xenmon.py ++ $(INSTALL_PYTHON_PROG) xenmon.py $(DESTDIR)$(sbindir)/xenmon.py + + .PHONY: uninstall + uninstall: +--- xen-4.11.0/tools/misc/xen-ringwatch.orig 2018-07-09 14:47:19.000000000 +0100 ++++ xen-4.11.0/tools/misc/xen-ringwatch 2018-09-10 21:18:53.191063128 +0100 +@@ -1,4 +1,4 @@ +-#!/usr/bin/python ++#!/usr/bin/python2 + # + # Copyright (C) 2011 Citrix Systems, Inc. + # +--- xen-4.11.0/tools/python/Makefile.orig 2018-07-09 14:47:19.000000000 +0100 ++++ xen-4.11.0/tools/python/Makefile 2018-09-10 21:21:07.097979007 +0100 +@@ -20,8 +20,8 @@ + setup.py install --record $(INSTALL_LOG) $(PYTHON_PREFIX_ARG) \ + --root="$(DESTDIR)" --force + +- $(INSTALL_PROG) scripts/convert-legacy-stream $(DESTDIR)$(LIBEXEC_BIN) +- $(INSTALL_PROG) scripts/verify-stream-v2 $(DESTDIR)$(LIBEXEC_BIN) ++ $(INSTALL_PYTHON_PROG) scripts/convert-legacy-stream $(DESTDIR)$(LIBEXEC_BIN) ++ $(INSTALL_PYTHON_PROG) scripts/verify-stream-v2 $(DESTDIR)$(LIBEXEC_BIN) + + .PHONY: uninstall + uninstall: +--- xen-4.11.0/tools/python/install-wrap.orig 2018-07-09 14:47:19.000000000 +0100 ++++ xen-4.11.0/tools/python/install-wrap 2018-09-11 20:09:57.803655357 +0100 +@@ -44,7 +44,7 @@ + destf="$dest" + for srcf in ${srcs}; do + if test -d "$dest"; then +- destf="$dest/${srcf%%*/}" ++ destf="$dest/${srcf##*/}" + fi + org="$(sed -n '2q; /^#! *\/usr\/bin\/env python *$/p' $srcf)" + if test "x$org" = x; then +--- xen-4.11.0/tools/misc/xencov_split.orig 2018-07-09 14:47:19.000000000 +0100 ++++ xen-4.11.0/tools/misc/xencov_split 2018-09-18 21:56:07.397893895 +0100 +@@ -1,4 +1,4 @@ +-#!/usr/bin/python ++#!/usr/bin/python2 + + import sys, os, os.path as path, struct, errno + from optparse import OptionParser diff --git a/xen.python3.12.patch b/xen.python3.12.patch deleted file mode 100644 index a6539c2..0000000 --- a/xen.python3.12.patch +++ /dev/null @@ -1,22 +0,0 @@ ---- xen-4.17.1/tools/python/Makefile.orig 2023-04-27 13:53:19.000000000 +0100 -+++ xen-4.17.1/tools/python/Makefile 2023-06-22 22:21:25.287486906 +0100 -@@ -4,7 +4,7 @@ - .PHONY: all - all: build - --PY_CFLAGS = $(CFLAGS) $(PY_NOOPT_CFLAGS) -+PY_CFLAGS = $(CFLAGS) $(PY_NOOPT_CFLAGS) -Wno-error=declaration-after-statement - PY_LDFLAGS = $(SHLIB_LDFLAGS) $(APPEND_LDFLAGS) - INSTALL_LOG = build/installed_files.txt - ---- xen-4.17.1/tools/pygrub/Makefile.orig 2023-04-27 13:53:19.000000000 +0100 -+++ xen-4.17.1/tools/pygrub/Makefile 2023-06-22 22:52:52.803047401 +0100 -@@ -2,7 +2,7 @@ - XEN_ROOT = $(CURDIR)/../.. - include $(XEN_ROOT)/tools/Rules.mk - --PY_CFLAGS = $(CFLAGS) $(PY_NOOPT_CFLAGS) -+PY_CFLAGS = $(CFLAGS) $(PY_NOOPT_CFLAGS) -Wno-error=declaration-after-statement - PY_LDFLAGS = $(SHLIB_LDFLAGS) $(APPEND_LDFLAGS) - INSTALL_LOG = build/installed_files.txt - diff --git a/xen.spec b/xen.spec index 1bea8ef..3566ea4 100644 --- a/xen.spec +++ b/xen.spec @@ -6,12 +6,18 @@ %define build_docs %{?_without_docs: 0} %{?!_without_docs: 1} # Build with stubdom unless rpmbuild was run with --without stubdom %define build_stubdom %{?_without_stubdom: 0} %{?!_without_stubdom: 1} +# Build with qemu-traditional unless rpmbuild was run with --without qemutrad +%define build_qemutrad %{?_without_qemutrad: 0} %{?!_without_qemutrad: 1} # build with ovmf from edk2-ovmf unless rpmbuild was run with --without ovmf %define build_ovmf %{?_without_ovmf: 0} %{?!_without_ovmf: 1} -# set to 0 for archs that don't use ovmf (reduces build dependencies) -%ifnarch x86_64 +# set to 0 for archs that don't use qemu or ovmf (reduces build dependencies) +%ifnarch x86_64 %{ix86} +%define build_qemutrad 0 %define build_ovmf 0 %endif +%if ! %build_qemutrad +%define build_stubdom 0 +%endif # Build with xen hypervisor unless rpmbuild was run with --without hyp %define build_hyp %{?_without_hyp: 0} %{?!_without_hyp: 1} # build xsm support unless rpmbuild was run with --without xsm @@ -36,9 +42,19 @@ # --without efi %define build_efi %{?_without_efi: 0} %{?!_without_efi: 1} # xen only supports efi boot images on x86_64 or aarch64 -%ifnarch x86_64 aarch64 +# i686 builds a x86_64 hypervisor so add that as well +%ifnarch x86_64 aarch64 %{ix86} %define build_efi 0 %endif +%if %build_efi && "%dist" < ".fc26" +%ifarch x86_64 +%define efiming 1 +%else +%define efiming 0 +%endif +%else +%define efiming 0 +%endif %if "%dist" >= ".fc20" %define with_systemd_presets 1 %else @@ -46,16 +62,15 @@ %endif # Hypervisor ABI -%define hv_abi 4.21 +%define hv_abi 4.11 Summary: Xen is a virtual machine monitor Name: xen -Version: 4.21.1 -Release: 10%{?dist} -# Automatically converted from old format: GPLv2+ and LGPLv2+ and BSD - review is highly recommended. -License: GPL-2.0-or-later AND LicenseRef-Callaway-LGPLv2+ AND LicenseRef-Callaway-BSD +Version: 4.11.2 +Release: 2%{?dist} +License: GPLv2+ and LGPLv2+ and BSD URL: http://xen.org/ -Source0: https://downloads.xenproject.org/release/xen/%{version}/xen-%{version}.tar.xz +Source0: https://downloads.xenproject.org/release/xen/%{version}/xen-%{version}.tar.gz Source2: %{name}.logrotate # used by stubdoms Source10: lwip-1.3.0.tar.gz @@ -66,60 +81,81 @@ Source14: grub-0.97.tar.gz Source15: polarssl-1.1.4-gpl.tgz # .config file for xen hypervisor Source21: xen.hypervisor.config -# mini-os xen-RELEASE-4.21.0 with .git and .gitignore stripped -Source22: mini-os-4.21.0.tar.xz -Patch1: xen.fedora.systemd.patch -Patch2: xen.ocaml.selinux.fix.patch -Patch3: xen.canonicalize.patch -Patch4: droplibvirtconflict.patch -Patch5: xen.gcc9.fixes.patch -Patch6: xen.gcc11.fixes.patch -Patch7: xen.gcc12.fixes.patch -Patch8: xen.efi.build.patch -Patch9: xen.python3.12.patch -Patch11: xen.json.nocpuid.patch -Patch12: xsa483.patch -Patch13: xsa484.patch -Patch14: xsa486.patch -Patch15: xen.git-90b20547b756a5cf9b0fec9fb0de5b361e8bf4c3.patch -Patch16: xsa490-4.21.patch -Patch17: xsa491-4.21.patch -Patch18: xsa492-4.21-01.patch -Patch19: xsa492-4.21-02.patch -Patch20: xsa492-4.21-03.patch -Patch21: xsa492-4.21-04.patch -Patch22: xsa492-4.21-05.patch -Patch23: xsa492-4.21-06.patch -Patch24: xsa492-4.21-07.patch -Patch25: xsa492-4.21-08.patch -Patch26: xsa492-4.21-09.patch -Patch27: xsa492-4.21-10.patch -Patch28: xsa492-4.21-11.patch -Patch29: xsa492-4.21-12.patch -Patch30: xsa492-4.21-13.patch -Patch31: xsa492-4.21-14.patch -Patch32: xsa492-4.21-15.patch -Patch33: xsa492-4.21-16.patch -Patch34: xsa492-4.21-17.patch -Patch35: xsa492-4.21-18.patch -Patch36: xsa492-4.21-19.patch -Patch37: xsa492-4.21-20.patch -Patch38: xsa493-4.21-01.patch -Patch39: xsa493-4.21-02.patch -Patch40: xsa493-4.21-03.patch -Patch41: xsa493-4.21-04.patch -Patch42: xsa494-4.21.patch +Patch1: xen-net-disable-iptables-on-bridge.patch +Patch2: xen.use.fedora.ipxe.patch +Patch3: xen.fedora.efi.build.patch +Patch4: CVE-2014-0150.patch +Patch5: xen.fedora.systemd.patch +Patch6: xen.ocaml.selinux.fix.patch +Patch7: xen.fedora.crypt.patch +Patch8: qemu.trad.CVE-2015-6815.patch +Patch9: qemu.trad.CVE-2015-5279.patch +Patch10: qemu.trad.CVE-2015-5278.patch +Patch11: qemu.trad.CVE-2015-7295.patch +Patch12: qemu.trad.CVE-2015-8345.patch +Patch13: qemu.trad.CVE-2015-7512.patch +Patch14: qemu.trad.CVE-2015-8504.patch +Patch15: qemu.trad.CVE-2016-1714.patch +Patch16: qemu.trad.CVE-2016-1981.patch +Patch17: qemu.trad.CVE-2016-2841.patch +Patch18: qemu.trad.CVE-2016-2538.patch +Patch19: qemu.trad.CVE-2016-2857.patch +Patch20: qemu.trad.CVE-2016-4001.patch +Patch21: qemu.trad.CVE-2016-4002.patch +Patch22: qemu.trad.CVE-2016-4439.patch +Patch23: qemu.trad.CVE-2016-4441.patch +Patch24: qemu.trad.CVE-2016-5238.patch +Patch25: qemu.trad.CVE-2016-5338.patch +Patch27: qemu.trad.CVE-2016-6351.patch +Patch28: xen.glibcfix.patch +Patch29: qemu.trad.CVE-2016-8669.patch +Patch30: qemu.trad.CVE-2016-8910.patch +Patch31: qemu.trad.bug1399055.patch +Patch32: qemu.trad.CVE-2016-9776.patch +Patch33: xen.gcc7.fix.patch +Patch34: xen.canonicalize.patch +Patch35: qemu.trad.CVE-2017-6505.patch +Patch36: qemu.trad.CVE-2017-7718.patch +Patch37: droplibvirtconflict.patch +Patch38: qemu.trad.CVE-2017-8309.patch +Patch39: qemu.trad.CVE-2017-9330.patch +Patch40: xen.gcc8.temp.fix.patch +Patch41: xen.drop.brctl.patch +Patch42: xen.stubdom.build.patch +Patch44: xen.vwprintw.fix.patch +Patch45: xen.python.env.patch +Patch46: xen.gcc9.fixes.patch +Patch47: xsa296.patch +Patch48: xsa298-4.11.patch +Patch49: xsa299-4.11-0001-x86-mm-L1TF-checks-don-t-leave-a-partial-entry.patch +Patch50: xsa301-4.11-1.patch +Patch51: xsa301-4.11-2.patch +Patch52: xsa301-4.11-3.patch +Patch53: xsa302-4.11-0001-IOMMU-add-missing-HVM-check.patch +Patch54: xsa302-4.11-0002-passthrough-quarantine-PCI-devices.patch +Patch55: xsa303-0001-xen-arm32-entry-Split-__DEFINE_ENTRY_TRAP-in-two.patch +Patch56: xsa303-0002-xen-arm32-entry-Fold-the-macro-SAVE_ALL-in-the-macro.patch +Patch57: xsa303-0003-xen-arm32-Don-t-blindly-unmask-interrupts-on-trap-wi.patch +Patch58: xsa303-0004-xen-arm64-Don-t-blindly-unmask-interrupts-on-trap-wi.patch +%if %build_qemutrad +BuildRequires: libidn-devel zlib-devel SDL-devel curl-devel +BuildRequires: libX11-devel gtk2-devel libaio-devel # build using Fedora seabios and ipxe packages for roms BuildRequires: seabios-bin ipxe-roms-qemu %ifarch %{ix86} x86_64 # for the VMX "bios" BuildRequires: dev86 %endif -BuildRequires: python3-devel ncurses-devel python3-setuptools +%endif +BuildRequires: python2-devel ncurses-devel BuildRequires: perl-interpreter perl-generators +%ifarch %{ix86} x86_64 +# so that x86_64 builds pick up glibc32 correctly +BuildRequires: /usr/include/gnu/stubs-32.h +%endif BuildRequires: gettext BuildRequires: gnutls-devel BuildRequires: openssl-devel @@ -130,13 +166,11 @@ BuildRequires: libuuid-devel # iasl needed to build hvmloader BuildRequires: acpica-tools # modern compressed kernels -BuildRequires: bzip2-devel xz-devel libzstd-devel +BuildRequires: bzip2-devel xz-devel # libfsimage BuildRequires: e2fsprogs-devel -# tools now require wget -BuildRequires: wget -# use json-c instead of yajl -BuildRequires: json-c-devel +# tools now require yajl and wget +BuildRequires: yajl-devel wget # remus support now needs libnl3 BuildRequires: libnl3-devel %if %with_xsm @@ -147,32 +181,36 @@ BuildRequires: checkpolicy m4 # cross compiler for building 64-bit hypervisor on ix86 BuildRequires: gcc-x86_64-linux-gnu %endif -BuildRequires: gcc make +BuildRequires: gcc Requires: iproute -Requires: python3-lxml +Requires: python2-lxml Requires: xen-runtime = %{version}-%{release} # Not strictly a dependency, but kpartx is by far the most useful tool right # now for accessing domU data from within a dom0 so bring it in when the user # installs xen. Requires: kpartx -ExclusiveArch: x86_64 aarch64 +ExclusiveArch: %{ix86} x86_64 armv7hl aarch64 +#ExclusiveArch: %#{ix86} x86_64 ia64 noarch %if %with_ocaml BuildRequires: ocaml, ocaml-findlib -BuildRequires: perl(Data::Dumper) +%endif +# efi image needs an ld that has -mi386pep option +%if %efiming +BuildRequires: mingw64-binutils %endif %if %with_systemd_presets Requires(post): systemd Requires(preun): systemd +Requires(postun): systemd BuildRequires: systemd %endif BuildRequires: systemd-devel -%ifarch aarch64 +%ifarch armv7hl aarch64 BuildRequires: libfdt-devel %endif -%if %build_hyp -BuildRequires: bison flex +%if %build_ovmf +BuildRequires: edk2-ovmf %endif -BuildRequires: hostname %description This package contains the XenD daemon and xm command line @@ -196,15 +234,6 @@ Requires: /usr/bin/qemu-img # Ensure we at least have a suitable kernel installed, though we can't # force user to actually boot it. Requires: xen-hypervisor-abi = %{hv_abi} -# perl is used in /etc/xen/scripts/locking.sh -Recommends: perl -%ifnarch aarch64 -# use /usr/bin/qemu-system-i386 in Fedora instead of qemu-xen -Recommends: qemu-system-x86-core -%endif -%if %build_ovmf -Recommends: edk2-ovmf-xen -%endif %description runtime This package contains the runtime programs and daemons which @@ -215,14 +244,6 @@ form the core Xen userspace environment. Summary: Libraries for Xen tools Provides: xen-hypervisor-abi = %{hv_abi} Requires: xen-licenses -%if %build_hyp -%ifarch %{ix86} -Recommends: grub2-pc-modules -%endif -%ifarch x86_64 -Recommends: grub2-pc-modules grub2-efi-x64-modules -%endif -%endif %description hypervisor This package contains the Xen hypervisor @@ -234,8 +255,14 @@ Summary: Xen documentation BuildArch: noarch Requires: xen-licenses # for the docs -BuildRequires: perl(Pod::Man) perl(Pod::Text) perl(File::Find) -BuildRequires: transfig pandoc perl(Pod::Html) +%if "%dist" >= ".fc18" +BuildRequires: texlive-times texlive-courier texlive-helvetic texlive-ntgclass +%endif +BuildRequires: transfig texi2html ghostscript texlive-latex +BuildRequires: perl(Pod::Man) perl(Pod::Text) texinfo graphviz +# optional requires for more documentation +#BuildRequires: pandoc discount +BuildRequires: discount %description doc This package contains the Xen documentation. @@ -279,112 +306,138 @@ This package contains libraries for developing ocaml tools to manage Xen virtual machines. %endif -%package test -Summary: internal xen tests -%description test -This package contains files used in testing the xen builds %prep %setup -q -%patch 1 -p1 -%patch 2 -p1 -%patch 3 -p1 -%patch 4 -p1 -%patch 5 -p1 -%patch 6 -p1 -%patch 7 -p1 -%patch 8 -p1 -%patch 9 -p1 -%patch 11 -p1 -%patch 12 -p1 -%patch 13 -p1 -%patch 14 -p1 -%patch 15 -p1 -%patch 16 -p1 -%patch 17 -p1 -%patch 18 -p1 -%patch 19 -p1 -%patch 20 -p1 -%patch 21 -p1 -%patch 22 -p1 -%patch 23 -p1 -%patch 24 -p1 -%patch 25 -p1 -%patch 26 -p1 -%patch 27 -p1 -%patch 28 -p1 -%patch 29 -p1 -%patch 30 -p1 -%patch 31 -p1 -%patch 32 -p1 -%patch 33 -p1 -%patch 34 -p1 -%patch 35 -p1 -%patch 36 -p1 -%patch 37 -p1 -%patch 38 -p1 -%patch 39 -p1 -%patch 40 -p1 -%patch 41 -p1 -%patch 42 -p1 +%patch1 -p1 +%patch4 -p1 +%patch5 -p1 +%patch6 -p1 +%patch7 -p1 +%patch8 -p1 +%patch9 -p1 +%patch10 -p1 +%patch11 -p1 +%patch12 -p1 +%patch13 -p1 +%patch14 -p1 +%patch15 -p1 +%patch16 -p1 +%patch17 -p1 +%patch18 -p1 +%patch19 -p1 +%patch20 -p1 +%patch21 -p1 +%patch22 -p1 +%patch23 -p1 +%patch24 -p1 +%patch25 -p1 +%patch28 -p1 +%patch33 -p1 +%patch34 -p1 +%patch37 -p1 +%patch2 -p1 +%patch3 -p1 +%patch40 -p1 +%patch41 -p1 +%patch42 -p1 +%patch44 -p1 +%patch45 -p1 +%patch46 -p1 +%patch47 -p1 +%patch48 -p1 +%patch49 -p1 +%patch50 -p1 +%patch51 -p1 +%patch52 -p1 +%ifarch %{ix86} x86_64 +%patch53 -p1 +%patch54 -p1 +%endif +%patch55 -p1 +%patch56 -p1 +%patch57 -p1 +%patch58 -p1 + +# qemu-xen-traditional patches +pushd tools/qemu-xen-traditional +%patch27 -p1 +%patch29 -p1 +%patch30 -p1 +%patch31 -p1 +%patch32 -p1 +%patch35 -p1 +%patch36 -p1 +%patch38 -p1 +%patch39 -p1 +popd + +# qemu-xen patches +pushd tools/qemu-xen +popd # stubdom sources cp -v %{SOURCE10} %{SOURCE11} %{SOURCE12} %{SOURCE13} %{SOURCE14} %{SOURCE15} stubdom # copy xen hypervisor .config file to change settings cp -v %{SOURCE21} xen/.config -# mini-os is now separate file -mkdir extras -tar -C extras -xf %{SOURCE22} + %build -# This package calls binutils components directly and would need to pass -# in flags to enable the LTO plugins -# Disable LTO -%define _lto_cflags %{nil} - %if !%build_ocaml %define ocaml_flags OCAML_TOOLS=n %endif +%if %efiming +%define efi_flags LD_EFI=/usr/x86_64-w64-mingw32/bin/ld +%endif %if %build_efi mkdir -p dist/install/boot/efi/efi/fedora %endif %if %build_ocaml mkdir -p dist/install%{_libdir}/ocaml/stublibs %endif -export EXTRA_CFLAGS_XEN_TOOLS="$RPM_OPT_FLAGS -Wno-error=use-after-free $LDFLAGS" -export PYTHON="/usr/bin/python3" -export LDFLAGS_SAVE=`echo $LDFLAGS | sed -e 's/-Wl,//g' -e 's/,/ /g' -e 's? -specs=[-a-z/0-9]*??g'` -export CFLAGS_SAVE="$CFLAGS" -CONFIG_EXTRA="" -%if %build_ovmf -CONFIG_EXTRA="$CONFIG_EXTRA --with-system-ovmf=/usr/share/edk2/xen/OVMF.fd" -%endif -%ifarch aarch64 -CONFIG_EXTRA="$CONFIG_EXTRA --with-system-ipxe=/usr/share/ipxe/10ec8139.rom" -%endif %if %(test -f /usr/share/seabios/bios-256k.bin && echo 1|| echo 0) -CONFIG_EXTRA="$CONFIG_EXTRA --with-system-seabios=/usr/share/seabios/bios-256k.bin" +%define seabiosloc /usr/share/seabios/bios-256k.bin %else -CONFIG_EXTRA="$CONFIG_EXTRA --disable-seabios" +%define seabiosloc /usr/share/seabios/bios.bin %endif -%if %with_systemd_presets -CONFIG_EXTRA="$CONFIG_EXTRA --enable-systemd" -%endif -./configure --prefix=%{_prefix} --libdir=%{_libdir} --libexecdir=%{_libexecdir} --with-system-qemu=/usr/bin/qemu-system-i386 --with-linux-backend-modules="xen-evtchn xen-gntdev xen-gntalloc xen-blkback xen-netback xen-pciback xen-scsiback xen-acpi-processor" $CONFIG_EXTRA -unset CFLAGS CXXFLAGS FFLAGS LDFLAGS -export LDFLAGS="$LDFLAGS_SAVE" -export CFLAGS=`echo "$CFLAGS_SAVE -Wno-error=address" | sed -e 's/-specs=\/usr\/lib\/rpm\/redhat/redhat-annobin-cc1//g'` - +#export XEN_VENDORVERSION="-%{release}" +export EXTRA_CFLAGS_XEN_TOOLS="$RPM_OPT_FLAGS" +export EXTRA_CFLAGS_QEMU_TRADITIONAL="$RPM_OPT_FLAGS" +export EXTRA_CFLAGS_QEMU_XEN="$RPM_OPT_FLAGS" +export PYTHON="/usr/bin/python2" %if %build_hyp -%make_build prefix=/usr xen +%if %build_crosshyp +%define efi_flags LD_EFI=false +XEN_TARGET_ARCH=x86_64 make %{?_smp_mflags} %{?efi_flags} prefix=/usr xen CC="/usr/bin/x86_64-linux-gnu-gcc `echo $RPM_OPT_FLAGS | sed -e 's/-m32//g' -e 's/-march=i686//g' -e 's/-mtune=atom//g' -e 's/-specs=\/usr\/lib\/rpm\/redhat\/redhat-annobin-cc1//g' -e 's/-fstack-clash-protection//g' -e 's/-mcet//g' -e 's/-fcf-protection//g'`" +%else +%ifarch armv7hl +make %{?_smp_mflags} %{?efi_flags} prefix=/usr xen CC="gcc `echo $RPM_OPT_FLAGS | sed -e 's/-mfloat-abi=hard//g' -e 's/-march=armv7-a//g'`" +%else +%ifarch aarch64 +make %{?_smp_mflags} %{?efi_flags} prefix=/usr xen CC="gcc $RPM_OPT_FLAGS" +%else +make %{?_smp_mflags} %{?efi_flags} prefix=/usr xen CC="gcc `echo $RPM_OPT_FLAGS | sed -e 's/-specs=\/usr\/lib\/rpm\/redhat\/redhat-annobin-cc1//g' -e 's/-fcf-protection//g'`" %endif -unset CFLAGS CXXFLAGS FFLAGS LDFLAGS - -%make_build %{?ocaml_flags} prefix=/usr tools +%endif +%endif +%endif +%if ! %build_qemutrad +CONFIG_EXTRA="--disable-qemu-traditional" +%else +CONFIG_EXTRA="" +%endif +%if %build_ovmf +CONFIG_EXTRA="$CONFIG_EXTRA --with-system-ovmf=%{_libexecdir}/%{name}/boot/ovmf.bin" +%endif +./configure --prefix=%{_prefix} --libdir=%{_libdir} --libexecdir=%{_libexecdir} --with-system-seabios=%{seabiosloc} --with-system-qemu=/usr/bin/qemu-system-i386 --with-linux-backend-modules="xen-evtchn xen-gntdev xen-gntalloc xen-blkback xen-netback xen-pciback xen-scsiback xen-acpi-processor" $CONFIG_EXTRA +make %{?_smp_mflags} %{?ocaml_flags} prefix=/usr tools %if %build_docs make prefix=/usr docs %endif export RPM_OPT_FLAGS_RED=`echo $RPM_OPT_FLAGS | sed -e 's/-m64//g' -e 's/--param=ssp-buffer-size=4//g' -e's/-fstack-protector-strong//'` +%ifarch %{ix86} +export EXTRA_CFLAGS_XEN_TOOLS="$RPM_OPT_FLAGS_RED" +%endif %if %build_stubdom %ifnarch armv7hl aarch64 make mini-os-dir @@ -392,7 +445,7 @@ make -C stubdom build %endif %ifarch x86_64 export EXTRA_CFLAGS_XEN_TOOLS="$RPM_OPT_FLAGS_RED" -XEN_TARGET_ARCH=x86_32 make -C stubdom pv-grub-if-enabled +XEN_TARGET_ARCH=x86_32 make -C stubdom pv-grub %endif %endif @@ -427,18 +480,30 @@ find %{buildroot} -print | xargs ls -ld | sed -e 's|.*%{buildroot}||' > f1.list rm -rf %{buildroot}/usr/*-xen-elf # hypervisor symlinks -rm -rf %{buildroot}/boot/xen-%{hv_abi}.gz +rm -rf %{buildroot}/boot/xen-4.0.gz rm -rf %{buildroot}/boot/xen-4.gz -rm -rf %{buildroot}/boot/xen.gz %if !%build_hyp rm -rf %{buildroot}/boot %endif # silly doc dir fun rm -fr %{buildroot}%{_datadir}/doc/xen +rm -rf %{buildroot}%{_datadir}/doc/qemu # Pointless helper -rm -f %{buildroot}%{_bindir}/xen-python-path +rm -f %{buildroot}%{_sbindir}/xen-python-path + +# qemu stuff (unused or available from upstream) +rm -rf %{buildroot}/usr/share/xen/man +rm -rf %{buildroot}/usr/bin/qemu-*-xen +ln -s qemu-img %{buildroot}/%{_bindir}/qemu-img-xen +ln -s qemu-img %{buildroot}/%{_bindir}/qemu-nbd-xen +for file in bios.bin openbios-sparc32 openbios-sparc64 ppc_rom.bin \ + pxe-e1000.bin pxe-ne2k_pci.bin pxe-pcnet.bin pxe-rtl8139.bin \ + vgabios.bin vgabios-cirrus.bin video.x openbios-ppc bamboo.dtb +do + rm -f %{buildroot}/%{_datadir}/xen/qemu/$file +done # README's not intended for end users rm -f %{buildroot}/%{_sysconfdir}/xen/README* @@ -451,17 +516,20 @@ rm -rf %{buildroot}/%{_libdir}/*.a %if %build_efi # clean up extra efi files -rm -f %{buildroot}/%{_libdir}/efi/xen-%{hv_abi}.efi -rm -f %{buildroot}/%{_libdir}/efi/xen-4.efi -rm -f %{buildroot}/%{_libdir}/efi/xen.efi -cp -p %{buildroot}/%{_libdir}/efi/xen-%{version}{,.notstripped}.efi -strip -s %{buildroot}/%{_libdir}/efi/xen-%{version}.efi +rm -rf %{buildroot}/%{_libdir}/efi +%ifarch %{ix86} +rm -rf %{buildroot}/usr/lib64/efi +%endif %endif %if ! %build_ocaml rm -rf %{buildroot}/%{_unitdir}/oxenstored.service %endif +%if %build_ovmf +cat /usr/share/OVMF/OVMF_{VARS,CODE}.fd >%{buildroot}%{_libexecdir}/%{name}/boot/ovmf.bin +%endif + ############ fixup files in /etc ############ # logrotate @@ -469,12 +537,10 @@ mkdir -p %{buildroot}%{_sysconfdir}/logrotate.d/ install -m 644 %{SOURCE2} %{buildroot}%{_sysconfdir}/logrotate.d/%{name} # init scripts -%define initdloc %(test -d /etc/rc.d/init.d/ && echo rc.d/init.d || echo init.d ) - -rm %{buildroot}%{_sysconfdir}/%{initdloc}/xen-watchdog -rm %{buildroot}%{_sysconfdir}/%{initdloc}/xencommons -rm %{buildroot}%{_sysconfdir}/%{initdloc}/xendomains -rm %{buildroot}%{_sysconfdir}/%{initdloc}/xendriverdomain +rm %{buildroot}%{_sysconfdir}/rc.d/init.d/xen-watchdog +rm %{buildroot}%{_sysconfdir}/rc.d/init.d/xencommons +rm %{buildroot}%{_sysconfdir}/rc.d/init.d/xendomains +rm %{buildroot}%{_sysconfdir}/rc.d/init.d/xendriverdomain ############ create dirs in /var ############ @@ -488,7 +554,7 @@ ln -s %{_libexecdir}/%{name} %{buildroot}/%{_libdir}/%{name} %endif ############ create symlink to qemu-system-i386 in /usr/bin ############ -ln -s ../../../bin/qemu-system-i386 %{buildroot}/%{_libexecdir}/%{name}/bin/qemu-system-i386 +ln -s /usr/bin/qemu-system-i386 %{buildroot}/%{_libexecdir}/%{name}/bin/qemu-system-i386 ############ debug packaging: list files ############ @@ -506,17 +572,6 @@ find . -path licensedir -prune -o -path stubdom/ioemu -prune -o \ install -m 644 $file licensedir/$file done -############ move sbin files to bin - -mv %{buildroot}/usr/sbin/* %{buildroot}/usr/bin/ - -############ remove xen*.efi.elf files to avoid debuginfo failure - -%ifarch x86_64 -rm dist/install/usr/lib/debug/xen-*.efi.elf -rm %{buildroot}/usr/lib/debug/xen-*.efi.elf -%endif - ############ all done now ############ %post @@ -537,6 +592,11 @@ if [ $1 == 0 ]; then fi %endif +%if %with_systemd_presets +%postun +%systemd_postun +%endif + %post runtime %if %with_systemd_presets %systemd_post xenstored.service xenconsoled.service @@ -557,8 +617,13 @@ if [ $1 == 0 ]; then fi %endif +%if %with_systemd_presets +%postun runtime +%systemd_postun +%endif + %posttrans runtime -if [ ! -L /usr/lib/xen -a -d /usr/lib/xen ] && [ -z "$(ls -A /usr/lib/xen)" ]; then +if [ ! -L /usr/lib/xen -a -d /usr/lib/xen -a -z "$(ls -A /usr/lib/xen)" ]; then rmdir /usr/lib/xen fi if [ ! -e /usr/lib/xen ]; then @@ -569,49 +634,71 @@ fi %if %build_hyp %post hypervisor -do_it() { - DIR=$1 - TARGET=$2 - if [ -d $DIR ]; then - if [ ! -d $TARGET ]; then - mkdir $TARGET - fi - for m in relocator.mod multiboot2.mod elf.mod; do - if [ -f $DIR/$m ]; then - if [ ! -f $TARGET/$m ] || ! cmp -s $DIR/$m $TARGET/$m; then - cp -p $DIR/$m $TARGET/$m - fi - fi - done - fi -} if [ $1 == 1 -a -f /sbin/grub2-mkconfig ]; then - for f in /boot/grub2/grub.cfg; do - if [ -f $f ]; then - /sbin/grub2-mkconfig -o $f - sed -i -e '/insmod module2/d' $f - fi - done -fi -if [ -f /sbin/grub2-mkconfig ]; then if [ -f /boot/grub2/grub.cfg ]; then - DIR=/usr/lib/grub/i386-pc - TARGET=/boot/grub2/i386-pc - do_it $DIR $TARGET - DIR=/usr/lib/grub/x86_64-efi - TARGET=/boot/grub2/x86_64-efi - do_it $DIR $TARGET + /sbin/grub2-mkconfig -o /boot/grub2/grub.cfg + sed -i -e '/insmod module2/d' /boot/grub2/grub.cfg + if [ -d /usr/lib/grub/i386-pc ]; then + if [ ! -d /boot/grub2/i386-pc ]; then + mkdir /boot/grub2/i386-pc + fi + if [ -f /usr/lib/grub/i386-pc/relocator.mod -a ! -f /boot/grub2/i386-pc/relocator.mod ]; then + cp -p /usr/lib/grub/i386-pc/relocator.mod /boot/grub2/i386-pc/relocator.mod + fi + if [ -f /usr/lib/grub/i386-pc/multiboot2.mod -a ! -f /boot/grub2/i386-pc/multiboot2.mod ]; then + cp -p /usr/lib/grub/i386-pc/multiboot2.mod /boot/grub2/i386-pc/multiboot2.mod + fi + fi + fi + if [ -f /boot/efi/EFI/fedora/grub.cfg ]; then + /sbin/grub2-mkconfig -o /boot/efi/EFI/fedora/grub.cfg + sed -i -e '/insmod module2/d' /boot/efi/EFI/fedora/grub.cfg + if [ -d /usr/lib/grub/x86_64-efi ]; then + if [ ! -d /boot/efi/EFI/fedora/x86_64-efi ]; then + mkdir /boot/efi/EFI/fedora/x86_64-efi + fi + if [ -f /usr/lib/grub/x86_64-efi/relocator.mod -a ! -f /boot/efi/EFI/fedora/x86_64-efi/relocator.mod ]; then + cp -p /usr/lib/grub/x86_64-efi/relocator.mod /boot/efi/EFI/fedora/x86_64-efi/relocator.mod + fi + if [ -f /usr/lib/grub/x86_64-efi/multiboot2.mod -a ! -f /boot/efi/EFI/fedora/x86_64-efi/multiboot2.mod ]; then + cp -p /usr/lib/grub/x86_64-efi/multiboot2.mod /boot/efi/EFI/fedora/x86_64-efi/multiboot2.mod + fi + fi fi fi %postun hypervisor if [ -f /sbin/grub2-mkconfig ]; then - for f in /boot/grub2/grub.cfg; do - if [ -f $f ]; then - /sbin/grub2-mkconfig -o $f - sed -i -e '/insmod module2/d' $f + if [ -f /boot/grub2/grub.cfg ]; then + /sbin/grub2-mkconfig -o /boot/grub2/grub.cfg + sed -i -e '/insmod module2/d' /boot/grub2/grub.cfg + if [ -d /usr/lib/grub/i386-pc -a $1 == 1 ]; then + if [ ! -d /boot/grub2/i386-pc ]; then + mkdir /boot/grub2/i386-pc + fi + if [ -f /usr/lib/grub/i386-pc/relocator.mod -a ! -f /boot/grub2/i386-pc/relocator.mod ]; then + cp -p /usr/lib/grub/i386-pc/relocator.mod /boot/grub2/i386-pc/relocator.mod + fi + if [ -f /usr/lib/grub/i386-pc/multiboot2.mod -a ! -f /boot/grub2/i386-pc/multiboot2.mod ]; then + cp -p /usr/lib/grub/i386-pc/multiboot2.mod /boot/grub2/i386-pc/multiboot2.mod + fi fi - done + fi + if [ -f /boot/efi/EFI/fedora/grub.cfg ]; then + /sbin/grub2-mkconfig -o /boot/efi/EFI/fedora/grub.cfg + sed -i -e '/insmod module2/d' /boot/efi/EFI/fedora/grub.cfg + if [ -d /usr/lib/grub/x86_64-efi -a $1 == 1 ]; then + if [ ! -d /boot/efi/EFI/fedora/x86_64-efi ]; then + mkdir /boot/efi/EFI/fedora/x86_64-efi + fi + if [ -f /usr/lib/grub/x86_64-efi/relocator.mod -a ! -f /boot/efi/EFI/fedora/x86_64-efi/relocator.mod ]; then + cp -p /usr/lib/grub/x86_64-efi/relocator.mod /boot/efi/EFI/fedora/x86_64-efi/relocator.mod + fi + if [ -f /usr/lib/grub/x86_64-efi/multiboot2.mod -a ! -f /boot/efi/EFI/fedora/x86_64-efi/multiboot2.mod ]; then + cp -p /usr/lib/grub/x86_64-efi/multiboot2.mod /boot/efi/EFI/fedora/x86_64-efi/multiboot2.mod + fi + fi + fi fi %endif @@ -633,14 +720,20 @@ if [ $1 == 0 ]; then /bin/systemctl disable oxenstored.service fi %endif + +%if %with_systemd_presets +%postun ocaml +%systemd_postun +%endif %endif # Base package only contains XenD/xm python stuff #files -f xen-xm.lang %files %doc COPYING README -%{python3_sitearch}/%{name} -%{python3_sitearch}/xen-*.egg-info +%{_bindir}/xencons +%{python2_sitearch}/%{name} +%{python2_sitearch}/xen-*.egg-info # Guest autostart links %dir %attr(0700,root,root) %{_sysconfdir}/%{name}/auto @@ -650,34 +743,8 @@ fi %{_unitdir}/xendomains.service %files libs -%{_libdir}/libxencall.so.1 -%{_libdir}/libxencall.so.1.3 -%{_libdir}/libxenctrl.so.4.* -%{_libdir}/libxendevicemodel.so.1 -%{_libdir}/libxendevicemodel.so.1.4 -%{_libdir}/libxenevtchn.so.1 -%{_libdir}/libxenevtchn.so.1.2 -%{_libdir}/libxenforeignmemory.so.1 -%{_libdir}/libxenforeignmemory.so.1.4 -%{_libdir}/libxenfsimage.so.4.* -%{_libdir}/libxengnttab.so.1 -%{_libdir}/libxengnttab.so.1.2 -%{_libdir}/libxenguest.so.4.* -%{_libdir}/libxenlight.so.4.* -%{_libdir}/libxenstat.so.4.* -%{_libdir}/libxenstore.so.4 -%{_libdir}/libxenstore.so.4.1 -%{_libdir}/libxentoolcore.so.1 -%{_libdir}/libxentoolcore.so.1.0 -%{_libdir}/libxentoollog.so.1 -%{_libdir}/libxentoollog.so.1.0 -%{_libdir}/libxenvchan.so.4.* -%{_libdir}/libxlutil.so.4.* -%{_libdir}/xenfsimage -%{_libdir}/libxenhypfs.so.1 -%{_libdir}/libxenhypfs.so.1.0 -%{_libdir}/libxenmanage.so.1 -%{_libdir}/libxenmanage.so.1.0 +%{_libdir}/*.so.* +%{_libdir}/fs # All runtime stuff except for XenD/xm python stuff %files runtime @@ -687,16 +754,16 @@ fi %dir %attr(0700,root,root) %{_sysconfdir}/%{name}/scripts/ %config %attr(0700,root,root) %{_sysconfdir}/%{name}/scripts/* -%{_sysconfdir}/bash_completion.d/xl +%{_sysconfdir}/bash_completion.d/xl.sh %{_unitdir}/proc-xen.mount +%{_unitdir}/var-lib-xenstored.mount %{_unitdir}/xenstored.service %{_unitdir}/xenconsoled.service %{_unitdir}/xen-watchdog.service %{_unitdir}/xen-qemu-dom0-disk-backend.service %{_unitdir}/xendriverdomain.service -%{_modulesloaddir}/xen.conf -%{_systemd_util_dir}/system-sleep/xen-watchdog-sleep.sh +/usr/lib/modules-load.d/xen.conf %config(noreplace) %{_sysconfdir}/sysconfig/xencommons %config(noreplace) %{_sysconfdir}/xen/xl.conf @@ -710,10 +777,19 @@ fi %dir %{_libexecdir}/%{name} %dir %{_libexecdir}/%{name}/bin %attr(0700,root,root) %{_libexecdir}/%{name}/bin/* +# QEMU runtime files +%if %build_qemutrad +%ifnarch armv7hl aarch64 +%dir %{_datadir}/%{name}/qemu +%dir %{_datadir}/%{name}/qemu/keymaps +%{_datadir}/%{name}/qemu/keymaps/* +%endif +%endif # man pages %if %build_docs %{_mandir}/man1/xentop.1* +%{_mandir}/man1/xentrace_format.1* %{_mandir}/man8/xentrace.8* %{_mandir}/man1/xl.1* %{_mandir}/man5/xl.cfg.5* @@ -728,25 +804,27 @@ fi %{_mandir}/man5/xl-network-configuration.5.gz %{_mandir}/man7/xen-pv-channel.7.gz %{_mandir}/man7/xl-numa-placement.7.gz -%{_mandir}/man1/xenhypfs.1.gz -%{_mandir}/man7/xen-vbd-interface.7.gz -%{_mandir}/man5/xl-pci-configuration.5.gz -%{_mandir}/man8/xenwatchdogd.8.gz %endif -%{python3_sitearch}/xenfsimage*.so -%{python3_sitearch}/grub -%{python3_sitearch}/pygrub-*.egg-info +%{python2_sitearch}/fsimage.so +%{python2_sitearch}/grub +%{python2_sitearch}/pygrub-*.egg-info # The firmware -%ifarch x86_64 +%ifarch %{ix86} x86_64 %dir %{_libexecdir}/%{name}/boot %{_libexecdir}/xen/boot/hvmloader +%ifnarch %{ix86} %{_libexecdir}/%{name}/boot/xen-shim /usr/lib/debug%{_libexecdir}/xen/boot/xen-shim-syms +%endif +%if %build_ovmf +%{_libexecdir}/xen/boot/ovmf.bin +%endif %if %build_stubdom +%{_libexecdir}/xen/boot/ioemu-stubdom.gz %{_libexecdir}/xen/boot/xenstore-stubdom.gz -%{_libexecdir}/xen/boot/xenstorepvh-stubdom.gz +%{_libexecdir}/xen/boot/pv-grub*.gz %endif %endif %if "%{_libdir}" != "/usr/lib" @@ -757,65 +835,66 @@ fi %dir %{_localstatedir}/lib/%{name} %dir %{_localstatedir}/lib/%{name}/dump %dir %{_localstatedir}/lib/%{name}/images +# Xenstore persistent state +%dir %{_localstatedir}/lib/xenstored # Xenstore runtime state %ghost %{_localstatedir}/run/xenstored # All xenstore CLI tools +%{_bindir}/qemu-*-xen %{_bindir}/xenstore %{_bindir}/xenstore-* +%{_bindir}/pygrub +%{_bindir}/xentrace* #%#{_bindir}/remus # XSM -%{_bindir}/flask-* +%{_sbindir}/flask-* # Misc stuff -%ifnarch aarch64 +%ifnarch armv7hl aarch64 %{_bindir}/xen-detect %endif %{_bindir}/xencov_split -%ifnarch aarch64 -%{_bindir}/gdbsx -%{_bindir}/xen-kdd +%ifnarch armv7hl aarch64 +%{_sbindir}/gdbsx +%{_sbindir}/kdd %endif -%ifnarch aarch64 -%{_bindir}/xen-hptool -%{_bindir}/xen-hvmcrash -%{_bindir}/xen-hvmctx +%{_sbindir}/xen-bugtool +%ifnarch armv7hl aarch64 +%{_sbindir}/xen-hptool +%{_sbindir}/xen-hvmcrash +%{_sbindir}/xen-hvmctx %endif -%{_bindir}/xenconsoled -%{_bindir}/xenlockprof -%{_bindir}/xenmon -%{_bindir}/xentop -%{_bindir}/xentrace_setmask -%{_bindir}/xenbaked -%{_bindir}/xenstored -%{_bindir}/xenpm -%{_bindir}/xenpmd -%{_bindir}/xenperf -%{_bindir}/xenwatchdogd -%{_bindir}/xl -%ifnarch aarch64 -%{_bindir}/xen-lowmemd +%{_sbindir}/xen-tmem-list-parse +%{_sbindir}/xenconsoled +%{_sbindir}/xenlockprof +%{_sbindir}/xenmon.py* +%{_sbindir}/xentop +%{_sbindir}/xentrace_setmask +%{_sbindir}/xenbaked +%{_sbindir}/xenstored +%{_sbindir}/xenpm +%{_sbindir}/xenpmd +%{_sbindir}/xenperf +%{_sbindir}/xenwatchdogd +%{_sbindir}/xl +%ifnarch armv7hl aarch64 +%{_sbindir}/xen-lowmemd %endif -%{_bindir}/xencov -%ifnarch aarch64 -%{_bindir}/xen-mfndump +%{_sbindir}/xen-ringwatch +%{_sbindir}/xencov +%ifnarch armv7hl aarch64 +%{_sbindir}/xen-mfndump %endif +%ifnarch armv7hl aarch64 %{_bindir}/xenalyze -%{_bindir}/xentrace -%{_bindir}/xentrace_setsize -%ifnarch aarch64 +%endif +%{_sbindir}/xentrace +%{_sbindir}/xentrace_setsize +%ifnarch armv7hl aarch64 %{_bindir}/xen-cpuid %endif -%{_bindir}/xen-livepatch -%{_bindir}/xen-diag -%ifnarch armv7hl aarch64 -%{_bindir}/xen-ucode -%{_bindir}/xen-memshare -%{_bindir}/xen-mceinj -%{_bindir}/xen-vmtrace -%endif -%{_bindir}/vchan-socket-proxy -%{_bindir}/xenhypfs -%{_bindir}/xen-access +%{_sbindir}/xen-livepatch +%{_sbindir}/xen-diag # Xen logfiles %dir %attr(0700,root,root) %{_localstatedir}/log/xen @@ -824,8 +903,9 @@ fi %files hypervisor %if %build_hyp -%ifnarch aarch64 +%ifnarch armv7hl aarch64 /boot/xen-*.gz +/boot/xen.gz /boot/xen*.config %else /boot/xen* @@ -834,10 +914,10 @@ fi %dir %attr(0755,root,root) /boot/flask /boot/flask/xenpolicy* %endif -/usr/lib/debug/xen* -%endif %if %build_efi -%{_libdir}/efi/*.efi +/boot/efi/EFI/fedora/*.efi +%endif +/usr/lib/debug/xen* %endif %if %build_docs @@ -853,7 +933,7 @@ fi %dir %{_includedir}/xenstore-compat %{_includedir}/xenstore-compat/* %{_libdir}/*.so -%{_libdir}/pkgconfig/* +/usr/share/pkgconfig/* %files licenses %doc licensedir/* @@ -866,7 +946,7 @@ fi %exclude %{_libdir}/ocaml/xen*/*.cmx %{_libdir}/ocaml/stublibs/*.so %{_libdir}/ocaml/stublibs/*.so.owner -%{_bindir}/oxenstored +%{_sbindir}/oxenstored %config(noreplace) %{_sysconfdir}/xen/oxenstored.conf %{_unitdir}/oxenstored.service @@ -874,854 +954,47 @@ fi %{_libdir}/ocaml/xen*/*.a %{_libdir}/ocaml/xen*/*.cmxa %{_libdir}/ocaml/xen*/*.cmx -%{_libdir}/ocaml/xsd_glue/* -%{_libexecdir}/xen/ocaml/xsd_glue/xenctrl_plugin/domain_getinfo_v1.cmxs %endif -%files test -%{_libexecdir}/xen/tests/* - %changelog -* Wed Jul 22 2026 Python Maint - 4.21.1-10 -- Rebuilt for Python 3.15.0b4 ABI change - -* Fri Jul 17 2026 Fedora Release Engineering - 4.21.1-9 -- Rebuilt for https://fedoraproject.org/wiki/Fedora_45_Mass_Rebuild - -* Thu Jul 09 2026 Jerry James - 4.21.1-8 -- OCaml 5.5.0 rebuild - -* Thu Jun 18 2026 Yaakov Selkowitz - 4.21.1-7 -- Rebuilt for openssl 4.0 - -* Thu Jun 18 2026 Michael Young - 4.21.1-6 -- x86 HVM I/O port list traversal [XSA-491, CVE-2026-42487] -- domctl lock open to abuse [XSA-492, CVE-2026-42489, CVE-2026-42490] -- Arm: Completion of memory accesses not guaranteed by completion of a TLBI - [XSA-493, CVE-2025-10263] -- x86: mismatched mapcache metadata [XSA-494, CVE-2026-42488] - -* Sat Jun 13 2026 Yaakov Selkowitz - 4.21.1-5 -- Rebuilt for openssl 4.0 - -* Wed Jun 03 2026 Python Maint - 4.21.1-4 -- Rebuilt for Python 3.15 - -* Tue May 12 2026 Michael Young - 4.21.1-3 -- x86: CPU Opcode Cache corruption [XSA-490,CVE-2025-54518] - -* Tue Apr 28 2026 Michael Young - 4.21.1-2 -- oxenstored keeps quota related use counts across domain destruction - [XSA-483, CVE-2026-23556] -- Xenstored DoS via XS_RESET_WATCHES command [XSA-484, CVE-2026-23557] -- grant table v2 race in status page mapping [XSA-486, CVE-2026-23558] -- x86: Floating Point Divider State Sampling [XSA-488, CVE-2025-54505] - -* Thu Mar 26 2026 Michael Young - 4.21.1-1 -- update to xen 4.21.1 - remove patches now included or superceded upstream - -* Tue Mar 17 2026 Michael Young - 4.21.0-5 -- Use after free of paging structures in EPT [XSA-480, CVE-2026-23554] -- Xenstored DoS by unprivileged domain [XSA-481, CVE-2026-23555] - -* Fri Feb 20 2026 Richard W.M. Jones - 4.21.0-4 -- OCaml 5.4.1 rebuild - -* Wed Jan 28 2026 Michael Young - 4.21.0-3 - x86: buffer overrun with shadow paging + tracing [XSA-477, CVE-2025-58150] - x86: incomplete IBPB for vCPU isolation [XSA-479, CVE-2026-23553] - -* Sat Jan 17 2026 Fedora Release Engineering - 4.21.0-2 -- Rebuilt for https://fedoraproject.org/wiki/Fedora_44_Mass_Rebuild - -* Wed Jan 07 2026 Michael Young - 4.21.0-1 -- update to xen 4.21.0 - rebase mini-os - use .xz xen tarball instead of .gz - fix quotes around sed command - update libxenstore version - package libxenmanage and xen-watchdog-sleep.sh files - add a new package for test files - renumber patches - use json-c instead of yajl -- fix bug in xen code when using json-c -- fix code issues detected by gcc16 - -* Thu Nov 13 2025 Michael Young - 4.20.2-2.fc44 -- update to xen 4.20.2 - remove patches now included or superceded upstream - -* Sun Oct 26 2025 Michael Young - 4.20.1-9 -- teecr32_el1 and teehbr32_el1 support dropped in binutils 2.45.50-5.fc44 - -* Fri Oct 24 2025 Michael Young -- Incorrect removal of permissions on PCI device unplug [XSA-476, - CVE-2025-58149] - -* Tue Oct 21 2025 Michael Young -- x86: Incorrect input sanitisation in Viridian hypercalls [XSA-475, - CVE-2025-58147, CVE-2025-58148] - -* Wed Oct 15 2025 Richard W.M. Jones - 4.20.1-7 -- OCaml 5.4.0 rebuild - -* Fri Sep 19 2025 Python Maint - 4.20.1-6 -- Rebuilt for Python 3.14.0rc3 bytecode - -* Wed Sep 10 2025 Michael Young - 4.20.1-5 -- Mutiple vulnerabilities in the Viridian interface [XSA-472, - CVE-2025-27466, CVE-2025-58142, CVE-2025-58143] -- Arm issues with page refcounting [XSA-473, CVE-2025-58144, - CVE-2025-58145] - -* Tue Sep 02 2025 Michael Young - 4.20.1-4 -- tools/xl: don't crash on NULL command line - -* Fri Aug 15 2025 Python Maint - 4.20.1-3 -- Rebuilt for Python 3.14.0rc2 bytecode - -* Fri Jul 25 2025 Fedora Release Engineering - 4.20.1-2 -- Rebuilt for https://fedoraproject.org/wiki/Fedora_43_Mass_Rebuild - -* Sun Jul 13 2025 Michael Young - 4.20.1-1 -- update to xen 4.20.1 - remove old qemu code for spac file - remove armv7hl and ix86 code from spec file - update configuration in xen.hypervisor.config - minios is now a separate file - package extra ocaml files - unset -specs=/usr/lib/rpm/redhat/redhat-annobin-cc1 for hypervisor build - rebase xen.efi.build.patch - includes fixes for security vulnerabilites - x86: Incorrect stubs exception handling for flags recovery [XSA-470, - CVE-2025-27465] - x86: Transitive Scheduler Attacks [XSA-471, CVE-2024-36350, - CVE-2024-36357] - -* Fri Jul 11 2025 Jerry James - 4.19.2-6 -- Rebuild to fix OCaml dependencies - -* Mon Jun 02 2025 Python Maint - 4.19.2-5 -- Rebuilt for Python 3.14 - -* Mon May 12 2025 Michael Young - 4.19.2-4 -- x86: Indirect Target Selection [XSA-469, CVE-2024-28956] - -* Mon Apr 07 2025 Michael Young - 4.19.2-2 -- update to xen-4.19.2 - remove patches now included or superceded upstream - remove xen*.efi.elf files to avoid debuginfo failure - -* Thu Feb 27 2025 Michael Young - 4.19.1-7 -- deadlock potential with VT-d and legacy PCI device pass-through - [XSA-467, CVE-2025-1713] - -* Thu Jan 23 2025 Michael Young - 4.19.1-6 -- adjust file locations now /usr/sbin is a symlink to /usr/bin -- remove debugedit fix as no longer needed - -* Sun Jan 19 2025 Fedora Release Engineering - 4.19.1-5 -- Rebuilt for https://fedoraproject.org/wiki/Fedora_42_Mass_Rebuild - -* Fri Jan 10 2025 Jerry James - 4.19.1-4 -- OCaml 5.3.0 rebuild for Fedora 42 - -* Thu Jan 09 2025 Michael Young - 4.19.1-3 -- work around debugedit bug to fix aarch64 builds - -* Sat Jan 04 2025 Andrea Perotti - 4.19.1-2 -- xen-hypervisor %post doesn't load all needed grub2 modules - (#2335558) - -* Thu Dec 05 2024 Michael Young - 4.19.1-1 -- update to xen-4.19.1 - remove patches now included or superceded upstream - -* Tue Nov 12 2024 Michael Young - 4.19.0-5 -- Deadlock in x86 HVM standard VGA handling [XSA-463, CVE-2024-45818] -- libxl leaks data to PVH guests via ACPI tables [XSA-464, CVE-2024-45819] -- additional patches so above applies cleanly - -* Tue Sep 24 2024 Michael Young - 4.19.0-4 -- x86: Deadlock in vlapic_error() [XSA-462, CVE-2024-45817] (#2314782) - -* Wed Sep 04 2024 Miroslav Suchý - 4.19.0-3 -- convert license to SPDX - -* Wed Aug 14 2024 Michael Young - 4.19.0-2 -- error handling in x86 IOMMU identity mapping [XSA-460, CVE-2024-31145] - (#2314784) -- PCI device pass-through with shared resources [XSA-461, CVE-2024-31146] - (#2314783) - -* Sat Aug 03 2024 Michael Young - 4.19.0-1 -- update to xen-4.19.0 - rebase xen.fedora.systemd.patch, xen.efi.build.patch - xen.ocaml5.fixes.patch and xen.gcc14.fixes.patch - remove patches now included or superceded upstream - now need to enable systemd explicitly - xentrace_format has gone, pygrub is now only in /usr/libexec/xen/bin/ - package xenwatchdogd.8.gz - use relative links for /usr/bin/qemu-system-i386 - -* Sat Jul 20 2024 Fedora Release Engineering - 4.18.2-5 -- Rebuilt for https://fedoraproject.org/wiki/Fedora_41_Mass_Rebuild - -* Tue Jul 16 2024 Michael Young - 4.18.2-4 -- double unlock in x86 guest IRQ handling [XSA-458, CVE-2024-31143] - (#2298690) - -* Fri Jun 07 2024 Python Maint - 4.18.2-3 -- Rebuilt for Python 3.13 - -* Mon Jun 03 2024 Michael Young - 4.18.2-2 -- x86: Native Branch History Injection [XSA-456 version 3, CVE-2024-2201] - -* Tue Apr 09 2024 Michael Young - 4.18.2-1 -- x86: Native Branch History Injection [XSA-456, CVE-2024-2201] -- update to xen 4.18.2, remove patches now included upstream - -* Tue Apr 09 2024 Michael Young - 4.18.1-2 -- x86 HVM hypercalls may trigger Xen bug check [XSA-454, CVE-2023-46842] -- x86: Incorrect logic for BTC/SRSO mitigations [XSA-455, CVE-2024-31142] - -* Wed Mar 20 2024 Michael Young - 4.18.1-1 -- update to xen-4.18.1 - rebase xen.gcc12.fixes.patch - remove patches now included or superceded upstream - -* Wed Mar 13 2024 Michael Young - 4.18.0-7 -- x86: Register File Data Sampling [XSA-452, CVE-2023-28746] -- GhostRace: Speculative Race Conditions [XSA-453, CVE-2024-2193] -- additional patches so above applies cleanly - -* Tue Feb 27 2024 Michael Young - 4.18.0-6 -- x86: shadow stack vs exceptions from emulation stubs - [XSA-451, - CVE-2023-46841] (#2266326) - -* Sun Feb 04 2024 Michael Young - 4.18.0-5 -- pci: phantom functions assigned to incorrect contexts [XSA-449, - CVE-2023-46839] -- VT-d: Failure to quarantine devices in !HVM build [XSA-450, - CVE-2023-46840] -- the glibc32 doesn't seem to add anything to the build so drop it - -* Sat Feb 03 2024 Michael Young - 4.18.0-4 -- build fixes for gcc14, replace stubs-32.h requirement with glibc32 - -* Sat Jan 27 2024 Fedora Release Engineering - 4.18.0-3 -- Rebuilt for https://fedoraproject.org/wiki/Fedora_40_Mass_Rebuild - -* Wed Dec 13 2023 Michael Young - 4.18.0-2 -- arm32: The cache may not be properly cleaned/invalidated (take two) - [XSA-447, CVE-2023-46837] -- rebuild for OCaml-5.1.1 - -* Wed Nov 29 2023 Michael Young - 4.18.0-1 -- update to xen-4.18.0 - rebase xen.canonicalize.patch and xen.ocaml5.fixes.patch - remove or adjust patches now included or superceded upstream -- xencons has been dropped - - -* Tue Nov 14 2023 Michael Young - 4.17.2-5 -- x86/AMD: mismatch in IOMMU quarantine page table levels [XSA-445, - CVE-2023-46835] -- x86: BTC/SRSO fixes not fully effective [XSA-446, CVE-2023-46836] - -* Tue Oct 10 2023 Michael Young - 4.17.2-4 -- xenstored: A transaction conflict can crash C Xenstored [XSA-440, - CVE-2023-34323] -- x86/AMD: missing IOMMU TLB flushing [XSA-442, CVE-2023-34326] -- Multiple vulnerabilities in libfsimage disk handling [XSA-443, - CVE-2023-34325] -- x86/AMD: Debug Mask handling [XSA-444, CVE-2023-34327, - CVE-2023-34328] - -* Sun Oct 08 2023 Michael Young - 4.17.2-3 -- rebuild (f40) for OCaml 5.1 - -* Tue Sep 26 2023 Michael Young - 4.17.2-2 -- arm32: The cache may not be properly cleaned/invalidated [XSA-437, - CVE-2023-34321] -- top-level shadow reference dropped too early for 64-bit PV guests - [XSA-438, CVE-2023-34322] -- x86/AMD: Divide speculative information leak [XSA-439, CVE-2023-20588] - -* Thu Aug 10 2023 Michael Young - 4.17.2-1 -- update to xen-4.17.2 which includes - x86/AMD: Speculative Return Stack Overflow [XSA-434, CVE-2023-20569] - x86/Intel: Gather Data Sampling [XSA-435, CVE-2022-40982] -- remove patches now included upstream - -* Tue Aug 01 2023 Michael Young - 4.17.1-9 -- arm: Guests can trigger a deadlock on Cortex-A77 [XSA-436, CVE-2023-34320] - (#2228238) - -* Mon Jul 31 2023 Michael Young - 4.17.1-8 -- bugfix for x86/AMD: Zenbleed [XSA-433, CVE-2023-20593] - -* Tue Jul 25 2023 Michael Young -- adjust OCaml patch condition so eln builds work - -* Mon Jul 24 2023 Michael Young - 4.17.1-7 -- x86/AMD: Zenbleed [XSA-433, CVE-2023-20593] -- omit OCaml 5 patch on fc38 - -* Sat Jul 22 2023 Fedora Release Engineering - 4.17.1-6 -- Rebuilt for https://fedoraproject.org/wiki/Fedora_39_Mass_Rebuild - -* Mon Jul 10 2023 Jerry James - 4.17.1-5 -- Add patch for OCaml 5.0.0 - -* Tue Jun 27 2023 Michael Young - 4.17.1-4 -- work around a build problem with python 3.12 - -* Tue Jun 13 2023 Python Maint - 4.17.1-3 -- Rebuilt for Python 3.12 - -* Tue May 16 2023 Michael Young - 4.17.1-2 -- Mishandling of guest SSBD selection on AMD hardware - [XSA-431, CVE-2022-42336] - -* Tue May 02 2023 Michael Young - 4.17.1-1 -- update to xen-4.17.1 - remove patches now included upstream - switch from patchN to patch N format for applying patches - -* Tue Apr 25 2023 Michael Young - 4.17.0-9 -- x86 shadow paging arbitrary pointer dereference [XSA-430, CVE-2022-42335] - -* Tue Mar 21 2023 Michael Young - 4.17.0-8 -- 3 security issues (#2180425) - x86 shadow plus log-dirty mode use-after-free [XSA-427, CVE-2022-42332] - x86/HVM pinned cache attributes mis-handling [XSA-428, CVE-2022-42333, - CVE-2022-42334] - x86: speculative vulnerability in 32bit SYSCALL path [XSA-429, - CVE-2022-42331] - -* Sat Feb 18 2023 Michael Young - 4.17.0-7 -- use OVMF.fd from new edk2-ovmf-xen package as ovmf.bin file - built from edk2-ovmf package no longer supports xen (#2170930) - -* Tue Feb 14 2023 Michael Young - 4.17.0-6 -- x86: Cross-Thread Return Address Predictions [XSA-426, CVE-2022-27672] - -* Wed Jan 25 2023 Michael Young - 4.17.0-5 -- Guests can cause Xenstore crash via soft reset [XSA-425, CVE-2022-42330] - (#2164520) - -* Tue Jan 24 2023 Michael Young -- now need BuildRequires for hostname - -* Sat Jan 21 2023 Fedora Release Engineering - 4.17.0-4 -- Rebuilt for https://fedoraproject.org/wiki/Fedora_38_Mass_Rebuild - -* Tue Jan 17 2023 Michael Young - 4.17.0-3 -- build fix for gcc13 - -* Sun Jan 08 2023 Michael Young - 4.17.0-2 -- fix clean up of init scripts if /etc/rc.d/init.d doesn't exist - -* Tue Dec 20 2022 Michael Young -- python3-setuptools BuildRequires is needed for python 3.12 - -* Tue Dec 13 2022 Michael Young - 4.17.0-1 -- update to xen-4.17.0 - rebase xen.fedora.systemd.patch and xen.canonicalize.patch - remove or adjust patches now included or superceded upstream - /var/lib/xenstored has moved to /run/xenstored - -* Tue Nov 08 2022 Michael Young - 4.16.2-4 -- x86: Multiple speculative security issues [XSA-422, CVE-2022-23824] - -* Tue Nov 01 2022 Michael Young - 4.16.2-3 -- x86: unintended memory sharing between guests [XSA-412, CVE-2022-42327] -- Xenstore: Guests can crash xenstored [XSA-414, CVE-2022-42309] -- Xenstore: Guests can create orphaned Xenstore nodes [XSA-415, - CVE-2022-42310] -- Xenstore: guests can let run xenstored out of memory [XSA-326, - CVE-2022-42311, CVE-2022-42312, CVE-2022-42313, CVE-2022-42314, - CVE-2022-42315, CVE-2022-42316, CVE-2022-42317, CVE-2022-42318] -- Xenstore: Guests can cause Xenstore to not free temporary memory - [XSA-416, CVE-2022-42319] -- Xenstore: Guests can get access to Xenstore nodes of deleted domains - [XSA-417, CVE-2022-42320] -- Xenstore: Guests can crash xenstored via exhausting the stack - [XSA-418, CVE-2022-42321] -- Xenstore: Cooperating guests can create arbitrary numbers of nodes - [XSA-419, CVE-2022-42322, CVE-2022-42323] -- Oxenstored 32->31 bit integer truncation issues [XSA-420, CVE-2022-42324] -- Xenstore: Guests can create arbitrary number of nodes via transactions - [XSA-421, CVE-2022-42325, CVE-2022-42326] - -* Fri Oct 14 2022 Michael Young - 4.16.2-2 -- Arm: unbounded memory consumption for 2nd-level page tables [XSA-409, - CVE-2022-33747] (#2135268) -- P2M pool freeing may take excessively long [XSA-410, CVE-2022-33746] - (#2135641) -- lock order inversion in transitive grant copy handling [XSA-411, - CVE-2022-33748] (#2135263) - -* Sat Sep 17 2022 Michael Young - 4.16.2-1 -- update to xen-4.16.2 - remove or adjust patches now included or superceded upstream - -* Tue Jul 26 2022 Michael Young - 4.16.1-8 -- insufficient TLB flush for x86 PV guests in shadow mode [XSA-408, - CVE-2022-33745] (#2112223) - -* Sat Jul 23 2022 Fedora Release Engineering - 4.16.1-7 -- Rebuilt for https://fedoraproject.org/wiki/Fedora_37_Mass_Rebuild - -* Tue Jul 12 2022 Michael Young - 4.16.1-6 -- Retbleed - arbitrary speculative code execution with return instructions - [XSA-407, CVE-2022-23816, CVE-2022-23825, CVE-2022-29900] - -* Tue Jul 05 2022 Michael Young - 4.16.1-5 -- Linux disk/nic frontends data leaks [XSA-403, CVE-2022-26365, - CVE-2022-33740, CVE-2022-33741, CVE-2022-33742] (#2104747) - -* Tue Jun 21 2022 Michael Young - 4.16.1-4 -- x86: MMIO Stale Data vulnerabilities [XSA-404, CVE-2022-21123, - CVE-2022-21125, CVE-2022-21166] - -* Mon Jun 13 2022 Python Maint - 4.16.1-3 -- Rebuilt for Python 3.11 (F37 build only) - -* Sat Jun 11 2022 Michael Young - 4.16.1-2 -- stop building for ix86 and armv7hl due to missing build dependency -- x86 pv: Race condition in typeref acquisition [XSA-401, CVE-2022-26362] -- x86 pv: Insufficient care with non-coherent mappings [ XSA-402, - CVE-2022-26363, CVE-2022-26364] -- additional patches so above applies cleanly - -* Thu Apr 14 2022 Michael Young - 4.16.1-1 -- update to xen-4.16.1 - remove or adjust patches now included or superceded upstream - renumber patches -- strip .efi file to help EFI partitions with limited space - -* Tue Apr 05 2022 Michael Young - 4.16.0-6 -- Racy interactions between dirty vram tracking and paging log dirty - hypercalls [XSA-397, CVE-2022-26356] -- race in VT-d domain ID cleanup [XSA-399, CVE-2022-26357] -- IOMMU: RMRR (VT-d) and unity map (AMD-Vi) handling issues [XSA-400, - CVE-2022-26358, CVE-2022-26359, CVE-2022-26360, CVE-2022-26361] -- additional patches so above applies cleanly - -* Mon Mar 21 2022 Michael Young - 4.16.0-5 -- fix build of xen*.efi file and package it in /usr/lib*/efi - -* Tue Mar 15 2022 Michael Young - 4.16.0-4 -- Multiple speculative security issues [XSA-398] -- additional patches so above applies cleanly - -* Sat Jan 29 2022 Michael Young - 4.16.0-3 -- adjust build script and patches for gcc12 and package note support - -* Sat Jan 29 2022 Michael Young -- arm: guest_physmap_remove_page not removing the p2m mappings [XSA-393, - CVE-2022-23033] (#2045044) -- A PV guest could DoS Xen while unmapping a grant [XSA-394, CVE-2022-23034] - (#2045042) -- Insufficient cleanup of passed-through device IRQs [XSA-395, - CVE-2022-23035] (#2045040) - -* Sat Jan 22 2022 Fedora Release Engineering - 4.16.0-2 -- Rebuilt for https://fedoraproject.org/wiki/Fedora_36_Mass_Rebuild - -* Mon Jan 10 2022 Michael Young - 4.16.0-1 -- update to xen-4.16.0 - rebase xen.canonicalize.patch and xen.gcc11.fixes.patch - drop xen.fedora.efi.build.patch which is no longer useful - remove or adjust patches now included or superceded upstream - update libxenstore libary versions - unpackage /boot/efi/EFI/fedora/xen*.efi - package xen-mceinj and xen-vmtrace -- don't build qemu-traditional or pv-grub by default (following upstream) -- fix some incorrect dependencies on building qemu-traditional -- change grub module package dependencies from Suggests to Recommends - and move to hypervisor package -- rework seabios configure logic (bios.bin is no longer useful) -- frontends vulnerable to backends [XSA-376] (document change only) - -* Tue Nov 23 2021 Michael Young - 4.15.1-4 -- guests may exceed their designated memory limit [XSA-385, CVE-2021-28706] -- PoD operations on misaligned GFNs [XSA-388, CVE-2021-28704, CVE-2021-28707 - CVE-2021-28708] -- issues with partially successful P2M updates on x86 [XSA-389, - CVE-2021-28705, CVE-2021-28709] -- certain VT-d IOMMUs may not work in shared page table mode [XSA-390, - CVE-2021-28710] - -* Wed Oct 06 2021 Michael Young - 4.15.1-3 -- rebuild (f36 only) for OCaml 4.13.1 - -* Tue Oct 05 2021 Michael Young - 4.15.1-2 -- PCI devices with RMRRs not deassigned correctly [XSA-386, CVE-2021-28702] - (#2011248) - -* Sun Sep 12 2021 Michael Young - 4.15.1-1 -- update to xen-4.15.1 - remove or adjust patches now included or superceded upstream - update libxencall version - -* Wed Sep 08 2021 Michael Young - 4.15.0-7 -- Another race in XENMAPSPACE_grant_table handling [XSA-384, CVE-2021-28701] - (#2002786) -- bugfix for XSA-380 -- stop editing grub files in /boot/efi/EFI/fedora - -* Wed Aug 25 2021 Michael Young - 4.15.0-6 -- IOMMU page mapping issues on x86 [XSA-378, CVE-2021-28694, - CVE-2021-28695, CVE-2021-28696] (#1997531) (#1997568) - (#1997537) -- grant table v2 status pages may remain accessible after de-allocation - [XSA-379, CVE-2021-28697] (#1997520) -- long running loops in grant table handling [XSA-380, CVE-2021-28698] - (#1997526) -- inadequate grant-v2 status frames array bounds check [XSA-382, - CVE-2021-28699] (#1997523) -- xen/arm: No memory limit for dom0less domUs [XSA-383, CVE-2021-28700] - (#1997527) -- grub x86_64-efi modules now go into /boot/grub2 - -* Thu Aug 12 2021 Michael Young - 4.15.0-5 - - work around build issue with GNU ld 2.37 (#1990344) - -* Fri Jul 23 2021 Fedora Release Engineering - 4.15.0-4 -- Rebuilt for https://fedoraproject.org/wiki/Fedora_35_Mass_Rebuild - -* Tue Jun 08 2021 Michael Young - 4.15.0-3 -- xen/arm: Boot modules are not scrubbed [XSA-372, CVE-2021-28693] - (#1970542) -- inappropriate x86 IOMMU timeout detection / handling - [XSA-373, CVE-2021-28692] (#1970540) -- Speculative Code Store Bypass [XSA-375, CVE-2021-0089, CVE-2021-26313] - (#1970531) -- x86: TSX Async Abort protections not restored after S3 - [XSA-377, CVE-2021-28690] (#1970546) - -* Fri Jun 04 2021 Python Maint - 4.15.0-2 -- Rebuilt for Python 3.10 - -* Wed May 05 2021 Michael Young - 4.15.0-1 -- update to xen-4.15.0 - adjust xen.canonicalize.patch - remove or adjust patches now included or superceded upstream - renumber patch - update libxendevicemodel libxenevtchn libxenforeignmemory versions - /etc/bash_completion.d/xl.sh is now xl - package xen-access xen-memshare xenstorepvh-stubdom.gz - xl-pci-configuration.5.gz -- adjust xen.ocaml.4.12.fixes.patch to work with earlier ocaml -- re-copy grub modules if they have changed - -* Fri Mar 19 2021 Michael Young - 4.14.1-8 -- HVM soft-reset crashes toolstack [XSA-368, CVE-2021-28687] (#1940610) -- adjust efi test to stop build failing - -* Tue Mar 02 2021 Michael Young - 4.14.1-6 -- build fixes for OCaml 4.12.0 - -* Tue Feb 16 2021 Michael Young - 4.14.1-5 -- Linux: display frontend "be-alloc" mode is unsupported (comment only) - [XSA-363, CVE-2021-26934] (#1929549) -- arm: The cache may not be cleaned for newly allocated scrubbed pages - [XSA-364, CVE-2021-26933] (#1929547) - -* Mon Feb 01 2021 Michael Young - 4.14.1-4 -- backport upstream zstd dom0 and guest patches -- add libzstd-devel BuildRequires -- add weak dependency on grub modules to improve initial boot setup - -* Wed Jan 27 2021 Fedora Release Engineering - 4.14.1-3 -- Rebuilt for https://fedoraproject.org/wiki/Fedora_34_Mass_Rebuild - -* Thu Jan 21 2021 Michael Young - 4.14.1-2 -- IRQ vector leak on x86 [XSA-360] - -* Sun Dec 20 2020 Michael Young - 4.14.1-1 -- update to 4.14.1 - adjust xen.canonicalize.patch - remove or adjust patches now included or superceded upstream - renumber patches - -* Tue Dec 15 2020 Michael Young - 4.14.0-14 -- xenstore watch notifications lacking permission checks [XSA-115, - CVE-2020-29480] (#1908091) -- Xenstore: new domains inheriting existing node permissions [XSA-322, - CVE-2020-29481] (#1908095) -- Xenstore: wrong path length check [XSA-323, CVE-2020-29482] (#1908096) -- Xenstore: guests can crash xenstored via watchs [XSA-324, CVE-2020-29484] - (#1908088) -- Xenstore: guests can disturb domain cleanup [XSA-325, CVE-2020-29483] - (#1908087) -- oxenstored memory leak in reset_watches [XSA-330, CVE-2020-29485] - (#1908000) -- undue recursion in x86 HVM context switch code [XSA-348, CVE-2020-29566] - (#1908085) -- oxenstored: node ownership can be changed by unprivileged clients - [XSA-352, CVE-2020-29486] (#1908003) -- oxenstored: permissions not checked on root node [XSA-353, CVE-2020-29479] - (#1908002) -- infinite loop when cleaning up IRQ vectors [XSA-356, CVE-2020-29567] - (#1907932) -- FIFO event channels control block related ordering [XSA-358, - CVE-2020-29570] (#1907931) -- FIFO event channels control structure ordering [XSA-359, CVE-2020-29571] - (#1908089) - -* Sat Dec 05 2020 Jeff Law - 4.14.0-13 -- Work around another gcc-11 stringop-overflow diagnostic - -* Tue Nov 24 2020 Michael Young - 4.14.0-12 -- stack corruption from XSA-346 change [XSA-355] - -* Mon Nov 23 2020 Michael Young - 4.14.0-11 -- support zstd compressed kernels (dom0 only) based on linux kernel code - -* Tue Nov 10 2020 Michael Young - 4.14.0-10 -- Information leak via power sidechannel [XSA-351, CVE-2020-28368] - (#1897146) -- add make as build requires - -* Tue Nov 03 2020 Michael Young - 4.14.0-9 -- revised patch for XSA-286 (mitigating performance impact) - -* Fri Oct 30 2020 Jeff Law - 4.14.0-8 -- Work around gcc-11 stringop-overflow diagnostics as well - -* Wed Oct 28 2020 Michael Young - 4.14.0-7 -- x86 PV guest INVLPG-like flushes may leave stale TLB entries - [XSA-286, CVE-2020-27674] (#1891092) -- simplify grub scripts (patches from Thierry Vignaud ) -- some fixes for gcc 11 - -* Tue Oct 20 2020 Michael Young - 4.14.0-6 -- x86: Race condition in Xen mapping code [XSA-345, CVE-2020-27672] - (#1891097) -- undue deferral of IOMMU TLB flushes [XSA-346, CVE-2020-27671] - (#1891093) -- unsafe AMD IOMMU page table updates [XSA-347, CVE-2020-27670] - (#1891088) - -* Tue Sep 22 2020 Michael Young - 4.14.0-5 -- x86 pv: Crash when handling guest access to MSR_MISC_ENABLE [XSA-333, - CVE-2020-25602] (#1881619) -- Missing unlock in XENMEM_acquire_resource error path [XSA-334, - CVE-2020-25598] (#1881616) -- race when migrating timers between x86 HVM vCPU-s [XSA-336, - CVE-2020-25604] (#1881618) -- PCI passthrough code reading back hardware registers [XSA-337, - CVE-2020-25595] (#1881587) -- once valid event channels may not turn invalid [XSA-338, CVE-2020-25597] - (#1881588) -- x86 pv guest kernel DoS via SYSENTER [XSA-339, CVE-2020-25596] - (#1881617) -- Missing memory barriers when accessing/allocating an event channel [XSA-340, - CVE-2020-25603] (#1881583) -- out of bounds event channels available to 32-bit x86 domains [XSA-342, - CVE-2020-25600] (#1881582) -- races with evtchn_reset() [XSA-343, CVE-2020-25599] (#1881581) -- lack of preemption in evtchn_reset() / evtchn_destroy() [XSA-344, - CVE-2020-25601] (#1881586) - -* Thu Sep 03 2020 Michael Young - 4.14.0-4 -- rebuild for OCaml 4.11.1 - -* Mon Aug 24 2020 Michael Young - 4.14.0-3 -- QEMU: usb: out-of-bounds r/w access issue [XSA-335, CVE-2020-14364] - (#1871850) - -* Wed Jul 29 2020 Fedora Release Engineering - 4.14.0-2 -- Rebuilt for https://fedoraproject.org/wiki/Fedora_33_Mass_Rebuild - -* Sun Jul 26 2020 Michael Young - 4.14.0-1 -- update to 4.14.0 - remove or adjust patches now included or superceded upstream - adjust xen.hypervisor.config - bison and flex packages now needed for hypervisor build - /usr/bin/vchan-socket-proxy and /usr/sbin/xenhypfs have been added - with associated libraries and man page -- re-enable pandoc for more documentation - adding xen-vbd-interface.7.gz -- revise documentation build dependencies - drop tex, texinfo, ghostscript, graphviz, discount - add perl(Pod::Html) perl(File::Find) -- additional build dependency for ocaml on perl(Data::Dumper) - -* Tue Jul 14 2020 Tom Stellard - 4.13.1-5 -- Use make macros -- https://fedoraproject.org/wiki/Changes/UseMakeBuildInstallMacro - -* Tue Jul 07 2020 Michael Young - 4.13.1-4 -- incorrect error handling in event channel port allocation leads to - DoS [XSA-317, CVE-2020-15566] (#1854465) -- inverted code paths in x86 dirty VRAM tracking leads to DoS - [XSA-319, CVE-2020-15563] (#1854463) -- xen: insufficient cache write-back under VT-d leads to DoS - [XSA-321, CVE-2020-15565] (#1854467) -- missing alignment check in VCPUOP_register_vcpu_info leads to DoS - [XSA-327, CVE-2020-15564] (#1854458) -- non-atomic modification of live EPT PTE leads to DoS - [XSA-328, CVE-2020-15567] (#1854464) - -* Tue Jun 30 2020 Jeff Law -Disable LTO - -* Wed Jun 10 2020 Michael Young - 4.13.1-3 -- Special Register Buffer speculative side channel [XSA-320] - -* Tue May 26 2020 Miro Hrončok - 4.13.1-2 -- Rebuilt for Python 3.9 - -* Tue May 19 2020 Michael Young - 4.13.1-1 -- update to 4.13.1 - remove patches now included or superceded upstream - -* Tue May 05 2020 Michael Young - 4.13.0-8 -- build aarch64 hypervisor with -mno-outline-atomics to fix gcc 10 build - -* Tue Apr 14 2020 Michael Young - 4.13.0-7 -- multiple xenoprof issues [XSA-313, CVE-2020-11740, CVE-2020-11741] - (#1823912, #1823914) -- Missing memory barriers in read-write unlock paths [XSA-314, - CVE-2020-11739] (#1823784) -- Bad error path in GNTTABOP_map_grant [XSA-316, CVE-2020-11743] (#1823926) -- Bad continuation handling in GNTTABOP_copy [XSA-318, CVE-2020-11742] - (#1823943) - -* Tue Mar 17 2020 Michael Young - 4.13.0-6 -- fix issues in pygrub dependency found by python 3.8 - -* Tue Mar 10 2020 Michael Young - 4.13.0-5 -- setting for --with-system-ipxe should be a rom file (#1778516) -- add weak depends on ipxe-roms-qemu and qemu-system-x86-core - -* Fri Jan 31 2020 Fedora Release Engineering - 4.13.0-4 -- Rebuilt for https://fedoraproject.org/wiki/Fedora_32_Mass_Rebuild - -* Wed Jan 22 2020 Michael Young - 4.13.0-3 -- build fixes for OCaml 4.10.0 and gcc 10 - -* Tue Jan 14 2020 Michael Young - 4.13.0-2 -- arm: a CPU may speculate past the ERET instruction [XSA-312] -- use more explicit library names -- add weak requires for perl (/etc/xen/scripts/locking.sh) - -* Wed Dec 18 2019 Michael Young - 4.13.0-1 -- update to 4.13.0 - remove patches now included or superceded upstream - adjust xen.hypervisor.config - /usr/sbin/xen-tmem-list-parse has been removed - pkgconfig files have moved to %%{_libdir}/pkgconfig - /usr/sbin/xen-ucode has been added (x86 only) - -* Sun Dec 15 2019 Michael Young - 4.12.1-9 -- fix build with OCaml 4.09.0 - -* Wed Dec 11 2019 Michael Young - 4.12.1-8 -- denial of service in find_next_bit() [XSA-307, CVE-2019-19581, - CVE-2019-19582] (#1782211) -- denial of service in HVM/PVH guest userspace code [XSA-308, - CVE-2019-19583] (#1782206) -- privilege escalation due to malicious PV guest [XSA-309, CVE-2019-19578] - (#1782210) -- Further issues with restartable PV type change operations [XSA-310, - CVE-2019-19580] (#1782207) -- vulnerability in dynamic height handling for AMD IOMMU pagetables - [XSA-311, CVE-2019-19577] (#1782208) -- add patches needed to apply XSA-311 - -* Tue Nov 26 2019 Michael Young - 4.12.1-7 -- Device quarantine for alternate pci assignment methods [XSA-306, - CVE-2019-19579] (#1780559) - -* Tue Nov 12 2019 Michael Young - 4.12.1-6 -- add missing XSA-299 patches - -* Tue Nov 12 2019 Michael Young - 4.12.1-5 -- x86: Machine Check Error on Page Size Change DoS [XSA-304, CVE-2018-12207] -- TSX Asynchronous Abort speculative side channel [XSA-305, CVE-2019-11135] - -* Thu Oct 31 2019 Michael Young - 4.12.1-4 -- VCPUOP_initialise DoS [XSA-296, CVE-2019-18420] (#1771368) +* Fri Nov 01 2019 Michael Young - 4.11.2-2 +- VCPUOP_initialise DoS [XSA-296, CVE-2019-18420] - missing descriptor table limit checking in x86 PV emulation [XSA-298, - CVE-2019-18425] (#1771341) + CVE-2019-18425] - Issues with restartable PV type change operations [XSA-299, CVE-2019-18421] - (#1767726) + (#1767726) - add-to-physmap can be abused to DoS Arm hosts [XSA-301, CVE-2019-18423] - (#1771345) - passed through PCI devices may corrupt host memory after deassignment - [XSA-302, CVE-2019-18424] (#1767731) + [XSA-302, CVE-2019-18424] (#1767731) - ARM: Interrupts are unconditionally unmasked in exception handlers - [XSA-303, CVE-2019-18422] (#1771443) + [XSA-303, CVE-2019-18422] -* Thu Oct 03 2019 Miro Hrončok - 4.12.1-3 -- Rebuilt for Python 3.8.0rc1 (#1748018) +* Mon Jul 01 2019 Michael Young - 4.11.2-1 +- update to 4.11.2 + remove patches now fixed upstream + adjust xen.use.fedora.ipxe.patch + drop parts of xen.gcc9.fixes.patch -* Mon Aug 19 2019 Miro Hrončok - 4.12.1-2 -- Rebuilt for Python 3.8 +* Sat Jun 15 2019 Michael Young - 4.11.1-6 +- Unlimited Arm Atomics Operations [XSA-295] (#1720760) -* Fri Aug 09 2019 Michael Young - 4.12.1-1 -- update to 4.12.1 - remove patches for issues now fixed upstream - adjust xen.gcc9.fixes.patch - -* Sat Jul 27 2019 Fedora Release Engineering - 4.12.0-5 -- Rebuilt for https://fedoraproject.org/wiki/Fedora_31_Mass_Rebuild - -* Wed Jun 19 2019 Michael Young - 4.12.0-4 -- Unlimited Arm Atomics Operations [XSA-295, CVE-2019-17349, - CVE-2019-17350] (#1720760) -- some debug files are now properly packaged in debuginfo rpms - -* Tue Jun 18 2019 Zbigniew Jędrzejewski-Szmek -- Fix build with python3.8 (#1704807) - -* Sat Jun 01 2019 Michael Young - 4.12.0-3 -- fix HVM DomU boot on some chipsets -- fix expected FTBFS with Python 3.8 (#1704807) -- adjust grub2 workaround - -* Tue May 14 2019 Michael Young - 4.12.0-2 +* Tue May 14 2019 Michael Young - 4.11.1-5 - Microarchitectural Data Sampling speculative side channel [XSA-297, CVE-2018-12126, CVE-2018-12127, CVE-2018-12130, CVE-2019-11091] - additional patches so above applies cleanly - work around grub2 issues in dom0 -* Fri Apr 05 2019 Michael Young - 4.12.0-1 -- update to 4.12.0 (#1694695) - remove patches for issues now fixed upstream - replace xen.use.fedora.ipxe.patch with --with-system-ipxe - drop xen.glibcfix.patch xen.gcc8.temp.fix.patch which are no longer needed - adjust xen.python.env.patch xen.gcc9.fixes.patch - xen.hypervisor.config refresh - kdd is now xen-kdd, xenmon.py is now xenmon, fsimage.so is now xenfsimage.so - fs libdir is now xenfsimage libdir - xen-ringwatch xen-bugtool have been dropped -- remove remaining traces of efiming and efi_flags logic -- switch from python2 to python3 -- drop systemd_postun and renumber patches - * Tue Mar 05 2019 Michael Young - 4.11.1-4 - xen: various flaws (#1685577) - grant table transfer issues on large hosts [XSA-284, CVE-2019-17340] - race with pass-through device hotplug [XSA-285, CVE-2019-17341] - x86: steal_page violates page_struct access discipline - [XSA-287, CVE-2019-17342] - x86: Inconsistent PV IOMMU discipline [XSA-288, CVE-2019-17343] - missing preemption in x86 PV page table unvalidation - [XSA-290, CVE-2019-17344] - x86/PV: page type reference counting issue with failed IOMMU update - [XSA-291, CVE-2019-17345] - x86: insufficient TLB flushing when using PCID [XSA-292, CVE-2019-17346] - x86: PV kernel context switch corruption [XSA-293, CVE-2019-17347] - x86 shadow: Insufficient TLB flushing when using PCID [XSA-294, - CVE-2019-17348] + grant table transfer issues on large hosts [XSA-284] + race with pass-through device hotplug [XSA-285] + x86: steal_page violates page_struct access discipline [XSA-287] + x86: Inconsistent PV IOMMU discipline [XSA-288] + missing preemption in x86 PV page table unvalidation [XSA-290] + x86/PV: page type reference counting issue with failed IOMMU update [XSA-291] + x86: insufficient TLB flushing when using PCID [XSA-292] + x86: PV kernel context switch corruption [XSA-293] + x86 shadow: Insufficient TLB flushing when using PCID [XSA-294] * Thu Feb 14 2019 Michael Young - 4.11.1-3 - add gcc9 build fixes (#1676229) diff --git a/xen.stubdom.build.patch b/xen.stubdom.build.patch new file mode 100644 index 0000000..05d1b7c --- /dev/null +++ b/xen.stubdom.build.patch @@ -0,0 +1,11 @@ +--- xen-4.11.0/stubdom/Makefile.orig 2018-07-09 14:47:19.000000000 +0100 ++++ xen-4.11.0/stubdom/Makefile 2018-07-11 22:33:00.764078143 +0100 +@@ -87,7 +87,7 @@ + patch -d $@ -p0 < newlib.patch + patch -d $@ -p0 < newlib-chk.patch + patch -d $@ -p1 < newlib-stdint-size_max-fix-from-1.17.0.patch +- find $@ -type f | xargs perl -i.bak \ ++ find $@ -type f | xargs egrep -l "tzname|daylight|timezone" | xargs perl -i.bak \ + -pe 's/\b_(tzname|daylight|timezone)\b/$$1/g' + touch $@ + diff --git a/xen.use.fedora.ipxe.patch b/xen.use.fedora.ipxe.patch new file mode 100644 index 0000000..25ab8d7 --- /dev/null +++ b/xen.use.fedora.ipxe.patch @@ -0,0 +1,33 @@ +--- xen-4.2.0/tools/firmware/hvmloader/Makefile.orig 2012-05-27 21:57:04.481812859 +0100 ++++ xen-4.2.0/tools/firmware/hvmloader/Makefile 2012-06-02 18:52:44.935034128 +0100 +@@ -48,7 +48,7 @@ + else + CIRRUSVGA_ROM := ../vgabios/VGABIOS-lgpl-latest.cirrus.bin + endif +-ETHERBOOT_ROMS := $(addprefix ../etherboot/ipxe/src/bin/, $(addsuffix .rom, $(ETHERBOOT_NICS))) ++ETHERBOOT_ROMS := $(addprefix /usr/share/ipxe/, $(addsuffix .rom, $(ETHERBOOT_NICS))) + endif + + ROMS := +--- xen-4.2.0/Config.mk.orig 2012-05-27 21:57:04.479812884 +0100 ++++ xen-4.2.0/Config.mk 2012-06-02 18:55:14.087169469 +0100 +@@ -206,7 +206,7 @@ + + SEABIOS_UPSTREAM_REVISION ?= rel-1.11.1 + +-ETHERBOOT_NICS ?= rtl8139 8086100e ++ETHERBOOT_NICS ?= 10ec8139 8086100e + + + QEMU_TRADITIONAL_REVISION ?= xen-4.11.2 +--- xen-4.2.0/tools/firmware/Makefile.orig 2012-05-27 21:57:04.480812871 +0100 ++++ xen-4.2.0/tools/firmware/Makefile 2012-06-02 19:03:52.254691484 +0100 +@@ -10,7 +10,7 @@ + SUBDIRS-$(CONFIG_SEABIOS) += seabios-dir + SUBDIRS-$(CONFIG_ROMBIOS) += rombios + SUBDIRS-$(CONFIG_ROMBIOS) += vgabios +-SUBDIRS-$(CONFIG_ROMBIOS) += etherboot ++#SUBDIRS-$(CONFIG_ROMBIOS) += etherboot + SUBDIRS-$(CONFIG_PV_SHIM) += xen-dir + SUBDIRS-y += hvmloader + diff --git a/xen.vwprintw.fix.patch b/xen.vwprintw.fix.patch new file mode 100644 index 0000000..5223f67 --- /dev/null +++ b/xen.vwprintw.fix.patch @@ -0,0 +1,11 @@ +--- xen-4.11.0/tools/xenstat/xentop/xentop.c.orig 2018-07-09 14:47:19.000000000 +0100 ++++ xen-4.11.0/tools/xenstat/xentop/xentop.c 2018-08-14 22:41:08.035898962 +0100 +@@ -301,7 +301,7 @@ + if (!batch) { + if((current_row() < lines()-1)) { + va_start(args, fmt); +- vwprintw(stdscr, (curses_str_t)fmt, args); ++ vw_printw(stdscr, (curses_str_t)fmt, args); + va_end(args); + } + } else { diff --git a/xsa296.patch b/xsa296.patch new file mode 100644 index 0000000..e71ea7f --- /dev/null +++ b/xsa296.patch @@ -0,0 +1,195 @@ +From: Andrew Cooper +Subject: xen/hypercall: Don't use BUG() for parameter checking in hypercall_create_continuation() + +Since c/s 1d429034 "hypercall: update vcpu_op to take an unsigned vcpuid", +which incorrectly swapped 'i' for 'u' in the parameter type list, guests have +been able to hit the BUG() in next_args()'s default case. + +Correct these back to 'i'. + +In addition, make adjustments to prevent this class of issue from occurring in +the future - crashing Xen is not an appropriate form of parameter checking. + +Capitalise NEXT_ARG() to catch all uses, to highlight that it is a macro doing +non-function-like things behind the scenes, and undef it when appropriate. +Implement a bad_fmt: block which prints an error, asserts unreachable, and +crashes the guest. + +On the ARM side, drop all parameter checking of p. It is asymmetric with the +x86 side, and akin to expecting memcpy() or sprintf() to check their src/fmt +parameter before use. A caller passing "" or something other than a string +literal will be obvious during code review. + +This is XSA-296. + +Signed-off-by: Andrew Cooper +Acked-by: Julien Grall + +diff --git a/xen/arch/arm/domain.c b/xen/arch/arm/domain.c +index 941bbff4fe..a3da8e9c08 100644 +--- a/xen/arch/arm/domain.c ++++ b/xen/arch/arm/domain.c +@@ -383,14 +383,15 @@ void sync_vcpu_execstate(struct vcpu *v) + /* Nothing to do -- no lazy switching */ + } + +-#define next_arg(fmt, args) ({ \ ++#define NEXT_ARG(fmt, args) \ ++({ \ + unsigned long __arg; \ + switch ( *(fmt)++ ) \ + { \ + case 'i': __arg = (unsigned long)va_arg(args, unsigned int); break; \ + case 'l': __arg = (unsigned long)va_arg(args, unsigned long); break; \ + case 'h': __arg = (unsigned long)va_arg(args, void *); break; \ +- default: __arg = 0; BUG(); \ ++ default: goto bad_fmt; \ + } \ + __arg; \ + }) +@@ -405,9 +406,6 @@ unsigned long hypercall_create_continuation( + unsigned int i; + va_list args; + +- /* All hypercalls take at least one argument */ +- BUG_ON( !p || *p == '\0' ); +- + current->hcall_preempted = true; + + va_start(args, format); +@@ -415,7 +413,7 @@ unsigned long hypercall_create_continuation( + if ( mcs->flags & MCSF_in_multicall ) + { + for ( i = 0; *p != '\0'; i++ ) +- mcs->call.args[i] = next_arg(p, args); ++ mcs->call.args[i] = NEXT_ARG(p, args); + + /* Return value gets written back to mcs->call.result */ + rc = mcs->call.result; +@@ -431,7 +429,7 @@ unsigned long hypercall_create_continuation( + + for ( i = 0; *p != '\0'; i++ ) + { +- arg = next_arg(p, args); ++ arg = NEXT_ARG(p, args); + + switch ( i ) + { +@@ -454,7 +452,7 @@ unsigned long hypercall_create_continuation( + + for ( i = 0; *p != '\0'; i++ ) + { +- arg = next_arg(p, args); ++ arg = NEXT_ARG(p, args); + + switch ( i ) + { +@@ -475,8 +473,16 @@ unsigned long hypercall_create_continuation( + va_end(args); + + return rc; ++ ++ bad_fmt: ++ gprintk(XENLOG_ERR, "Bad hypercall continuation format '%c'\n", *p); ++ ASSERT_UNREACHABLE(); ++ domain_crash(current->domain); ++ return 0; + } + ++#undef NEXT_ARG ++ + void startup_cpu_idle_loop(void) + { + struct vcpu *v = current; +diff --git a/xen/arch/x86/hypercall.c b/xen/arch/x86/hypercall.c +index d483dbaa6b..4643e5eb43 100644 +--- a/xen/arch/x86/hypercall.c ++++ b/xen/arch/x86/hypercall.c +@@ -80,14 +80,15 @@ const hypercall_args_t hypercall_args_table[NR_hypercalls] = + #undef COMP + #undef ARGS + +-#define next_arg(fmt, args) ({ \ ++#define NEXT_ARG(fmt, args) \ ++({ \ + unsigned long __arg; \ + switch ( *(fmt)++ ) \ + { \ + case 'i': __arg = (unsigned long)va_arg(args, unsigned int); break; \ + case 'l': __arg = (unsigned long)va_arg(args, unsigned long); break; \ + case 'h': __arg = (unsigned long)va_arg(args, void *); break; \ +- default: __arg = 0; BUG(); \ ++ default: goto bad_fmt; \ + } \ + __arg; \ + }) +@@ -109,7 +110,7 @@ unsigned long hypercall_create_continuation( + if ( mcs->flags & MCSF_in_multicall ) + { + for ( i = 0; *p != '\0'; i++ ) +- mcs->call.args[i] = next_arg(p, args); ++ mcs->call.args[i] = NEXT_ARG(p, args); + } + else + { +@@ -121,7 +122,7 @@ unsigned long hypercall_create_continuation( + { + for ( i = 0; *p != '\0'; i++ ) + { +- arg = next_arg(p, args); ++ arg = NEXT_ARG(p, args); + switch ( i ) + { + case 0: regs->rdi = arg; break; +@@ -137,7 +138,7 @@ unsigned long hypercall_create_continuation( + { + for ( i = 0; *p != '\0'; i++ ) + { +- arg = next_arg(p, args); ++ arg = NEXT_ARG(p, args); + switch ( i ) + { + case 0: regs->rbx = arg; break; +@@ -154,8 +155,16 @@ unsigned long hypercall_create_continuation( + va_end(args); + + return op; ++ ++ bad_fmt: ++ gprintk(XENLOG_ERR, "Bad hypercall continuation format '%c'\n", *p); ++ ASSERT_UNREACHABLE(); ++ domain_crash(curr->domain); ++ return 0; + } + ++#undef NEXT_ARG ++ + int hypercall_xlat_continuation(unsigned int *id, unsigned int nr, + unsigned int mask, ...) + { +diff --git a/xen/common/compat/domain.c b/xen/common/compat/domain.c +index 39877b3ab2..2531fa7421 100644 +--- a/xen/common/compat/domain.c ++++ b/xen/common/compat/domain.c +@@ -81,7 +81,7 @@ int compat_vcpu_op(int cmd, unsigned int vcpuid, XEN_GUEST_HANDLE_PARAM(void) ar + } + + if ( rc == -ERESTART ) +- rc = hypercall_create_continuation(__HYPERVISOR_vcpu_op, "iuh", ++ rc = hypercall_create_continuation(__HYPERVISOR_vcpu_op, "iih", + cmd, vcpuid, arg); + + break; +diff --git a/xen/common/domain.c b/xen/common/domain.c +index 2308588052..65bcd85e34 100644 +--- a/xen/common/domain.c ++++ b/xen/common/domain.c +@@ -1411,7 +1411,7 @@ long do_vcpu_op(int cmd, unsigned int vcpuid, XEN_GUEST_HANDLE_PARAM(void) arg) + + rc = arch_initialise_vcpu(v, arg); + if ( rc == -ERESTART ) +- rc = hypercall_create_continuation(__HYPERVISOR_vcpu_op, "iuh", ++ rc = hypercall_create_continuation(__HYPERVISOR_vcpu_op, "iih", + cmd, vcpuid, arg); + + break; diff --git a/xsa298-4.11.patch b/xsa298-4.11.patch new file mode 100644 index 0000000..9f649e3 --- /dev/null +++ b/xsa298-4.11.patch @@ -0,0 +1,87 @@ +From: Jan Beulich +Subject: x86/PV: check GDT/LDT limits during emulation + +Accesses beyond the LDT limit originating from emulation would trigger +the ASSERT() in pv_map_ldt_shadow_page(). On production builds such +accesses would cause an attempt to promote the touched page (offset from +the present LDT base address) to a segment descriptor one. If this +happens to succeed, guest user mode would be able to elevate its +privileges to that of the guest kernel. This is particularly easy when +there's no LDT at all, in which case the LDT base stored internally to +Xen is simply zero. + +Also adjust the ASSERT() that was triggering: It was off by one to +begin with, and for production builds we also better use +ASSERT_UNREACHABLE() instead with suitable recovery code afterwards. + +This is XSA-298. + +Reported-by: Andrew Cooper +Signed-off-by: Jan Beulich +Reviewed-by: Andrew Cooper + +--- a/xen/arch/x86/pv/emul-gate-op.c ++++ b/xen/arch/x86/pv/emul-gate-op.c +@@ -51,7 +51,13 @@ static int read_gate_descriptor(unsigned + const struct desc_struct *pdesc = gdt_ldt_desc_ptr(gate_sel); + + if ( (gate_sel < 4) || +- ((gate_sel >= FIRST_RESERVED_GDT_BYTE) && !(gate_sel & 4)) || ++ /* ++ * We're interested in call gates only, which occupy a single ++ * seg_desc_t for 32-bit and a consecutive pair of them for 64-bit. ++ */ ++ ((gate_sel >> 3) + !is_pv_32bit_vcpu(v) >= ++ (gate_sel & 4 ? v->arch.pv_vcpu.ldt_ents ++ : v->arch.pv_vcpu.gdt_ents)) || + __get_user(desc, pdesc) ) + return 0; + +@@ -70,7 +76,7 @@ static int read_gate_descriptor(unsigned + if ( !is_pv_32bit_vcpu(v) ) + { + if ( (*ar & 0x1f00) != 0x0c00 || +- (gate_sel >= FIRST_RESERVED_GDT_BYTE - 8 && !(gate_sel & 4)) || ++ /* Limit check done above already. */ + __get_user(desc, pdesc + 1) || + (desc.b & 0x1f00) ) + return 0; +--- a/xen/arch/x86/pv/emulate.c ++++ b/xen/arch/x86/pv/emulate.c +@@ -31,7 +31,14 @@ int pv_emul_read_descriptor(unsigned int + { + struct desc_struct desc; + +- if ( sel < 4) ++ if ( sel < 4 || ++ /* ++ * Don't apply the GDT limit here, as the selector may be a Xen ++ * provided one. __get_user() will fail (without taking further ++ * action) for ones falling in the gap between guest populated ++ * and Xen ones. ++ */ ++ ((sel & 4) && (sel >> 3) >= v->arch.pv_vcpu.ldt_ents) ) + desc.b = desc.a = 0; + else if ( __get_user(desc, gdt_ldt_desc_ptr(sel)) ) + return 0; +--- a/xen/arch/x86/pv/mm.c ++++ b/xen/arch/x86/pv/mm.c +@@ -92,12 +92,16 @@ bool pv_map_ldt_shadow_page(unsigned int + BUG_ON(unlikely(in_irq())); + + /* +- * Hardware limit checking should guarantee this property. NB. This is ++ * Prior limit checking should guarantee this property. NB. This is + * safe as updates to the LDT can only be made by MMUEXT_SET_LDT to the + * current vcpu, and vcpu_reset() will block until this vcpu has been + * descheduled before continuing. + */ +- ASSERT((offset >> 3) <= curr->arch.pv_vcpu.ldt_ents); ++ if ( unlikely((offset >> 3) >= curr->arch.pv_vcpu.ldt_ents) ) ++ { ++ ASSERT_UNREACHABLE(); ++ return false; ++ } + + if ( is_pv_32bit_domain(currd) ) + linear = (uint32_t)linear; diff --git a/xsa299-4.11-0001-x86-mm-L1TF-checks-don-t-leave-a-partial-entry.patch b/xsa299-4.11-0001-x86-mm-L1TF-checks-don-t-leave-a-partial-entry.patch new file mode 100644 index 0000000..6475328 --- /dev/null +++ b/xsa299-4.11-0001-x86-mm-L1TF-checks-don-t-leave-a-partial-entry.patch @@ -0,0 +1,94 @@ +From 852df269d247e177d5f2e9b8f3a4301a6fdd76bd Mon Sep 17 00:00:00 2001 +From: George Dunlap +Date: Thu, 10 Oct 2019 17:57:49 +0100 +Subject: [PATCH 01/11] x86/mm: L1TF checks don't leave a partial entry + +On detection of a potential L1TF issue, most validation code returns +-ERESTART to allow the switch to shadow mode to happen and cause the +original operation to be restarted. + +However, in the validation code, the return value -ERESTART has been +repurposed to indicate 1) the function has partially completed +something which needs to be undone, and 2) calling put_page_type() +should cleanly undo it. This causes problems in several places. + +For L1 tables, on receiving an -ERESTART return from alloc_l1_table(), +alloc_page_type() will set PGT_partial on the page. If for some +reason the original operation never restarts, then on domain +destruction, relinquish_memory() will call free_page_type() on the +page. + +Unfortunately, alloc_ and free_l1_table() aren't set up to deal with +PGT_partial. When returning a failure, alloc_l1_table() always +de-validates whatever it's validated so far, and free_l1_table() +always devalidates the whole page. This means that if +relinquish_memory() calls free_page_type() on an L1 that didn't +complete due to an L1TF, it will call put_page_from_l1e() on "page +entries" that have never been validated. + +For L2+ tables, setting rc to ERESTART causes the rest of the +alloc_lN_table() function to *think* that the entry in question will +have PGT_partial set. This will cause it to set partial_pte = 1. If +relinqush_memory() then calls free_page_type() on one of those pages, +then free_lN_table() will call put_page_from_lNe() on the entry when +it shouldn't. + +Rather than indicating -ERESTART, indicate -EINTR. This is the code +to indicate that nothing has changed from when you started the call +(which is effectively how alloc_l1_table() handles errors). + +mod_lN_entry() shouldn't have any of these types of problems, so leave +potential changes there for a clean-up patch later. + +This is part of XSA-299. + +Reported-by: George Dunlap +Signed-off-by: George Dunlap +Reviewed-by: Jan Beulich +--- + xen/arch/x86/mm.c | 8 ++++---- + 1 file changed, 4 insertions(+), 4 deletions(-) + +diff --git a/xen/arch/x86/mm.c b/xen/arch/x86/mm.c +index e6a4cb28f8..8ced185b49 100644 +--- a/xen/arch/x86/mm.c ++++ b/xen/arch/x86/mm.c +@@ -1110,7 +1110,7 @@ get_page_from_l2e( + int rc; + + if ( !(l2e_get_flags(l2e) & _PAGE_PRESENT) ) +- return pv_l1tf_check_l2e(d, l2e) ? -ERESTART : 1; ++ return pv_l1tf_check_l2e(d, l2e) ? -EINTR : 1; + + if ( unlikely((l2e_get_flags(l2e) & L2_DISALLOW_MASK)) ) + { +@@ -1142,7 +1142,7 @@ get_page_from_l3e( + int rc; + + if ( !(l3e_get_flags(l3e) & _PAGE_PRESENT) ) +- return pv_l1tf_check_l3e(d, l3e) ? -ERESTART : 1; ++ return pv_l1tf_check_l3e(d, l3e) ? -EINTR : 1; + + if ( unlikely((l3e_get_flags(l3e) & l3_disallow_mask(d))) ) + { +@@ -1175,7 +1175,7 @@ get_page_from_l4e( + int rc; + + if ( !(l4e_get_flags(l4e) & _PAGE_PRESENT) ) +- return pv_l1tf_check_l4e(d, l4e) ? -ERESTART : 1; ++ return pv_l1tf_check_l4e(d, l4e) ? -EINTR : 1; + + if ( unlikely((l4e_get_flags(l4e) & L4_DISALLOW_MASK)) ) + { +@@ -1404,7 +1404,7 @@ static int alloc_l1_table(struct page_info *page) + { + if ( !(l1e_get_flags(pl1e[i]) & _PAGE_PRESENT) ) + { +- ret = pv_l1tf_check_l1e(d, pl1e[i]) ? -ERESTART : 0; ++ ret = pv_l1tf_check_l1e(d, pl1e[i]) ? -EINTR : 0; + if ( ret ) + goto out; + } +-- +2.23.0 + diff --git a/xsa299-4.11-0002-x86-mm-Don-t-re-set-PGT_pinned-on-a-partially-de-val.patch b/xsa299-4.11-0002-x86-mm-Don-t-re-set-PGT_pinned-on-a-partially-de-val.patch new file mode 100644 index 0000000..a369f93 --- /dev/null +++ b/xsa299-4.11-0002-x86-mm-Don-t-re-set-PGT_pinned-on-a-partially-de-val.patch @@ -0,0 +1,99 @@ +From 6bdddd7980eac0cc883945d823986f24682ca47a Mon Sep 17 00:00:00 2001 +From: George Dunlap +Date: Thu, 10 Oct 2019 17:57:49 +0100 +Subject: [PATCH 02/11] x86/mm: Don't re-set PGT_pinned on a partially + de-validated page + +When unpinning pagetables, if an operation is interrupted, +relinquish_memory() re-sets PGT_pinned so that the un-pin will +pickedup again when the hypercall restarts. + +This is appropriate when put_page_and_type_preemptible() returns +-EINTR, which indicates that the page is back in its initial state +(i.e., completely validated). However, for -ERESTART, this leads to a +state where a page has both PGT_pinned and PGT_partial set. + +This happens to work at the moment, although it's not really a +"canonical" state; but in subsequent patches, where we need to make a +distinction in handling between PGT_validated and PGT_partial pages, +this causes issues. + +Move to a "canonical" state by: +- Only re-setting PGT_pinned on -EINTR +- Re-dropping the refcount held by PGT_pinned on -ERESTART + +In the latter case, the PGT_partial bit will be cleared further down +with the rest of the other PGT_partial pages. + +While here, clean up some trainling whitespace. + +This is part of XSA-299. + +Reported-by: George Dunlap +Signed-off-by: George Dunlap +Reviewed-by: Jan Beulich +--- + xen/arch/x86/domain.c | 31 ++++++++++++++++++++++++++++--- + 1 file changed, 28 insertions(+), 3 deletions(-) + +diff --git a/xen/arch/x86/domain.c b/xen/arch/x86/domain.c +index 29f892c04c..8fbecbb169 100644 +--- a/xen/arch/x86/domain.c ++++ b/xen/arch/x86/domain.c +@@ -112,7 +112,7 @@ static void play_dead(void) + * this case, heap corruption or #PF can occur (when heap debugging is + * enabled). For example, even printk() can involve tasklet scheduling, + * which touches per-cpu vars. +- * ++ * + * Consider very carefully when adding code to *dead_idle. Most hypervisor + * subsystems are unsafe to call. + */ +@@ -1838,9 +1838,34 @@ static int relinquish_memory( + break; + case -ERESTART: + case -EINTR: ++ /* ++ * -EINTR means PGT_validated has been re-set; re-set ++ * PGT_pinned again so that it gets picked up next time ++ * around. ++ * ++ * -ERESTART, OTOH, means PGT_partial is set instead. Put ++ * it back on the list, but don't set PGT_pinned; the ++ * section below will finish off de-validation. But we do ++ * need to drop the general ref associated with ++ * PGT_pinned, since put_page_and_type_preemptible() ++ * didn't do it. ++ * ++ * NB we can do an ASSERT for PGT_validated, since we ++ * "own" the type ref; but theoretically, the PGT_partial ++ * could be cleared by someone else. ++ */ ++ if ( ret == -EINTR ) ++ { ++ ASSERT(page->u.inuse.type_info & PGT_validated); ++ set_bit(_PGT_pinned, &page->u.inuse.type_info); ++ } ++ else ++ put_page(page); ++ + ret = -ERESTART; ++ ++ /* Put the page back on the list and drop the ref we grabbed above */ + page_list_add(page, list); +- set_bit(_PGT_pinned, &page->u.inuse.type_info); + put_page(page); + goto out; + default: +@@ -2062,7 +2087,7 @@ void vcpu_kick(struct vcpu *v) + * pending flag. These values may fluctuate (after all, we hold no + * locks) but the key insight is that each change will cause + * evtchn_upcall_pending to be polled. +- * ++ * + * NB2. We save the running flag across the unblock to avoid a needless + * IPI for domains that we IPI'd to unblock. + */ +-- +2.23.0 + diff --git a/xsa299-4.11-0003-x86-mm-Separate-out-partial_pte-tristate-into-indivi.patch b/xsa299-4.11-0003-x86-mm-Separate-out-partial_pte-tristate-into-indivi.patch new file mode 100644 index 0000000..fa6914b --- /dev/null +++ b/xsa299-4.11-0003-x86-mm-Separate-out-partial_pte-tristate-into-indivi.patch @@ -0,0 +1,609 @@ +From 7c0a37005f52d10903ce22851b52ae9b6f4f0ee2 Mon Sep 17 00:00:00 2001 +From: George Dunlap +Date: Thu, 10 Oct 2019 17:57:49 +0100 +Subject: [PATCH 03/11] x86/mm: Separate out partial_pte tristate into + individual flags + +At the moment, partial_pte is a tri-state that contains two distinct bits +of information: + +1. If zero, the pte at index [nr_validated_ptes] is un-validated. If + non-zero, the pte was last seen with PGT_partial set. + +2. If positive, the pte at index [nr_validated_ptes] does not hold a + general reference count. If negative, it does. + +To make future patches more clear, separate out this functionality +into two distinct, named bits: PTF_partial_set (for #1) and +PTF_partial_general_ref (for #2). + +Additionally, a number of functions which need this information also +take other flags to control behavior (such as `preemptible` and +`defer`). These are hard to read in the caller (since you only see +'true' or 'false'), and ugly when many are added together. In +preparation for adding yet another flag in a future patch, collapse +all of these into a single `flag` variable. + +NB that this does mean checking for what was previously the '-1' +condition a bit more ugly in the put_page_from_lNe functions (since +you have to check for both partial_set and general ref); but this +clause will go away in a future patch. + +Also note that the original comment had an off-by-one error: +partial_flags (like partial_pte before it) concerns +plNe[nr_validated_ptes], not plNe[nr_validated_ptes+1]. + +No functional change intended. + +This is part of XSA-299. + +Reported-by: George Dunlap +Signed-off-by: George Dunlap +Reviewed-by: Jan Beulich +--- + xen/arch/x86/mm.c | 164 +++++++++++++++++++++++---------------- + xen/include/asm-x86/mm.h | 41 ++++++---- + 2 files changed, 127 insertions(+), 78 deletions(-) + +diff --git a/xen/arch/x86/mm.c b/xen/arch/x86/mm.c +index 8ced185b49..1c4f54e328 100644 +--- a/xen/arch/x86/mm.c ++++ b/xen/arch/x86/mm.c +@@ -610,20 +610,34 @@ static int alloc_segdesc_page(struct page_info *page) + static int _get_page_type(struct page_info *page, unsigned long type, + bool preemptible); + ++/* ++ * The following flags are used to specify behavior of various get and ++ * put commands. The first two are also stored in page->partial_flags ++ * to indicate the state of the page pointed to by ++ * page->pte[page->nr_validated_entries]. See the comment in mm.h for ++ * more information. ++ */ ++#define PTF_partial_set (1 << 0) ++#define PTF_partial_general_ref (1 << 1) ++#define PTF_preemptible (1 << 2) ++#define PTF_defer (1 << 3) ++ + static int get_page_and_type_from_mfn( + mfn_t mfn, unsigned long type, struct domain *d, +- int partial, int preemptible) ++ unsigned int flags) + { + struct page_info *page = mfn_to_page(mfn); + int rc; ++ bool preemptible = flags & PTF_preemptible, ++ partial_ref = flags & PTF_partial_general_ref; + +- if ( likely(partial >= 0) && ++ if ( likely(!partial_ref) && + unlikely(!get_page_from_mfn(mfn, d)) ) + return -EINVAL; + + rc = _get_page_type(page, type, preemptible); + +- if ( unlikely(rc) && partial >= 0 && ++ if ( unlikely(rc) && !partial_ref && + (!preemptible || page != current->arch.old_guest_table) ) + put_page(page); + +@@ -1104,7 +1118,7 @@ get_page_from_l1e( + define_get_linear_pagetable(l2); + static int + get_page_from_l2e( +- l2_pgentry_t l2e, unsigned long pfn, struct domain *d, int partial) ++ l2_pgentry_t l2e, unsigned long pfn, struct domain *d, unsigned int flags) + { + unsigned long mfn = l2e_get_pfn(l2e); + int rc; +@@ -1119,8 +1133,9 @@ get_page_from_l2e( + return -EINVAL; + } + +- rc = get_page_and_type_from_mfn(_mfn(mfn), PGT_l1_page_table, d, +- partial, false); ++ ASSERT(!(flags & PTF_preemptible)); ++ ++ rc = get_page_and_type_from_mfn(_mfn(mfn), PGT_l1_page_table, d, flags); + if ( unlikely(rc == -EINVAL) && get_l2_linear_pagetable(l2e, pfn, d) ) + rc = 0; + +@@ -1137,7 +1152,7 @@ get_page_from_l2e( + define_get_linear_pagetable(l3); + static int + get_page_from_l3e( +- l3_pgentry_t l3e, unsigned long pfn, struct domain *d, int partial) ++ l3_pgentry_t l3e, unsigned long pfn, struct domain *d, unsigned int flags) + { + int rc; + +@@ -1152,7 +1167,7 @@ get_page_from_l3e( + } + + rc = get_page_and_type_from_mfn( +- l3e_get_mfn(l3e), PGT_l2_page_table, d, partial, 1); ++ l3e_get_mfn(l3e), PGT_l2_page_table, d, flags | PTF_preemptible); + if ( unlikely(rc == -EINVAL) && + !is_pv_32bit_domain(d) && + get_l3_linear_pagetable(l3e, pfn, d) ) +@@ -1170,7 +1185,7 @@ get_page_from_l3e( + define_get_linear_pagetable(l4); + static int + get_page_from_l4e( +- l4_pgentry_t l4e, unsigned long pfn, struct domain *d, int partial) ++ l4_pgentry_t l4e, unsigned long pfn, struct domain *d, unsigned int flags) + { + int rc; + +@@ -1185,7 +1200,7 @@ get_page_from_l4e( + } + + rc = get_page_and_type_from_mfn( +- l4e_get_mfn(l4e), PGT_l3_page_table, d, partial, 1); ++ l4e_get_mfn(l4e), PGT_l3_page_table, d, flags | PTF_preemptible); + if ( unlikely(rc == -EINVAL) && get_l4_linear_pagetable(l4e, pfn, d) ) + rc = 0; + +@@ -1275,7 +1290,7 @@ void put_page_from_l1e(l1_pgentry_t l1e, struct domain *l1e_owner) + * Note also that this automatically deals correctly with linear p.t.'s. + */ + static int put_page_from_l2e(l2_pgentry_t l2e, unsigned long pfn, +- int partial, bool defer) ++ unsigned int flags) + { + int rc = 0; + +@@ -1295,12 +1310,13 @@ static int put_page_from_l2e(l2_pgentry_t l2e, unsigned long pfn, + struct page_info *pg = l2e_get_page(l2e); + struct page_info *ptpg = mfn_to_page(_mfn(pfn)); + +- if ( unlikely(partial > 0) ) ++ if ( (flags & (PTF_partial_set | PTF_partial_general_ref)) == ++ PTF_partial_set ) + { +- ASSERT(!defer); ++ ASSERT(!(flags & PTF_defer)); + rc = _put_page_type(pg, true, ptpg); + } +- else if ( defer ) ++ else if ( flags & PTF_defer ) + { + current->arch.old_guest_ptpg = ptpg; + current->arch.old_guest_table = pg; +@@ -1317,7 +1333,7 @@ static int put_page_from_l2e(l2_pgentry_t l2e, unsigned long pfn, + } + + static int put_page_from_l3e(l3_pgentry_t l3e, unsigned long pfn, +- int partial, bool defer) ++ unsigned int flags) + { + struct page_info *pg; + int rc; +@@ -1340,13 +1356,14 @@ static int put_page_from_l3e(l3_pgentry_t l3e, unsigned long pfn, + + pg = l3e_get_page(l3e); + +- if ( unlikely(partial > 0) ) ++ if ( (flags & (PTF_partial_set | PTF_partial_general_ref)) == ++ PTF_partial_set ) + { +- ASSERT(!defer); ++ ASSERT(!(flags & PTF_defer)); + return _put_page_type(pg, true, mfn_to_page(_mfn(pfn))); + } + +- if ( defer ) ++ if ( flags & PTF_defer ) + { + current->arch.old_guest_ptpg = mfn_to_page(_mfn(pfn)); + current->arch.old_guest_table = pg; +@@ -1361,7 +1378,7 @@ static int put_page_from_l3e(l3_pgentry_t l3e, unsigned long pfn, + } + + static int put_page_from_l4e(l4_pgentry_t l4e, unsigned long pfn, +- int partial, bool defer) ++ unsigned int flags) + { + int rc = 1; + +@@ -1370,13 +1387,14 @@ static int put_page_from_l4e(l4_pgentry_t l4e, unsigned long pfn, + { + struct page_info *pg = l4e_get_page(l4e); + +- if ( unlikely(partial > 0) ) ++ if ( (flags & (PTF_partial_set | PTF_partial_general_ref)) == ++ PTF_partial_set ) + { +- ASSERT(!defer); ++ ASSERT(!(flags & PTF_defer)); + return _put_page_type(pg, true, mfn_to_page(_mfn(pfn))); + } + +- if ( defer ) ++ if ( flags & PTF_defer ) + { + current->arch.old_guest_ptpg = mfn_to_page(_mfn(pfn)); + current->arch.old_guest_table = pg; +@@ -1483,12 +1501,13 @@ static int alloc_l2_table(struct page_info *page, unsigned long type) + unsigned long pfn = mfn_x(page_to_mfn(page)); + l2_pgentry_t *pl2e; + unsigned int i; +- int rc = 0, partial = page->partial_pte; ++ int rc = 0; ++ unsigned int partial_flags = page->partial_flags; + + pl2e = map_domain_page(_mfn(pfn)); + + for ( i = page->nr_validated_ptes; i < L2_PAGETABLE_ENTRIES; +- i++, partial = 0 ) ++ i++, partial_flags = 0 ) + { + if ( i > page->nr_validated_ptes && hypercall_preempt_check() ) + { +@@ -1498,18 +1517,19 @@ static int alloc_l2_table(struct page_info *page, unsigned long type) + } + + if ( !is_guest_l2_slot(d, type, i) || +- (rc = get_page_from_l2e(pl2e[i], pfn, d, partial)) > 0 ) ++ (rc = get_page_from_l2e(pl2e[i], pfn, d, partial_flags)) > 0 ) + continue; + + if ( rc == -ERESTART ) + { + page->nr_validated_ptes = i; +- page->partial_pte = partial ?: 1; ++ /* Set 'set', retain 'general ref' */ ++ page->partial_flags = partial_flags | PTF_partial_set; + } + else if ( rc == -EINTR && i ) + { + page->nr_validated_ptes = i; +- page->partial_pte = 0; ++ page->partial_flags = 0; + rc = -ERESTART; + } + else if ( rc < 0 && rc != -EINTR ) +@@ -1518,7 +1538,7 @@ static int alloc_l2_table(struct page_info *page, unsigned long type) + if ( i ) + { + page->nr_validated_ptes = i; +- page->partial_pte = 0; ++ page->partial_flags = 0; + current->arch.old_guest_ptpg = NULL; + current->arch.old_guest_table = page; + } +@@ -1542,7 +1562,8 @@ static int alloc_l3_table(struct page_info *page) + unsigned long pfn = mfn_x(page_to_mfn(page)); + l3_pgentry_t *pl3e; + unsigned int i; +- int rc = 0, partial = page->partial_pte; ++ int rc = 0; ++ unsigned int partial_flags = page->partial_flags; + + pl3e = map_domain_page(_mfn(pfn)); + +@@ -1557,7 +1578,7 @@ static int alloc_l3_table(struct page_info *page) + memset(pl3e + 4, 0, (L3_PAGETABLE_ENTRIES - 4) * sizeof(*pl3e)); + + for ( i = page->nr_validated_ptes; i < L3_PAGETABLE_ENTRIES; +- i++, partial = 0 ) ++ i++, partial_flags = 0 ) + { + if ( i > page->nr_validated_ptes && hypercall_preempt_check() ) + { +@@ -1574,20 +1595,22 @@ static int alloc_l3_table(struct page_info *page) + else + rc = get_page_and_type_from_mfn( + l3e_get_mfn(pl3e[i]), +- PGT_l2_page_table | PGT_pae_xen_l2, d, partial, 1); ++ PGT_l2_page_table | PGT_pae_xen_l2, d, ++ partial_flags | PTF_preemptible); + } +- else if ( (rc = get_page_from_l3e(pl3e[i], pfn, d, partial)) > 0 ) ++ else if ( (rc = get_page_from_l3e(pl3e[i], pfn, d, partial_flags)) > 0 ) + continue; + + if ( rc == -ERESTART ) + { + page->nr_validated_ptes = i; +- page->partial_pte = partial ?: 1; ++ /* Set 'set', leave 'general ref' set if this entry was set */ ++ page->partial_flags = partial_flags | PTF_partial_set; + } + else if ( rc == -EINTR && i ) + { + page->nr_validated_ptes = i; +- page->partial_pte = 0; ++ page->partial_flags = 0; + rc = -ERESTART; + } + if ( rc < 0 ) +@@ -1604,7 +1627,7 @@ static int alloc_l3_table(struct page_info *page) + if ( i ) + { + page->nr_validated_ptes = i; +- page->partial_pte = 0; ++ page->partial_flags = 0; + current->arch.old_guest_ptpg = NULL; + current->arch.old_guest_table = page; + } +@@ -1736,19 +1759,21 @@ static int alloc_l4_table(struct page_info *page) + unsigned long pfn = mfn_x(page_to_mfn(page)); + l4_pgentry_t *pl4e = map_domain_page(_mfn(pfn)); + unsigned int i; +- int rc = 0, partial = page->partial_pte; ++ int rc = 0; ++ unsigned int partial_flags = page->partial_flags; + + for ( i = page->nr_validated_ptes; i < L4_PAGETABLE_ENTRIES; +- i++, partial = 0 ) ++ i++, partial_flags = 0 ) + { + if ( !is_guest_l4_slot(d, i) || +- (rc = get_page_from_l4e(pl4e[i], pfn, d, partial)) > 0 ) ++ (rc = get_page_from_l4e(pl4e[i], pfn, d, partial_flags)) > 0 ) + continue; + + if ( rc == -ERESTART ) + { + page->nr_validated_ptes = i; +- page->partial_pte = partial ?: 1; ++ /* Set 'set', leave 'general ref' set if this entry was set */ ++ page->partial_flags = partial_flags | PTF_partial_set; + } + else if ( rc < 0 ) + { +@@ -1758,7 +1783,7 @@ static int alloc_l4_table(struct page_info *page) + if ( i ) + { + page->nr_validated_ptes = i; +- page->partial_pte = 0; ++ page->partial_flags = 0; + if ( rc == -EINTR ) + rc = -ERESTART; + else +@@ -1811,19 +1836,20 @@ static int free_l2_table(struct page_info *page) + struct domain *d = page_get_owner(page); + unsigned long pfn = mfn_x(page_to_mfn(page)); + l2_pgentry_t *pl2e; +- int rc = 0, partial = page->partial_pte; +- unsigned int i = page->nr_validated_ptes - !partial; ++ int rc = 0; ++ unsigned int partial_flags = page->partial_flags, ++ i = page->nr_validated_ptes - !(partial_flags & PTF_partial_set); + + pl2e = map_domain_page(_mfn(pfn)); + + for ( ; ; ) + { + if ( is_guest_l2_slot(d, page->u.inuse.type_info, i) ) +- rc = put_page_from_l2e(pl2e[i], pfn, partial, false); ++ rc = put_page_from_l2e(pl2e[i], pfn, partial_flags); + if ( rc < 0 ) + break; + +- partial = 0; ++ partial_flags = 0; + + if ( !i-- ) + break; +@@ -1845,12 +1871,14 @@ static int free_l2_table(struct page_info *page) + else if ( rc == -ERESTART ) + { + page->nr_validated_ptes = i; +- page->partial_pte = partial ?: -1; ++ page->partial_flags = (partial_flags & PTF_partial_set) ? ++ partial_flags : ++ (PTF_partial_set | PTF_partial_general_ref); + } + else if ( rc == -EINTR && i < L2_PAGETABLE_ENTRIES - 1 ) + { + page->nr_validated_ptes = i + 1; +- page->partial_pte = 0; ++ page->partial_flags = 0; + rc = -ERESTART; + } + +@@ -1862,18 +1890,19 @@ static int free_l3_table(struct page_info *page) + struct domain *d = page_get_owner(page); + unsigned long pfn = mfn_x(page_to_mfn(page)); + l3_pgentry_t *pl3e; +- int rc = 0, partial = page->partial_pte; +- unsigned int i = page->nr_validated_ptes - !partial; ++ int rc = 0; ++ unsigned int partial_flags = page->partial_flags, ++ i = page->nr_validated_ptes - !(partial_flags & PTF_partial_set); + + pl3e = map_domain_page(_mfn(pfn)); + + for ( ; ; ) + { +- rc = put_page_from_l3e(pl3e[i], pfn, partial, 0); ++ rc = put_page_from_l3e(pl3e[i], pfn, partial_flags); + if ( rc < 0 ) + break; + +- partial = 0; ++ partial_flags = 0; + if ( rc == 0 ) + pl3e[i] = unadjust_guest_l3e(pl3e[i], d); + +@@ -1892,12 +1921,14 @@ static int free_l3_table(struct page_info *page) + if ( rc == -ERESTART ) + { + page->nr_validated_ptes = i; +- page->partial_pte = partial ?: -1; ++ page->partial_flags = (partial_flags & PTF_partial_set) ? ++ partial_flags : ++ (PTF_partial_set | PTF_partial_general_ref); + } + else if ( rc == -EINTR && i < L3_PAGETABLE_ENTRIES - 1 ) + { + page->nr_validated_ptes = i + 1; +- page->partial_pte = 0; ++ page->partial_flags = 0; + rc = -ERESTART; + } + return rc > 0 ? 0 : rc; +@@ -1908,26 +1939,29 @@ static int free_l4_table(struct page_info *page) + struct domain *d = page_get_owner(page); + unsigned long pfn = mfn_x(page_to_mfn(page)); + l4_pgentry_t *pl4e = map_domain_page(_mfn(pfn)); +- int rc = 0, partial = page->partial_pte; +- unsigned int i = page->nr_validated_ptes - !partial; ++ int rc = 0; ++ unsigned partial_flags = page->partial_flags, ++ i = page->nr_validated_ptes - !(partial_flags & PTF_partial_set); + + do { + if ( is_guest_l4_slot(d, i) ) +- rc = put_page_from_l4e(pl4e[i], pfn, partial, 0); ++ rc = put_page_from_l4e(pl4e[i], pfn, partial_flags); + if ( rc < 0 ) + break; +- partial = 0; ++ partial_flags = 0; + } while ( i-- ); + + if ( rc == -ERESTART ) + { + page->nr_validated_ptes = i; +- page->partial_pte = partial ?: -1; ++ page->partial_flags = (partial_flags & PTF_partial_set) ? ++ partial_flags : ++ (PTF_partial_set | PTF_partial_general_ref); + } + else if ( rc == -EINTR && i < L4_PAGETABLE_ENTRIES - 1 ) + { + page->nr_validated_ptes = i + 1; +- page->partial_pte = 0; ++ page->partial_flags = 0; + rc = -ERESTART; + } + +@@ -2203,7 +2237,7 @@ static int mod_l2_entry(l2_pgentry_t *pl2e, + return -EBUSY; + } + +- put_page_from_l2e(ol2e, pfn, 0, true); ++ put_page_from_l2e(ol2e, pfn, PTF_defer); + + return rc; + } +@@ -2271,7 +2305,7 @@ static int mod_l3_entry(l3_pgentry_t *pl3e, + if ( !create_pae_xen_mappings(d, pl3e) ) + BUG(); + +- put_page_from_l3e(ol3e, pfn, 0, 1); ++ put_page_from_l3e(ol3e, pfn, PTF_defer); + return rc; + } + +@@ -2334,7 +2368,7 @@ static int mod_l4_entry(l4_pgentry_t *pl4e, + return -EFAULT; + } + +- put_page_from_l4e(ol4e, pfn, 0, 1); ++ put_page_from_l4e(ol4e, pfn, PTF_defer); + return rc; + } + +@@ -2598,7 +2632,7 @@ int free_page_type(struct page_info *page, unsigned long type, + if ( !(type & PGT_partial) ) + { + page->nr_validated_ptes = 1U << PAGETABLE_ORDER; +- page->partial_pte = 0; ++ page->partial_flags = 0; + } + + switch ( type & PGT_type_mask ) +@@ -2889,7 +2923,7 @@ static int _get_page_type(struct page_info *page, unsigned long type, + if ( !(x & PGT_partial) ) + { + page->nr_validated_ptes = 0; +- page->partial_pte = 0; ++ page->partial_flags = 0; + } + page->linear_pt_count = 0; + rc = alloc_page_type(page, type, preemptible); +@@ -3064,7 +3098,7 @@ int new_guest_cr3(mfn_t mfn) + return 0; + } + +- rc = get_page_and_type_from_mfn(mfn, PGT_root_page_table, d, 0, 1); ++ rc = get_page_and_type_from_mfn(mfn, PGT_root_page_table, d, PTF_preemptible); + switch ( rc ) + { + case 0: +@@ -3452,7 +3486,7 @@ long do_mmuext_op( + if ( op.arg1.mfn != 0 ) + { + rc = get_page_and_type_from_mfn( +- _mfn(op.arg1.mfn), PGT_root_page_table, currd, 0, 1); ++ _mfn(op.arg1.mfn), PGT_root_page_table, currd, PTF_preemptible); + + if ( unlikely(rc) ) + { +diff --git a/xen/include/asm-x86/mm.h b/xen/include/asm-x86/mm.h +index 1ea173c555..46cba52941 100644 +--- a/xen/include/asm-x86/mm.h ++++ b/xen/include/asm-x86/mm.h +@@ -228,19 +228,34 @@ struct page_info + * setting the flag must not drop that reference, whereas the instance + * clearing it will have to. + * +- * If @partial_pte is positive then PTE at @nr_validated_ptes+1 has +- * been partially validated. This implies that the general reference +- * to the page (acquired from get_page_from_lNe()) would be dropped +- * (again due to the apparent failure) and hence must be re-acquired +- * when resuming the validation, but must not be dropped when picking +- * up the page for invalidation. ++ * If partial_flags & PTF_partial_set is set, then the page at ++ * at @nr_validated_ptes had PGT_partial set as a result of an ++ * operation on the current page. (That page may or may not ++ * still have PGT_partial set.) + * +- * If @partial_pte is negative then PTE at @nr_validated_ptes+1 has +- * been partially invalidated. This is basically the opposite case of +- * above, i.e. the general reference to the page was not dropped in +- * put_page_from_lNe() (due to the apparent failure), and hence it +- * must be dropped when the put operation is resumed (and completes), +- * but it must not be acquired if picking up the page for validation. ++ * If PTF_partial_general_ref is set, then the PTE at ++ * @nr_validated_ptef holds a general reference count for the ++ * page. ++ * ++ * This happens: ++ * - During de-validation, if de-validation of the page was ++ * interrupted ++ * - During validation, if an invalid entry is encountered and ++ * validation is preemptible ++ * - During validation, if PTF_partial_general_ref was set on ++ * this entry to begin with (perhaps because we're picking ++ * up from a partial de-validation). ++ * ++ * When resuming validation, if PTF_partial_general_ref is clear, ++ * then a general reference must be re-acquired; if it is set, no ++ * reference should be acquired. ++ * ++ * When resuming de-validation, if PTF_partial_general_ref is ++ * clear, no reference should be dropped; if it is set, a ++ * reference should be dropped. ++ * ++ * NB that PTF_partial_set and PTF_partial_general_ref are ++ * defined in mm.c, the only place where they are used. + * + * The 3rd field, @linear_pt_count, indicates + * - by a positive value, how many same-level page table entries a page +@@ -251,7 +266,7 @@ struct page_info + struct { + u16 nr_validated_ptes:PAGETABLE_ORDER + 1; + u16 :16 - PAGETABLE_ORDER - 1 - 2; +- s16 partial_pte:2; ++ u16 partial_flags:2; + s16 linear_pt_count; + }; + +-- +2.23.0 + diff --git a/xsa299-4.11-0004-x86-mm-Use-flags-for-_put_page_type-rather-than-a-bo.patch b/xsa299-4.11-0004-x86-mm-Use-flags-for-_put_page_type-rather-than-a-bo.patch new file mode 100644 index 0000000..767ebc1 --- /dev/null +++ b/xsa299-4.11-0004-x86-mm-Use-flags-for-_put_page_type-rather-than-a-bo.patch @@ -0,0 +1,140 @@ +From 20b8a6702c6839bafd252789396b443d4b5c5474 Mon Sep 17 00:00:00 2001 +From: George Dunlap +Date: Thu, 10 Oct 2019 17:57:49 +0100 +Subject: [PATCH 04/11] x86/mm: Use flags for _put_page_type rather than a + boolean + +This is in mainly in preparation for _put_page_type taking the +partial_flags value in the future. It also makes it easier to read in +the caller (since you see a flag name rather than `true` or `false`). + +No functional change intended. + +This is part of XSA-299. + +Reported-by: George Dunlap +Signed-off-by: George Dunlap +Reviewed-by: Jan Beulich +--- + xen/arch/x86/mm.c | 25 +++++++++++++------------ + 1 file changed, 13 insertions(+), 12 deletions(-) + +diff --git a/xen/arch/x86/mm.c b/xen/arch/x86/mm.c +index 1c4f54e328..e2fba15d86 100644 +--- a/xen/arch/x86/mm.c ++++ b/xen/arch/x86/mm.c +@@ -1207,7 +1207,7 @@ get_page_from_l4e( + return rc; + } + +-static int _put_page_type(struct page_info *page, bool preemptible, ++static int _put_page_type(struct page_info *page, unsigned int flags, + struct page_info *ptpg); + + void put_page_from_l1e(l1_pgentry_t l1e, struct domain *l1e_owner) +@@ -1314,7 +1314,7 @@ static int put_page_from_l2e(l2_pgentry_t l2e, unsigned long pfn, + PTF_partial_set ) + { + ASSERT(!(flags & PTF_defer)); +- rc = _put_page_type(pg, true, ptpg); ++ rc = _put_page_type(pg, PTF_preemptible, ptpg); + } + else if ( flags & PTF_defer ) + { +@@ -1323,7 +1323,7 @@ static int put_page_from_l2e(l2_pgentry_t l2e, unsigned long pfn, + } + else + { +- rc = _put_page_type(pg, true, ptpg); ++ rc = _put_page_type(pg, PTF_preemptible, ptpg); + if ( likely(!rc) ) + put_page(pg); + } +@@ -1360,7 +1360,7 @@ static int put_page_from_l3e(l3_pgentry_t l3e, unsigned long pfn, + PTF_partial_set ) + { + ASSERT(!(flags & PTF_defer)); +- return _put_page_type(pg, true, mfn_to_page(_mfn(pfn))); ++ return _put_page_type(pg, PTF_preemptible, mfn_to_page(_mfn(pfn))); + } + + if ( flags & PTF_defer ) +@@ -1370,7 +1370,7 @@ static int put_page_from_l3e(l3_pgentry_t l3e, unsigned long pfn, + return 0; + } + +- rc = _put_page_type(pg, true, mfn_to_page(_mfn(pfn))); ++ rc = _put_page_type(pg, PTF_preemptible, mfn_to_page(_mfn(pfn))); + if ( likely(!rc) ) + put_page(pg); + +@@ -1391,7 +1391,7 @@ static int put_page_from_l4e(l4_pgentry_t l4e, unsigned long pfn, + PTF_partial_set ) + { + ASSERT(!(flags & PTF_defer)); +- return _put_page_type(pg, true, mfn_to_page(_mfn(pfn))); ++ return _put_page_type(pg, PTF_preemptible, mfn_to_page(_mfn(pfn))); + } + + if ( flags & PTF_defer ) +@@ -1401,7 +1401,7 @@ static int put_page_from_l4e(l4_pgentry_t l4e, unsigned long pfn, + return 0; + } + +- rc = _put_page_type(pg, true, mfn_to_page(_mfn(pfn))); ++ rc = _put_page_type(pg, PTF_preemptible, mfn_to_page(_mfn(pfn))); + if ( likely(!rc) ) + put_page(pg); + } +@@ -2701,10 +2701,11 @@ static int _put_final_page_type(struct page_info *page, unsigned long type, + } + + +-static int _put_page_type(struct page_info *page, bool preemptible, ++static int _put_page_type(struct page_info *page, unsigned int flags, + struct page_info *ptpg) + { + unsigned long nx, x, y = page->u.inuse.type_info; ++ bool preemptible = flags & PTF_preemptible; + + ASSERT(current_locked_page_ne_check(page)); + +@@ -2911,7 +2912,7 @@ static int _get_page_type(struct page_info *page, unsigned long type, + + if ( unlikely(iommu_ret) ) + { +- _put_page_type(page, false, NULL); ++ _put_page_type(page, 0, NULL); + rc = iommu_ret; + goto out; + } +@@ -2938,7 +2939,7 @@ static int _get_page_type(struct page_info *page, unsigned long type, + + void put_page_type(struct page_info *page) + { +- int rc = _put_page_type(page, false, NULL); ++ int rc = _put_page_type(page, 0, NULL); + ASSERT(rc == 0); + (void)rc; + } +@@ -2955,7 +2956,7 @@ int get_page_type(struct page_info *page, unsigned long type) + + int put_page_type_preemptible(struct page_info *page) + { +- return _put_page_type(page, true, NULL); ++ return _put_page_type(page, PTF_preemptible, NULL); + } + + int get_page_type_preemptible(struct page_info *page, unsigned long type) +@@ -2972,7 +2973,7 @@ int put_old_guest_table(struct vcpu *v) + if ( !v->arch.old_guest_table ) + return 0; + +- switch ( rc = _put_page_type(v->arch.old_guest_table, true, ++ switch ( rc = _put_page_type(v->arch.old_guest_table, PTF_preemptible, + v->arch.old_guest_ptpg) ) + { + case -EINTR: +-- +2.23.0 + diff --git a/xsa299-4.11-0005-x86-mm-Rework-get_page_and_type_from_mfn-conditional.patch b/xsa299-4.11-0005-x86-mm-Rework-get_page_and_type_from_mfn-conditional.patch new file mode 100644 index 0000000..523ea25 --- /dev/null +++ b/xsa299-4.11-0005-x86-mm-Rework-get_page_and_type_from_mfn-conditional.patch @@ -0,0 +1,79 @@ +From 7b3f9f9a797459902bebba962e31be5cbfe7b515 Mon Sep 17 00:00:00 2001 +From: George Dunlap +Date: Thu, 10 Oct 2019 17:57:49 +0100 +Subject: [PATCH 05/11] x86/mm: Rework get_page_and_type_from_mfn conditional + +Make it easier to read by declaring the conditions in which we will +retain the ref, rather than the conditions under which we release it. + +The only way (page == current->arch.old_guest_table) can be true is if +preemptible is true; so remove this from the query itself, and add an +ASSERT() to that effect on the opposite path. + +No functional change intended. + +NB that alloc_lN_table() mishandle the "linear pt failure" situation +described in the comment; this will be addressed in a future patch. + +This is part of XSA-299. + +Reported-by: George Dunlap +Signed-off-by: George Dunlap +Reviewed-by: Jan Beulich +--- + xen/arch/x86/mm.c | 39 +++++++++++++++++++++++++++++++++++++-- + 1 file changed, 37 insertions(+), 2 deletions(-) + +diff --git a/xen/arch/x86/mm.c b/xen/arch/x86/mm.c +index e2fba15d86..eaf7b14245 100644 +--- a/xen/arch/x86/mm.c ++++ b/xen/arch/x86/mm.c +@@ -637,8 +637,43 @@ static int get_page_and_type_from_mfn( + + rc = _get_page_type(page, type, preemptible); + +- if ( unlikely(rc) && !partial_ref && +- (!preemptible || page != current->arch.old_guest_table) ) ++ /* ++ * Retain the refcount if: ++ * - page is fully validated (rc == 0) ++ * - page is not validated (rc < 0) but: ++ * - We came in with a reference (partial_ref) ++ * - page is partially validated but there's been an error ++ * (page == current->arch.old_guest_table) ++ * ++ * The partial_ref-on-error clause is worth an explanation. There ++ * are two scenarios where partial_ref might be true coming in: ++ * - mfn has been partially demoted as type `type`; i.e. has ++ * PGT_partial set ++ * - mfn has been partially demoted as L(type+1) (i.e., a linear ++ * page; e.g. we're being called from get_page_from_l2e with ++ * type == PGT_l1_table, but the mfn is PGT_l2_table) ++ * ++ * If there's an error, in the first case, _get_page_type will ++ * either return -ERESTART, in which case we want to retain the ++ * ref (as the caller will consider it retained), or -EINVAL, in ++ * which case old_guest_table will be set; in both cases, we need ++ * to retain the ref. ++ * ++ * In the second case, if there's an error, _get_page_type() can ++ * *only* return -EINVAL, and *never* set old_guest_table. In ++ * that case we also want to retain the reference, to allow the ++ * page to continue to be torn down (i.e., PGT_partial cleared) ++ * safely. ++ * ++ * Also note that we shouldn't be able to leave with the reference ++ * count retained unless we succeeded, or the operation was ++ * preemptible. ++ */ ++ if ( likely(!rc) || partial_ref ) ++ /* nothing */; ++ else if ( page == current->arch.old_guest_table ) ++ ASSERT(preemptible); ++ else + put_page(page); + + return rc; +-- +2.23.0 + diff --git a/xsa299-4.11-0006-x86-mm-Have-alloc_l-23-_table-clear-partial_flags-wh.patch b/xsa299-4.11-0006-x86-mm-Have-alloc_l-23-_table-clear-partial_flags-wh.patch new file mode 100644 index 0000000..c611801 --- /dev/null +++ b/xsa299-4.11-0006-x86-mm-Have-alloc_l-23-_table-clear-partial_flags-wh.patch @@ -0,0 +1,101 @@ +From d28893777be56ef51562ed32502377974f738fd3 Mon Sep 17 00:00:00 2001 +From: George Dunlap +Date: Thu, 10 Oct 2019 17:57:49 +0100 +Subject: [PATCH 06/11] x86/mm: Have alloc_l[23]_table clear partial_flags when + preempting + +In order to allow recursive pagetable promotions and demotions to be +interrupted, Xen must keep track of the state of the sub-pages +promoted or demoted. This is stored in two elements in the page +struct: nr_entries_validated and partial_flags. + +The rule is that entries [0, nr_entries_validated) should always be +validated and hold a general reference count. If partial_flags is +zero, then [nr_entries_validated] is not validated and no reference +count is held. If PTF_partial_set is set, then [nr_entries_validated] +is partially validated. + +At the moment, a distinction is made between promotion and demotion +with regard to whether the entry itself "holds" a general reference +count: when entry promotion is interrupted (i.e., returns -ERESTART), +the entry is not considered to hold a reference; when entry demotion +is interrupted, the entry is still considered to hold a general +reference. + +PTF_partial_general_ref is used to distinguish between these cases. +If clear, it's a partial promotion => no general reference count held +by the entry; if set, it's partial demotion, so a general reference +count held. Because promotions and demotions can be interleaved, this +value is passed to get_page_and_type_from_mfn and put_page_from_l*e, +to be able to properly handle reference counts. + +Unfortunately, when alloc_l[23]_table check hypercall_preempt_check() +and return -ERESTART, they set nr_entries_validated, but don't clear +partial_flags. + +If we were picking up from a previously-interrupted promotion, that +means that PTF_partial_set would be set even though +[nr_entries_validated] was not partially validated. This means that +if the page in this state were de-validated, put_page_type() would +erroneously be called on that entry. + +Perhaps worse, if we were racing with a de-validation, then we might +leave both PTF_partial_set and PTF_partial_general_ref; and when +de-validation picked up again, both the type and the general ref would +be erroneously dropped from [nr_entries_validated]. + +In a sense, the real issue here is code duplication. Rather than +duplicate the interruption code, set rc to -EINTR and fall through to +the code which already handles that case correctly. + +Given the logic at this point, it should be impossible for +partial_flags to be non-zero; add an ASSERT() to catch any changes. + +This is part of XSA-299. + +Reported-by: George Dunlap +Signed-off-by: George Dunlap +Reviewed-by: Jan Beulich +--- + xen/arch/x86/mm.c | 18 ++++-------------- + 1 file changed, 4 insertions(+), 14 deletions(-) + +diff --git a/xen/arch/x86/mm.c b/xen/arch/x86/mm.c +index eaf7b14245..053465cb7c 100644 +--- a/xen/arch/x86/mm.c ++++ b/xen/arch/x86/mm.c +@@ -1545,13 +1545,8 @@ static int alloc_l2_table(struct page_info *page, unsigned long type) + i++, partial_flags = 0 ) + { + if ( i > page->nr_validated_ptes && hypercall_preempt_check() ) +- { +- page->nr_validated_ptes = i; +- rc = -ERESTART; +- break; +- } +- +- if ( !is_guest_l2_slot(d, type, i) || ++ rc = -EINTR; ++ else if ( !is_guest_l2_slot(d, type, i) || + (rc = get_page_from_l2e(pl2e[i], pfn, d, partial_flags)) > 0 ) + continue; + +@@ -1616,13 +1611,8 @@ static int alloc_l3_table(struct page_info *page) + i++, partial_flags = 0 ) + { + if ( i > page->nr_validated_ptes && hypercall_preempt_check() ) +- { +- page->nr_validated_ptes = i; +- rc = -ERESTART; +- break; +- } +- +- if ( is_pv_32bit_domain(d) && (i == 3) ) ++ rc = -EINTR; ++ else if ( is_pv_32bit_domain(d) && (i == 3) ) + { + if ( !(l3e_get_flags(pl3e[i]) & _PAGE_PRESENT) || + (l3e_get_flags(pl3e[i]) & l3_disallow_mask(d)) ) +-- +2.23.0 + diff --git a/xsa299-4.11-0007-x86-mm-Always-retain-a-general-ref-on-partial.patch b/xsa299-4.11-0007-x86-mm-Always-retain-a-general-ref-on-partial.patch new file mode 100644 index 0000000..0d54cc5 --- /dev/null +++ b/xsa299-4.11-0007-x86-mm-Always-retain-a-general-ref-on-partial.patch @@ -0,0 +1,374 @@ +From f608a53c25806a7a4318cbe225bc5f5bbf154d69 Mon Sep 17 00:00:00 2001 +From: George Dunlap +Date: Thu, 10 Oct 2019 17:57:49 +0100 +Subject: [PATCH 07/11] x86/mm: Always retain a general ref on partial + +In order to allow recursive pagetable promotions and demotions to be +interrupted, Xen must keep track of the state of the sub-pages +promoted or demoted. This is stored in two elements in the page struct: +nr_entries_validated and partial_flags. + +The rule is that entries [0, nr_entries_validated) should always be +validated and hold a general reference count. If partial_flags is +zero, then [nr_entries_validated] is not validated and no reference +count is held. If PTF_partial_set is set, then [nr_entries_validated] +is partially validated. + +At the moment, a distinction is made between promotion and demotion +with regard to whether the entry itself "holds" a general reference +count: when entry promotion is interrupted (i.e., returns -ERESTART), +the entry is not considered to hold a reference; when entry demotion +is interrupted, the entry is still considered to hold a general +reference. + +PTF_partial_general_ref is used to distinguish between these cases. +If clear, it's a partial promotion => no general reference count held +by the entry; if set, it's partial demotion, so a general reference +count held. Because promotions and demotions can be interleaved, this +value is passed to get_page_and_type_from_mfn and put_page_from_l*e, +to be able to properly handle reference counts. + +Unfortunately, because a refcount is not held, it is possible to +engineer a situation where PFT_partial_set is set but the page in +question has been assigned to another domain. A sketch is provided in +the appendix. + +Fix this by having the parent page table entry hold a general +reference count whenever PFT_partial_set is set. (For clarity of +change, keep two separate flags. These will be collapsed in a +subsequent changeset.) + +This has two basic implications. On the put_page_from_lNe() side, +this mean that the (partial_set && !partial_ref) case can never happen, +and no longer needs to be special-cased. + +Secondly, because both flags are set together, there's no need to carry over +existing bits from partial_pte. + +(NB there is still another issue with calling _put_page_type() on a +page which had PGT_partial set; that will be handled in a subsequent +patch.) + +On the get_page_and_type_from_mfn() side, we need to distinguish +between callers which hold a reference on partial (i.e., +alloc_lN_table()), and those which do not (new_cr3, PIN_LN_TABLE, and +so on): pass a flag if the type should be retained on interruption. + +NB that since l1 promotion can't be preempted, that get_page_from_l2e +can't return -ERESTART. + +This is part of XSA-299. + +Reported-by: George Dunlap +Signed-off-by: George Dunlap +Reviewed-by: Jan Beulich +----- +* Appendix: Engineering PTF_partial_set while a page belongs to a + foreign domain + +Suppose A is a page which can be promoted to an l3, and B is a page +which can be promoted to an l2, and A[x] points to B. B has +PGC_allocated set but no other general references. + +V1: PIN_L3 A. + A is validated, B is validated. + A.type_count = 1 | PGT_validated | PGT_pinned + B.type_count = 1 | PGT_validated + B.count = 2 | PGC_allocated (A[x] holds a general ref) + +V1: UNPIN A. + A begins de-validation. + Arrange to be interrupted when i < x + V1->old_guest_table = A + V1->old_guest_table_ref_held = false + A.type_count = 1 | PGT_partial + A.nr_validated_entries = i < x + B.type_count = 0 + B.count = 1 | PGC_allocated + +V2: MOD_L4_ENTRY to point some l4e to A. + Picks up re-validation of A. + Arrange to be interrupted halfway through B's validation + B.type_count = 1 | PGT_partial + B.count = 2 | PGC_allocated (PGT_partial holds a general ref) + A.type_count = 1 | PGT_partial + A.nr_validated_entries = x + A.partial_pte = PTF_partial_set + +V3: MOD_L3_ENTRY to point some other l3e (not in A) to B. + Validates B. + B.type_count = 1 | PGT_validated + B.count = 2 | PGC_allocated ("other l3e" holds a general ref) + +V3: MOD_L3_ENTRY to clear l3e pointing to B. + Devalidates B. + B.type_count = 0 + B.count = 1 | PGC_allocated + +V3: decrease_reservation(B) + Clears PGC_allocated + B.count = 0 => B is freed + +B gets assigned to a different domain + +V1: Restarts UNPIN of A + put_old_guest_table(A) + ... + free_l3_table(A) + +Now since A.partial_flags has PTF_partial_set, free_l3_table() will +call put_page_from_l3e() on A[x], which points to B, while B is owned +by another domain. + +If A[x] held a general refcount for B on partial validation, as it does +for partial de-validation, then B would still have a reference count of +1 after PGC_allocated was freed; so B wouldn't be freed until after +put_page_from_l3e() had happend on A[x]. +--- + xen/arch/x86/mm.c | 84 +++++++++++++++++++++++----------------- + xen/include/asm-x86/mm.h | 15 ++++--- + 2 files changed, 58 insertions(+), 41 deletions(-) + +diff --git a/xen/arch/x86/mm.c b/xen/arch/x86/mm.c +index 053465cb7c..68a9e74002 100644 +--- a/xen/arch/x86/mm.c ++++ b/xen/arch/x86/mm.c +@@ -617,10 +617,11 @@ static int _get_page_type(struct page_info *page, unsigned long type, + * page->pte[page->nr_validated_entries]. See the comment in mm.h for + * more information. + */ +-#define PTF_partial_set (1 << 0) +-#define PTF_partial_general_ref (1 << 1) +-#define PTF_preemptible (1 << 2) +-#define PTF_defer (1 << 3) ++#define PTF_partial_set (1 << 0) ++#define PTF_partial_general_ref (1 << 1) ++#define PTF_preemptible (1 << 2) ++#define PTF_defer (1 << 3) ++#define PTF_retain_ref_on_restart (1 << 4) + + static int get_page_and_type_from_mfn( + mfn_t mfn, unsigned long type, struct domain *d, +@@ -629,7 +630,11 @@ static int get_page_and_type_from_mfn( + struct page_info *page = mfn_to_page(mfn); + int rc; + bool preemptible = flags & PTF_preemptible, +- partial_ref = flags & PTF_partial_general_ref; ++ partial_ref = flags & PTF_partial_general_ref, ++ partial_set = flags & PTF_partial_set, ++ retain_ref = flags & PTF_retain_ref_on_restart; ++ ++ ASSERT(partial_ref == partial_set); + + if ( likely(!partial_ref) && + unlikely(!get_page_from_mfn(mfn, d)) ) +@@ -642,13 +647,15 @@ static int get_page_and_type_from_mfn( + * - page is fully validated (rc == 0) + * - page is not validated (rc < 0) but: + * - We came in with a reference (partial_ref) ++ * - page is partially validated (rc == -ERESTART), and the ++ * caller has asked the ref to be retained in that case + * - page is partially validated but there's been an error + * (page == current->arch.old_guest_table) + * + * The partial_ref-on-error clause is worth an explanation. There + * are two scenarios where partial_ref might be true coming in: +- * - mfn has been partially demoted as type `type`; i.e. has +- * PGT_partial set ++ * - mfn has been partially promoted / demoted as type `type`; ++ * i.e. has PGT_partial set + * - mfn has been partially demoted as L(type+1) (i.e., a linear + * page; e.g. we're being called from get_page_from_l2e with + * type == PGT_l1_table, but the mfn is PGT_l2_table) +@@ -671,7 +678,8 @@ static int get_page_and_type_from_mfn( + */ + if ( likely(!rc) || partial_ref ) + /* nothing */; +- else if ( page == current->arch.old_guest_table ) ++ else if ( page == current->arch.old_guest_table || ++ (retain_ref && rc == -ERESTART) ) + ASSERT(preemptible); + else + put_page(page); +@@ -1348,8 +1356,8 @@ static int put_page_from_l2e(l2_pgentry_t l2e, unsigned long pfn, + if ( (flags & (PTF_partial_set | PTF_partial_general_ref)) == + PTF_partial_set ) + { +- ASSERT(!(flags & PTF_defer)); +- rc = _put_page_type(pg, PTF_preemptible, ptpg); ++ /* partial_set should always imply partial_ref */ ++ BUG(); + } + else if ( flags & PTF_defer ) + { +@@ -1394,8 +1402,8 @@ static int put_page_from_l3e(l3_pgentry_t l3e, unsigned long pfn, + if ( (flags & (PTF_partial_set | PTF_partial_general_ref)) == + PTF_partial_set ) + { +- ASSERT(!(flags & PTF_defer)); +- return _put_page_type(pg, PTF_preemptible, mfn_to_page(_mfn(pfn))); ++ /* partial_set should always imply partial_ref */ ++ BUG(); + } + + if ( flags & PTF_defer ) +@@ -1425,8 +1433,8 @@ static int put_page_from_l4e(l4_pgentry_t l4e, unsigned long pfn, + if ( (flags & (PTF_partial_set | PTF_partial_general_ref)) == + PTF_partial_set ) + { +- ASSERT(!(flags & PTF_defer)); +- return _put_page_type(pg, PTF_preemptible, mfn_to_page(_mfn(pfn))); ++ /* partial_set should always imply partial_ref */ ++ BUG(); + } + + if ( flags & PTF_defer ) +@@ -1550,13 +1558,22 @@ static int alloc_l2_table(struct page_info *page, unsigned long type) + (rc = get_page_from_l2e(pl2e[i], pfn, d, partial_flags)) > 0 ) + continue; + +- if ( rc == -ERESTART ) +- { +- page->nr_validated_ptes = i; +- /* Set 'set', retain 'general ref' */ +- page->partial_flags = partial_flags | PTF_partial_set; +- } +- else if ( rc == -EINTR && i ) ++ /* ++ * It shouldn't be possible for get_page_from_l2e to return ++ * -ERESTART, since we never call this with PTF_preemptible. ++ * (alloc_l1_table may return -EINTR on an L1TF-vulnerable ++ * entry.) ++ * ++ * NB that while on a "clean" promotion, we can never get ++ * PGT_partial. It is possible to arrange for an l2e to ++ * contain a partially-devalidated l2; but in that case, both ++ * of the following functions will fail anyway (the first ++ * because the page in question is not an l1; the second ++ * because the page is not fully validated). ++ */ ++ ASSERT(rc != -ERESTART); ++ ++ if ( rc == -EINTR && i ) + { + page->nr_validated_ptes = i; + page->partial_flags = 0; +@@ -1565,6 +1582,7 @@ static int alloc_l2_table(struct page_info *page, unsigned long type) + else if ( rc < 0 && rc != -EINTR ) + { + gdprintk(XENLOG_WARNING, "Failure in alloc_l2_table: slot %#x\n", i); ++ ASSERT(current->arch.old_guest_table == NULL); + if ( i ) + { + page->nr_validated_ptes = i; +@@ -1621,16 +1639,17 @@ static int alloc_l3_table(struct page_info *page) + rc = get_page_and_type_from_mfn( + l3e_get_mfn(pl3e[i]), + PGT_l2_page_table | PGT_pae_xen_l2, d, +- partial_flags | PTF_preemptible); ++ partial_flags | PTF_preemptible | PTF_retain_ref_on_restart); + } +- else if ( (rc = get_page_from_l3e(pl3e[i], pfn, d, partial_flags)) > 0 ) ++ else if ( (rc = get_page_from_l3e(pl3e[i], pfn, d, ++ partial_flags | PTF_retain_ref_on_restart)) > 0 ) + continue; + + if ( rc == -ERESTART ) + { + page->nr_validated_ptes = i; + /* Set 'set', leave 'general ref' set if this entry was set */ +- page->partial_flags = partial_flags | PTF_partial_set; ++ page->partial_flags = PTF_partial_set | PTF_partial_general_ref; + } + else if ( rc == -EINTR && i ) + { +@@ -1791,14 +1810,15 @@ static int alloc_l4_table(struct page_info *page) + i++, partial_flags = 0 ) + { + if ( !is_guest_l4_slot(d, i) || +- (rc = get_page_from_l4e(pl4e[i], pfn, d, partial_flags)) > 0 ) ++ (rc = get_page_from_l4e(pl4e[i], pfn, d, ++ partial_flags | PTF_retain_ref_on_restart)) > 0 ) + continue; + + if ( rc == -ERESTART ) + { + page->nr_validated_ptes = i; + /* Set 'set', leave 'general ref' set if this entry was set */ +- page->partial_flags = partial_flags | PTF_partial_set; ++ page->partial_flags = PTF_partial_set | PTF_partial_general_ref; + } + else if ( rc < 0 ) + { +@@ -1896,9 +1916,7 @@ static int free_l2_table(struct page_info *page) + else if ( rc == -ERESTART ) + { + page->nr_validated_ptes = i; +- page->partial_flags = (partial_flags & PTF_partial_set) ? +- partial_flags : +- (PTF_partial_set | PTF_partial_general_ref); ++ page->partial_flags = PTF_partial_set | PTF_partial_general_ref; + } + else if ( rc == -EINTR && i < L2_PAGETABLE_ENTRIES - 1 ) + { +@@ -1946,9 +1964,7 @@ static int free_l3_table(struct page_info *page) + if ( rc == -ERESTART ) + { + page->nr_validated_ptes = i; +- page->partial_flags = (partial_flags & PTF_partial_set) ? +- partial_flags : +- (PTF_partial_set | PTF_partial_general_ref); ++ page->partial_flags = PTF_partial_set | PTF_partial_general_ref; + } + else if ( rc == -EINTR && i < L3_PAGETABLE_ENTRIES - 1 ) + { +@@ -1979,9 +1995,7 @@ static int free_l4_table(struct page_info *page) + if ( rc == -ERESTART ) + { + page->nr_validated_ptes = i; +- page->partial_flags = (partial_flags & PTF_partial_set) ? +- partial_flags : +- (PTF_partial_set | PTF_partial_general_ref); ++ page->partial_flags = PTF_partial_set | PTF_partial_general_ref; + } + else if ( rc == -EINTR && i < L4_PAGETABLE_ENTRIES - 1 ) + { +diff --git a/xen/include/asm-x86/mm.h b/xen/include/asm-x86/mm.h +index 46cba52941..dc9cb869dd 100644 +--- a/xen/include/asm-x86/mm.h ++++ b/xen/include/asm-x86/mm.h +@@ -238,22 +238,25 @@ struct page_info + * page. + * + * This happens: +- * - During de-validation, if de-validation of the page was ++ * - During validation or de-validation, if the operation was + * interrupted + * - During validation, if an invalid entry is encountered and + * validation is preemptible + * - During validation, if PTF_partial_general_ref was set on +- * this entry to begin with (perhaps because we're picking +- * up from a partial de-validation). ++ * this entry to begin with (perhaps because it picked up a ++ * previous operation) + * +- * When resuming validation, if PTF_partial_general_ref is clear, +- * then a general reference must be re-acquired; if it is set, no +- * reference should be acquired. ++ * When resuming validation, if PTF_partial_general_ref is ++ * clear, then a general reference must be re-acquired; if it ++ * is set, no reference should be acquired. + * + * When resuming de-validation, if PTF_partial_general_ref is + * clear, no reference should be dropped; if it is set, a + * reference should be dropped. + * ++ * NB at the moment, PTF_partial_set should be set if and only if ++ * PTF_partial_general_ref is set. ++ * + * NB that PTF_partial_set and PTF_partial_general_ref are + * defined in mm.c, the only place where they are used. + * +-- +2.23.0 + diff --git a/xsa299-4.11-0008-x86-mm-Collapse-PTF_partial_set-and-PTF_partial_gene.patch b/xsa299-4.11-0008-x86-mm-Collapse-PTF_partial_set-and-PTF_partial_gene.patch new file mode 100644 index 0000000..dd847e8 --- /dev/null +++ b/xsa299-4.11-0008-x86-mm-Collapse-PTF_partial_set-and-PTF_partial_gene.patch @@ -0,0 +1,227 @@ +From 6811df7fb7a1d4bb5a75fec9cf41519b5c86c605 Mon Sep 17 00:00:00 2001 +From: George Dunlap +Date: Thu, 10 Oct 2019 17:57:49 +0100 +Subject: [PATCH 08/11] x86/mm: Collapse PTF_partial_set and + PTF_partial_general_ref into one + +...now that they are equivalent. No functional change intended. + +Reported-by: George Dunlap +Signed-off-by: George Dunlap +Reviewed-by: Jan Beulich +--- + xen/arch/x86/mm.c | 50 +++++++++++----------------------------- + xen/include/asm-x86/mm.h | 29 +++++++++++------------ + 2 files changed, 26 insertions(+), 53 deletions(-) + +diff --git a/xen/arch/x86/mm.c b/xen/arch/x86/mm.c +index 68a9e74002..4970b19aff 100644 +--- a/xen/arch/x86/mm.c ++++ b/xen/arch/x86/mm.c +@@ -612,13 +612,12 @@ static int _get_page_type(struct page_info *page, unsigned long type, + + /* + * The following flags are used to specify behavior of various get and +- * put commands. The first two are also stored in page->partial_flags +- * to indicate the state of the page pointed to by ++ * put commands. The first is also stored in page->partial_flags to ++ * indicate the state of the page pointed to by + * page->pte[page->nr_validated_entries]. See the comment in mm.h for + * more information. + */ + #define PTF_partial_set (1 << 0) +-#define PTF_partial_general_ref (1 << 1) + #define PTF_preemptible (1 << 2) + #define PTF_defer (1 << 3) + #define PTF_retain_ref_on_restart (1 << 4) +@@ -630,13 +629,10 @@ static int get_page_and_type_from_mfn( + struct page_info *page = mfn_to_page(mfn); + int rc; + bool preemptible = flags & PTF_preemptible, +- partial_ref = flags & PTF_partial_general_ref, + partial_set = flags & PTF_partial_set, + retain_ref = flags & PTF_retain_ref_on_restart; + +- ASSERT(partial_ref == partial_set); +- +- if ( likely(!partial_ref) && ++ if ( likely(!partial_set) && + unlikely(!get_page_from_mfn(mfn, d)) ) + return -EINVAL; + +@@ -646,14 +642,14 @@ static int get_page_and_type_from_mfn( + * Retain the refcount if: + * - page is fully validated (rc == 0) + * - page is not validated (rc < 0) but: +- * - We came in with a reference (partial_ref) ++ * - We came in with a reference (partial_set) + * - page is partially validated (rc == -ERESTART), and the + * caller has asked the ref to be retained in that case + * - page is partially validated but there's been an error + * (page == current->arch.old_guest_table) + * +- * The partial_ref-on-error clause is worth an explanation. There +- * are two scenarios where partial_ref might be true coming in: ++ * The partial_set-on-error clause is worth an explanation. There ++ * are two scenarios where partial_set might be true coming in: + * - mfn has been partially promoted / demoted as type `type`; + * i.e. has PGT_partial set + * - mfn has been partially demoted as L(type+1) (i.e., a linear +@@ -676,7 +672,7 @@ static int get_page_and_type_from_mfn( + * count retained unless we succeeded, or the operation was + * preemptible. + */ +- if ( likely(!rc) || partial_ref ) ++ if ( likely(!rc) || partial_set ) + /* nothing */; + else if ( page == current->arch.old_guest_table || + (retain_ref && rc == -ERESTART) ) +@@ -1353,13 +1349,7 @@ static int put_page_from_l2e(l2_pgentry_t l2e, unsigned long pfn, + struct page_info *pg = l2e_get_page(l2e); + struct page_info *ptpg = mfn_to_page(_mfn(pfn)); + +- if ( (flags & (PTF_partial_set | PTF_partial_general_ref)) == +- PTF_partial_set ) +- { +- /* partial_set should always imply partial_ref */ +- BUG(); +- } +- else if ( flags & PTF_defer ) ++ if ( flags & PTF_defer ) + { + current->arch.old_guest_ptpg = ptpg; + current->arch.old_guest_table = pg; +@@ -1399,13 +1389,6 @@ static int put_page_from_l3e(l3_pgentry_t l3e, unsigned long pfn, + + pg = l3e_get_page(l3e); + +- if ( (flags & (PTF_partial_set | PTF_partial_general_ref)) == +- PTF_partial_set ) +- { +- /* partial_set should always imply partial_ref */ +- BUG(); +- } +- + if ( flags & PTF_defer ) + { + current->arch.old_guest_ptpg = mfn_to_page(_mfn(pfn)); +@@ -1430,13 +1413,6 @@ static int put_page_from_l4e(l4_pgentry_t l4e, unsigned long pfn, + { + struct page_info *pg = l4e_get_page(l4e); + +- if ( (flags & (PTF_partial_set | PTF_partial_general_ref)) == +- PTF_partial_set ) +- { +- /* partial_set should always imply partial_ref */ +- BUG(); +- } +- + if ( flags & PTF_defer ) + { + current->arch.old_guest_ptpg = mfn_to_page(_mfn(pfn)); +@@ -1649,7 +1625,7 @@ static int alloc_l3_table(struct page_info *page) + { + page->nr_validated_ptes = i; + /* Set 'set', leave 'general ref' set if this entry was set */ +- page->partial_flags = PTF_partial_set | PTF_partial_general_ref; ++ page->partial_flags = PTF_partial_set; + } + else if ( rc == -EINTR && i ) + { +@@ -1818,7 +1794,7 @@ static int alloc_l4_table(struct page_info *page) + { + page->nr_validated_ptes = i; + /* Set 'set', leave 'general ref' set if this entry was set */ +- page->partial_flags = PTF_partial_set | PTF_partial_general_ref; ++ page->partial_flags = PTF_partial_set; + } + else if ( rc < 0 ) + { +@@ -1916,7 +1892,7 @@ static int free_l2_table(struct page_info *page) + else if ( rc == -ERESTART ) + { + page->nr_validated_ptes = i; +- page->partial_flags = PTF_partial_set | PTF_partial_general_ref; ++ page->partial_flags = PTF_partial_set; + } + else if ( rc == -EINTR && i < L2_PAGETABLE_ENTRIES - 1 ) + { +@@ -1964,7 +1940,7 @@ static int free_l3_table(struct page_info *page) + if ( rc == -ERESTART ) + { + page->nr_validated_ptes = i; +- page->partial_flags = PTF_partial_set | PTF_partial_general_ref; ++ page->partial_flags = PTF_partial_set; + } + else if ( rc == -EINTR && i < L3_PAGETABLE_ENTRIES - 1 ) + { +@@ -1995,7 +1971,7 @@ static int free_l4_table(struct page_info *page) + if ( rc == -ERESTART ) + { + page->nr_validated_ptes = i; +- page->partial_flags = PTF_partial_set | PTF_partial_general_ref; ++ page->partial_flags = PTF_partial_set; + } + else if ( rc == -EINTR && i < L4_PAGETABLE_ENTRIES - 1 ) + { +diff --git a/xen/include/asm-x86/mm.h b/xen/include/asm-x86/mm.h +index dc9cb869dd..c6ba9e4d73 100644 +--- a/xen/include/asm-x86/mm.h ++++ b/xen/include/asm-x86/mm.h +@@ -233,7 +233,7 @@ struct page_info + * operation on the current page. (That page may or may not + * still have PGT_partial set.) + * +- * If PTF_partial_general_ref is set, then the PTE at ++ * Additionally, if PTF_partial_set is set, then the PTE at + * @nr_validated_ptef holds a general reference count for the + * page. + * +@@ -242,23 +242,20 @@ struct page_info + * interrupted + * - During validation, if an invalid entry is encountered and + * validation is preemptible +- * - During validation, if PTF_partial_general_ref was set on +- * this entry to begin with (perhaps because it picked up a ++ * - During validation, if PTF_partial_set was set on this ++ * entry to begin with (perhaps because it picked up a + * previous operation) + * +- * When resuming validation, if PTF_partial_general_ref is +- * clear, then a general reference must be re-acquired; if it +- * is set, no reference should be acquired. ++ * When resuming validation, if PTF_partial_set is clear, then ++ * a general reference must be re-acquired; if it is set, no ++ * reference should be acquired. + * +- * When resuming de-validation, if PTF_partial_general_ref is +- * clear, no reference should be dropped; if it is set, a +- * reference should be dropped. ++ * When resuming de-validation, if PTF_partial_set is clear, ++ * no reference should be dropped; if it is set, a reference ++ * should be dropped. + * +- * NB at the moment, PTF_partial_set should be set if and only if +- * PTF_partial_general_ref is set. +- * +- * NB that PTF_partial_set and PTF_partial_general_ref are +- * defined in mm.c, the only place where they are used. ++ * NB that PTF_partial_set is defined in mm.c, the only place ++ * where it is used. + * + * The 3rd field, @linear_pt_count, indicates + * - by a positive value, how many same-level page table entries a page +@@ -268,8 +265,8 @@ struct page_info + */ + struct { + u16 nr_validated_ptes:PAGETABLE_ORDER + 1; +- u16 :16 - PAGETABLE_ORDER - 1 - 2; +- u16 partial_flags:2; ++ u16 :16 - PAGETABLE_ORDER - 1 - 1; ++ u16 partial_flags:1; + s16 linear_pt_count; + }; + +-- +2.23.0 + diff --git a/xsa299-4.11-0009-x86-mm-Properly-handle-linear-pagetable-promotion-fa.patch b/xsa299-4.11-0009-x86-mm-Properly-handle-linear-pagetable-promotion-fa.patch new file mode 100644 index 0000000..f62d774 --- /dev/null +++ b/xsa299-4.11-0009-x86-mm-Properly-handle-linear-pagetable-promotion-fa.patch @@ -0,0 +1,106 @@ +From a6098b8920b02149220641cb13358e9012b5fc4d Mon Sep 17 00:00:00 2001 +From: George Dunlap +Date: Thu, 10 Oct 2019 17:57:49 +0100 +Subject: [PATCH 09/11] x86/mm: Properly handle linear pagetable promotion + failures + +In order to allow recursive pagetable promotions and demotions to be +interrupted, Xen must keep track of the state of the sub-pages +promoted or demoted. This is stored in two elements in the page +struct: nr_entries_validated and partial_flags. + +The rule is that entries [0, nr_entries_validated) should always be +validated and hold a general reference count. If partial_flags is +zero, then [nr_entries_validated] is not validated and no reference +count is held. If PTF_partial_set is set, then [nr_entries_validated] +is partially validated, and a general reference count is held. + +Unfortunately, in cases where an entry began with PTF_partial_set set, +and get_page_from_lNe() returns -EINVAL, the PTF_partial_set bit is +erroneously dropped. (This scenario can be engineered mainly by the +use of interleaving of promoting and demoting a page which has "linear +pagetable" entries; see the appendix for a sketch.) This means that +we will "leak" a general reference count on the page in question, +preventing the page from being freed. + +Fix this by setting page->partial_flags to the partial_flags local +variable. + +This is part of XSA-299. + +Reported-by: George Dunlap +Signed-off-by: George Dunlap +Reviewed-by: Jan Beulich +----- +Appendix + +Suppose A and B can both be promoted to L2 pages, and A[x] points to B. + +V1: PIN_L2 B. + B.type_count = 1 | PGT_validated + B.count = 2 | PGC_allocated + +V1: MOD_L3_ENTRY pointing something to A. + In the process of validating A[x], grab an extra type / ref on B: + B.type_count = 2 | PGT_validated + B.count = 3 | PGC_allocated + A.type_count = 1 | PGT_validated + A.count = 2 | PGC_allocated + +V1: UNPIN B. + B.type_count = 1 | PGT_validate + B.count = 2 | PGC_allocated + +V1: MOD_L3_ENTRY removing the reference to A. + De-validate A, down to A[x], which points to B. + Drop the final type on B. Arrange to be interrupted. + B.type_count = 1 | PGT_partial + B.count = 2 | PGC_allocated + A.type_count = 1 | PGT_partial + A.nr_validated_entries = x + A.partial_pte = -1 + +V2: MOD_L3_ENTRY adds a reference to A. + +At this point, get_page_from_l2e(A[x]) tries +get_page_and_type_from_mfn(), which fails because it's the wrong type; +and get_l2_linear_pagetable() also fails, because B isn't validated as +an l2 anymore. +--- + xen/arch/x86/mm.c | 6 +++--- + 1 file changed, 3 insertions(+), 3 deletions(-) + +diff --git a/xen/arch/x86/mm.c b/xen/arch/x86/mm.c +index 4970b19aff..cfb7538403 100644 +--- a/xen/arch/x86/mm.c ++++ b/xen/arch/x86/mm.c +@@ -1562,7 +1562,7 @@ static int alloc_l2_table(struct page_info *page, unsigned long type) + if ( i ) + { + page->nr_validated_ptes = i; +- page->partial_flags = 0; ++ page->partial_flags = partial_flags; + current->arch.old_guest_ptpg = NULL; + current->arch.old_guest_table = page; + } +@@ -1647,7 +1647,7 @@ static int alloc_l3_table(struct page_info *page) + if ( i ) + { + page->nr_validated_ptes = i; +- page->partial_flags = 0; ++ page->partial_flags = partial_flags; + current->arch.old_guest_ptpg = NULL; + current->arch.old_guest_table = page; + } +@@ -1804,7 +1804,7 @@ static int alloc_l4_table(struct page_info *page) + if ( i ) + { + page->nr_validated_ptes = i; +- page->partial_flags = 0; ++ page->partial_flags = partial_flags; + if ( rc == -EINTR ) + rc = -ERESTART; + else +-- +2.23.0 + diff --git a/xsa299-4.11-0010-x86-mm-Fix-nested-de-validation-on-error.patch b/xsa299-4.11-0010-x86-mm-Fix-nested-de-validation-on-error.patch new file mode 100644 index 0000000..643ef53 --- /dev/null +++ b/xsa299-4.11-0010-x86-mm-Fix-nested-de-validation-on-error.patch @@ -0,0 +1,169 @@ +From eabd77b59f4006128501d6e15f9e620dfb349420 Mon Sep 17 00:00:00 2001 +From: George Dunlap +Date: Thu, 10 Oct 2019 17:57:49 +0100 +Subject: [PATCH 10/11] x86/mm: Fix nested de-validation on error + +If an invalid entry is discovered when validating a page-table tree, +the entire tree which has so far been validated must be de-validated. +Since this may take a long time, alloc_l[2-4]_table() set current +vcpu's old_guest_table immediately; put_old_guest_table() will make +sure that put_page_type() will be called to finish off the +de-validation before any other MMU operations can happen on the vcpu. + +The invariant for partial pages should be: + +* Entries [0, nr_validated_ptes) should be completely validated; + put_page_type() will de-validate these. + +* If [nr_validated_ptes] is partially validated, partial_flags should + set PTF_partiaL_set. put_page_type() will be called on this page to + finish off devalidation, and the appropriate refcount adjustments + will be done. + +alloc_l[2-3]_table() indicates partial validation to its callers by +setting current->old_guest_table. + +Unfortunately, this is mishandled. + +Take the case where validating lNe[x] returns an error. + +First, alloc_l3_table() doesn't check old_guest_table at all; as a +result, partial_flags is not set when it should be. nr_validated_ptes +is set to x; and since PFT_partial_set clear, de-validation resumes at +nr_validated_ptes-1. This means that the l2 page at pl3e[x] will not +have put_page_type() called on it when de-validating the rest of the +l3: it will be stuck in the PGT_partial state until the domain is +destroyed, or until it is re-used as an l2. (Any other page type will +fail.) + +Worse, alloc_l4_table(), rather than setting PTF_partial_set as it +should, sets nr_validated_ptes to x+1. When de-validating, since +partial is 0, this will correctly resume calling put_page_type at [x]; +but, if the put_page_type() is never called, but instead +get_page_type() is called, validation will pick up at [x+1], +neglecting to validate [x]. If the rest of the validation succeeds, +the l4 will be validated even though [x] is invalid. + +Fix this in both cases by setting PTF_partial_set if old_guest_table +is set. + +While here, add some safety catches: +- old_guest_table must point to the page contained in + [nr_validated_ptes]. +- alloc_l1_page shouldn't set old_guest_table + +If we experience one of these situations in production builds, it's +safer to avoid calling put_page_type for the pages in question. If +they have PGT_partial set, they will be cleaned up on domain +destruction; if not, we have no idea whether a type count is safe to +drop. Retaining an extra type ref that should have been dropped may +trigger a BUG() on the free_domain_page() path, but dropping a type +count that shouldn't be dropped may cause a privilege escalation. + +This is part of XSA-299. + +Reported-by: George Dunlap +Signed-off-by: George Dunlap +Reviewed-by: Jan Beulich +--- + xen/arch/x86/mm.c | 55 ++++++++++++++++++++++++++++++++++++++++++++++- + 1 file changed, 54 insertions(+), 1 deletion(-) + +diff --git a/xen/arch/x86/mm.c b/xen/arch/x86/mm.c +index cfb7538403..aa03cb8b40 100644 +--- a/xen/arch/x86/mm.c ++++ b/xen/arch/x86/mm.c +@@ -1561,6 +1561,20 @@ static int alloc_l2_table(struct page_info *page, unsigned long type) + ASSERT(current->arch.old_guest_table == NULL); + if ( i ) + { ++ /* ++ * alloc_l1_table() doesn't set old_guest_table; it does ++ * its own tear-down immediately on failure. If it ++ * did we'd need to check it and set partial_flags as we ++ * do in alloc_l[34]_table(). ++ * ++ * Note on the use of ASSERT: if it's non-null and ++ * hasn't been cleaned up yet, it should have ++ * PGT_partial set; and so the type will be cleaned up ++ * on domain destruction. Unfortunately, we would ++ * leak the general ref held by old_guest_table; but ++ * leaking a page is less bad than a host crash. ++ */ ++ ASSERT(current->arch.old_guest_table == NULL); + page->nr_validated_ptes = i; + page->partial_flags = partial_flags; + current->arch.old_guest_ptpg = NULL; +@@ -1588,6 +1602,7 @@ static int alloc_l3_table(struct page_info *page) + unsigned int i; + int rc = 0; + unsigned int partial_flags = page->partial_flags; ++ l3_pgentry_t l3e = l3e_empty(); + + pl3e = map_domain_page(_mfn(pfn)); + +@@ -1634,7 +1649,11 @@ static int alloc_l3_table(struct page_info *page) + rc = -ERESTART; + } + if ( rc < 0 ) ++ { ++ /* XSA-299 Backport: Copy l3e for checking */ ++ l3e = pl3e[i]; + break; ++ } + + pl3e[i] = adjust_guest_l3e(pl3e[i], d); + } +@@ -1648,6 +1667,24 @@ static int alloc_l3_table(struct page_info *page) + { + page->nr_validated_ptes = i; + page->partial_flags = partial_flags; ++ if ( current->arch.old_guest_table ) ++ { ++ /* ++ * We've experienced a validation failure. If ++ * old_guest_table is set, "transfer" the general ++ * reference count to pl3e[nr_validated_ptes] by ++ * setting PTF_partial_set. ++ * ++ * As a precaution, check that old_guest_table is the ++ * page pointed to by pl3e[nr_validated_ptes]. If ++ * not, it's safer to leak a type ref on production ++ * builds. ++ */ ++ if ( current->arch.old_guest_table == l3e_get_page(l3e) ) ++ page->partial_flags = PTF_partial_set; ++ else ++ ASSERT_UNREACHABLE(); ++ } + current->arch.old_guest_ptpg = NULL; + current->arch.old_guest_table = page; + } +@@ -1810,7 +1847,23 @@ static int alloc_l4_table(struct page_info *page) + else + { + if ( current->arch.old_guest_table ) +- page->nr_validated_ptes++; ++ { ++ /* ++ * We've experienced a validation failure. If ++ * old_guest_table is set, "transfer" the general ++ * reference count to pl3e[nr_validated_ptes] by ++ * setting PTF_partial_set. ++ * ++ * As a precaution, check that old_guest_table is the ++ * page pointed to by pl4e[nr_validated_ptes]. If ++ * not, it's safer to leak a type ref on production ++ * builds. ++ */ ++ if ( current->arch.old_guest_table == l4e_get_page(pl4e[i]) ) ++ page->partial_flags = PTF_partial_set; ++ else ++ ASSERT_UNREACHABLE(); ++ } + current->arch.old_guest_ptpg = NULL; + current->arch.old_guest_table = page; + } +-- +2.23.0 + diff --git a/xsa299-4.11-0011-x86-mm-Don-t-drop-a-type-ref-unless-you-held-a-ref-t.patch b/xsa299-4.11-0011-x86-mm-Don-t-drop-a-type-ref-unless-you-held-a-ref-t.patch new file mode 100644 index 0000000..24970da --- /dev/null +++ b/xsa299-4.11-0011-x86-mm-Don-t-drop-a-type-ref-unless-you-held-a-ref-t.patch @@ -0,0 +1,413 @@ +From f0086e3ac65c8bcabb84c1c29ab00b0c8a187555 Mon Sep 17 00:00:00 2001 +From: George Dunlap +Date: Thu, 10 Oct 2019 17:57:50 +0100 +Subject: [PATCH 11/11] x86/mm: Don't drop a type ref unless you held a ref to + begin with + +Validation and de-validation of pagetable trees may take arbitrarily +large amounts of time, and so must be preemptible. This is indicated +by setting the PGT_partial bit in the type_info, and setting +nr_validated_entries and partial_flags appropriately. Specifically, +if the entry at [nr_validated_entries] is partially validated, +partial_flags should have the PGT_partial_set bit set, and the entry +should hold a general reference count. During de-validation, +put_page_type() is called on partially validated entries. + +Unfortunately, there are a number of issues with the current algorithm. + +First, doing a "normal" put_page_type() is not safe when no type ref +is held: there is nothing to stop another vcpu from coming along and +picking up validation again: at which point the put_page_type may drop +the only page ref on an in-use page. Some examples are listed in the +appendix. + +The core issue is that put_page_type() is being called both to clean +up PGT_partial, and to drop a type count; and has no way of knowing +which is which; and so if in between, PGT_partial is cleared, +put_page_type() will drop the type ref erroneously. + +What is needed is to distinguish between two states: +- Dropping a type ref which is held +- Cleaning up a page which has been partially de/validated + +Fix this by telling put_page_type() which of the two activities you +intend. + +When cleaning up a partial de/validation, take no action unless you +find a page partially validated. + +If put_page_type() is called without PTF_partial_set, and finds the +page in a PGT_partial state anyway, then there's certainly been a +misaccounting somewhere, and carrying on would almost certainly cause +a security issue, so crash the host instead. + +In put_page_from_lNe, pass partial_flags on to _put_page_type(). + +old_guest_table may be set either with a fully validated page (when +using the "deferred put" pattern), or with a partially validated page +(when a normal "de-validation" is interrupted, or when a validation +fails part-way through due to invalid entries). Add a flag, +old_guest_table_partial, to indicate which of these it is, and use +that to pass the appropriate flag to _put_page_type(). + +While here, delete stray trailing whitespace. + +This is part of XSA-299. + +Reported-by: George Dunlap +Signed-off-by: George Dunlap +Reviewed-by: Jan Beulich +----- +Appendix: + +Suppose page A, when interpreted as an l3 pagetable, contains all +valid entries; and suppose A[x] points to page B, which when +interpreted as an l2 pagetable, contains all valid entries. + +P1: PIN_L3_TABLE + A -> PGT_l3_table | 1 | valid + B -> PGT_l2_table | 1 | valid + +P1: UNPIN_TABLE + > Arrange to interrupt after B has been de-validated + B: + type_info -> PGT_l2_table | 0 + A: + type_info -> PGT_l3_table | 1 | partial + nr_validated_enties -> (less than x) + +P2: mod_l4_entry to point to A + > Arrange for this to be interrupted while B is being validated + B: + type_info -> PGT_l2_table | 1 | partial + (nr_validated_entires &c set as appropriate) + A: + type_info -> PGT_l3_table | 1 | partial + nr_validated_entries -> x + partial_pte = 1 + +P3: mod_l3_entry some other unrelated l3 to point to B: + B: + type_info -> PGT_l2_table | 1 + +P1: Restart UNPIN_TABLE + +At this point, since A.nr_validate_entries == x and A.partial_pte != +0, free_l3_table() will call put_page_from_l3e() on pl3e[x], dropping +its type count to 0 while it's still being pointed to by some other l3 + +A similar issue arises with old_guest_table. Consider the following +scenario: + +Suppose A is a page which, when interpreted as an l2, has valid entries +until entry x, which is invalid. + +V1: PIN_L2_TABLE(A) + + A -> PGT_l2_table | 1 | PGT_partial + V1 -> old_guest_table = A + + +V2: PIN_L2_TABLE(A) + + A -> PGT_l2_table | 1 | PGT_partial + V2 -> old_guest_table = A + + put_old_guest_table() + _put_page_type(A) + A -> PGT_l2_table | 0 + +V1: + put_old_guest_table() + _put_page_type(A) # UNDERFLOW + +Indeed, it is possible to engineer for old_guest_table for every vcpu +a guest has to point to the same page. +--- + xen/arch/x86/domain.c | 6 +++ + xen/arch/x86/mm.c | 99 +++++++++++++++++++++++++++++++----- + xen/include/asm-x86/domain.h | 4 +- + 3 files changed, 95 insertions(+), 14 deletions(-) + +diff --git a/xen/arch/x86/domain.c b/xen/arch/x86/domain.c +index 8fbecbb169..c880568dd4 100644 +--- a/xen/arch/x86/domain.c ++++ b/xen/arch/x86/domain.c +@@ -1074,9 +1074,15 @@ int arch_set_info_guest( + rc = -ERESTART; + /* Fallthrough */ + case -ERESTART: ++ /* ++ * NB that we're putting the kernel-mode table ++ * here, which we've already successfully ++ * validated above; hence partial = false; ++ */ + v->arch.old_guest_ptpg = NULL; + v->arch.old_guest_table = + pagetable_get_page(v->arch.guest_table); ++ v->arch.old_guest_table_partial = false; + v->arch.guest_table = pagetable_null(); + break; + default: +diff --git a/xen/arch/x86/mm.c b/xen/arch/x86/mm.c +index aa03cb8b40..c701c7ef14 100644 +--- a/xen/arch/x86/mm.c ++++ b/xen/arch/x86/mm.c +@@ -1353,10 +1353,11 @@ static int put_page_from_l2e(l2_pgentry_t l2e, unsigned long pfn, + { + current->arch.old_guest_ptpg = ptpg; + current->arch.old_guest_table = pg; ++ current->arch.old_guest_table_partial = false; + } + else + { +- rc = _put_page_type(pg, PTF_preemptible, ptpg); ++ rc = _put_page_type(pg, flags | PTF_preemptible, ptpg); + if ( likely(!rc) ) + put_page(pg); + } +@@ -1379,6 +1380,7 @@ static int put_page_from_l3e(l3_pgentry_t l3e, unsigned long pfn, + unsigned long mfn = l3e_get_pfn(l3e); + int writeable = l3e_get_flags(l3e) & _PAGE_RW; + ++ ASSERT(!(flags & PTF_partial_set)); + ASSERT(!(mfn & ((1UL << (L3_PAGETABLE_SHIFT - PAGE_SHIFT)) - 1))); + do { + put_data_page(mfn_to_page(_mfn(mfn)), writeable); +@@ -1391,12 +1393,14 @@ static int put_page_from_l3e(l3_pgentry_t l3e, unsigned long pfn, + + if ( flags & PTF_defer ) + { ++ ASSERT(!(flags & PTF_partial_set)); + current->arch.old_guest_ptpg = mfn_to_page(_mfn(pfn)); + current->arch.old_guest_table = pg; ++ current->arch.old_guest_table_partial = false; + return 0; + } + +- rc = _put_page_type(pg, PTF_preemptible, mfn_to_page(_mfn(pfn))); ++ rc = _put_page_type(pg, flags | PTF_preemptible, mfn_to_page(_mfn(pfn))); + if ( likely(!rc) ) + put_page(pg); + +@@ -1415,12 +1419,15 @@ static int put_page_from_l4e(l4_pgentry_t l4e, unsigned long pfn, + + if ( flags & PTF_defer ) + { ++ ASSERT(!(flags & PTF_partial_set)); + current->arch.old_guest_ptpg = mfn_to_page(_mfn(pfn)); + current->arch.old_guest_table = pg; ++ current->arch.old_guest_table_partial = false; + return 0; + } + +- rc = _put_page_type(pg, PTF_preemptible, mfn_to_page(_mfn(pfn))); ++ rc = _put_page_type(pg, flags | PTF_preemptible, ++ mfn_to_page(_mfn(pfn))); + if ( likely(!rc) ) + put_page(pg); + } +@@ -1525,6 +1532,14 @@ static int alloc_l2_table(struct page_info *page, unsigned long type) + + pl2e = map_domain_page(_mfn(pfn)); + ++ /* ++ * NB that alloc_l2_table will never set partial_pte on an l2; but ++ * free_l2_table might if a linear_pagetable entry is interrupted ++ * partway through de-validation. In that circumstance, ++ * get_page_from_l2e() will always return -EINVAL; and we must ++ * retain the type ref by doing the normal partial_flags tracking. ++ */ ++ + for ( i = page->nr_validated_ptes; i < L2_PAGETABLE_ENTRIES; + i++, partial_flags = 0 ) + { +@@ -1579,6 +1594,7 @@ static int alloc_l2_table(struct page_info *page, unsigned long type) + page->partial_flags = partial_flags; + current->arch.old_guest_ptpg = NULL; + current->arch.old_guest_table = page; ++ current->arch.old_guest_table_partial = true; + } + } + if ( rc < 0 ) +@@ -1681,12 +1697,16 @@ static int alloc_l3_table(struct page_info *page) + * builds. + */ + if ( current->arch.old_guest_table == l3e_get_page(l3e) ) ++ { ++ ASSERT(current->arch.old_guest_table_partial); + page->partial_flags = PTF_partial_set; ++ } + else + ASSERT_UNREACHABLE(); + } + current->arch.old_guest_ptpg = NULL; + current->arch.old_guest_table = page; ++ current->arch.old_guest_table_partial = true; + } + while ( i-- > 0 ) + pl3e[i] = unadjust_guest_l3e(pl3e[i], d); +@@ -1860,12 +1880,16 @@ static int alloc_l4_table(struct page_info *page) + * builds. + */ + if ( current->arch.old_guest_table == l4e_get_page(pl4e[i]) ) ++ { ++ ASSERT(current->arch.old_guest_table_partial); + page->partial_flags = PTF_partial_set; ++ } + else + ASSERT_UNREACHABLE(); + } + current->arch.old_guest_ptpg = NULL; + current->arch.old_guest_table = page; ++ current->arch.old_guest_table_partial = true; + } + } + } +@@ -2782,6 +2806,28 @@ static int _put_page_type(struct page_info *page, unsigned int flags, + x = y; + nx = x - 1; + ++ /* ++ * Is this expected to do a full reference drop, or only ++ * cleanup partial validation / devalidation? ++ * ++ * If the former, the caller must hold a "full" type ref; ++ * which means the page must be validated. If the page is ++ * *not* fully validated, continuing would almost certainly ++ * open up a security hole. An exception to this is during ++ * domain destruction, where PGT_validated can be dropped ++ * without dropping a type ref. ++ * ++ * If the latter, do nothing unless type PGT_partial is set. ++ * If it is set, the type count must be 1. ++ */ ++ if ( !(flags & PTF_partial_set) ) ++ BUG_ON((x & PGT_partial) || ++ !((x & PGT_validated) || page_get_owner(page)->is_dying)); ++ else if ( !(x & PGT_partial) ) ++ return 0; ++ else ++ BUG_ON((x & PGT_count_mask) != 1); ++ + ASSERT((x & PGT_count_mask) != 0); + + switch ( nx & (PGT_locked | PGT_count_mask) ) +@@ -3041,17 +3087,34 @@ int put_old_guest_table(struct vcpu *v) + if ( !v->arch.old_guest_table ) + return 0; + +- switch ( rc = _put_page_type(v->arch.old_guest_table, PTF_preemptible, +- v->arch.old_guest_ptpg) ) ++ rc = _put_page_type(v->arch.old_guest_table, ++ PTF_preemptible | ++ ( v->arch.old_guest_table_partial ? ++ PTF_partial_set : 0 ), ++ v->arch.old_guest_ptpg); ++ ++ if ( rc == -ERESTART || rc == -EINTR ) + { +- case -EINTR: +- case -ERESTART: ++ v->arch.old_guest_table_partial = (rc == -ERESTART); + return -ERESTART; +- case 0: +- put_page(v->arch.old_guest_table); + } + ++ /* ++ * It shouldn't be possible for _put_page_type() to return ++ * anything else at the moment; but if it does happen in ++ * production, leaking the type ref is probably the best thing to ++ * do. Either way, drop the general ref held by old_guest_table. ++ */ ++ ASSERT(rc == 0); ++ ++ put_page(v->arch.old_guest_table); + v->arch.old_guest_table = NULL; ++ v->arch.old_guest_ptpg = NULL; ++ /* ++ * Safest default if someone sets old_guest_table without ++ * explicitly setting old_guest_table_partial. ++ */ ++ v->arch.old_guest_table_partial = true; + + return rc; + } +@@ -3201,11 +3264,11 @@ int new_guest_cr3(mfn_t mfn) + switch ( rc = put_page_and_type_preemptible(page) ) + { + case -EINTR: +- rc = -ERESTART; +- /* fallthrough */ + case -ERESTART: + curr->arch.old_guest_ptpg = NULL; + curr->arch.old_guest_table = page; ++ curr->arch.old_guest_table_partial = (rc == -ERESTART); ++ rc = -ERESTART; + break; + default: + BUG_ON(rc); +@@ -3479,6 +3542,7 @@ long do_mmuext_op( + { + curr->arch.old_guest_ptpg = NULL; + curr->arch.old_guest_table = page; ++ curr->arch.old_guest_table_partial = false; + } + } + } +@@ -3513,6 +3577,11 @@ long do_mmuext_op( + case -ERESTART: + curr->arch.old_guest_ptpg = NULL; + curr->arch.old_guest_table = page; ++ /* ++ * EINTR means we still hold the type ref; ERESTART ++ * means PGT_partial holds the type ref ++ */ ++ curr->arch.old_guest_table_partial = (rc == -ERESTART); + rc = 0; + break; + default: +@@ -3581,11 +3650,15 @@ long do_mmuext_op( + switch ( rc = put_page_and_type_preemptible(page) ) + { + case -EINTR: +- rc = -ERESTART; +- /* fallthrough */ + case -ERESTART: + curr->arch.old_guest_ptpg = NULL; + curr->arch.old_guest_table = page; ++ /* ++ * EINTR means we still hold the type ref; ++ * ERESTART means PGT_partial holds the ref ++ */ ++ curr->arch.old_guest_table_partial = (rc == -ERESTART); ++ rc = -ERESTART; + break; + default: + BUG_ON(rc); +diff --git a/xen/include/asm-x86/domain.h b/xen/include/asm-x86/domain.h +index 1ac5a96c08..360c38bd83 100644 +--- a/xen/include/asm-x86/domain.h ++++ b/xen/include/asm-x86/domain.h +@@ -309,7 +309,7 @@ struct arch_domain + + struct paging_domain paging; + struct p2m_domain *p2m; +- /* To enforce lock ordering in the pod code wrt the ++ /* To enforce lock ordering in the pod code wrt the + * page_alloc lock */ + int page_alloc_unlock_level; + +@@ -542,6 +542,8 @@ struct arch_vcpu + struct page_info *old_guest_table; /* partially destructed pagetable */ + struct page_info *old_guest_ptpg; /* containing page table of the */ + /* former, if any */ ++ bool old_guest_table_partial; /* Are we dropping a type ref, or just ++ * finishing up a partial de-validation? */ + /* guest_table holds a ref to the page, and also a type-count unless + * shadow refcounts are in use */ + pagetable_t shadow_table[4]; /* (MFN) shadow(s) of guest */ +-- +2.23.0 + diff --git a/xsa301-4.11-1.patch b/xsa301-4.11-1.patch new file mode 100644 index 0000000..4d528fe --- /dev/null +++ b/xsa301-4.11-1.patch @@ -0,0 +1,80 @@ +From 21dfe8f707febd62869d4ebbaa155736870bebec Mon Sep 17 00:00:00 2001 +From: Julien Grall +Date: Wed, 2 Oct 2019 12:06:50 +0100 +Subject: [PATCH 1/3] xen/arm: p2m: Avoid aliasing guest physical frame + +The P2M helpers implementation is quite lax and will end up to ignore +the unused top bits of a guest physical frame. + +This effectively means that p2m_set_entry() will create a mapping for a +different frame (it is always equal to gfn & (mask unused bits)). Yet +p2m->max_mapped_gfn will be updated using the original frame. + +At the moment, p2m_get_entry() and p2m_resolve_translation_fault() +assume that p2m_get_root_pointer() will always return a non-NULL pointer +when the GFN is smaller than p2m->max_mapped_gfn. + +Unfortunately, because of the aliasing described above, it would be +possible to set p2m->max_mapped_gfn high enough so it covers frame that +would lead p2m_get_root_pointer() to return NULL. + +As we don't sanity check the guest physical frame provided by a guest, a +malicious guest could craft a series of hypercalls that will hit the +BUG_ON() and therefore DoS Xen. + +To prevent aliasing, the function p2m_get_root_pointer() is now reworked +to return NULL If any of the unused top bits are not zero. The caller +can then decide what's the appropriate action to do. Since the two paths +(i.e. P2M_ROOT_PAGES == 1 and P2M_ROOT_PAGES != 1) are now very +similarly, take the opportunity to consolidate them making the code a +bit simpler. + +With this change, p2m_get_entry() will not try to insert a mapping as +the root pointer is invalid. + +Note that root_table is now switch to unsigned long as unsigned int is +not enough to hold part of a GFN. + +This is part of XSA-301. + +Reported-by: Julien Grall +Signed-off-by: Julien Grall +Reviewed-by: Stefano Stabellini +--- + xen/arch/arm/p2m.c | 17 +++++------------ + 1 file changed, 5 insertions(+), 12 deletions(-) + +diff --git a/xen/arch/arm/p2m.c b/xen/arch/arm/p2m.c +index d43c3aa896..3967ee7306 100644 +--- a/xen/arch/arm/p2m.c ++++ b/xen/arch/arm/p2m.c +@@ -177,21 +177,14 @@ void p2m_tlb_flush_sync(struct p2m_domain *p2m) + static lpae_t *p2m_get_root_pointer(struct p2m_domain *p2m, + gfn_t gfn) + { +- unsigned int root_table; +- +- if ( P2M_ROOT_PAGES == 1 ) +- return __map_domain_page(p2m->root); ++ unsigned long root_table; + + /* +- * Concatenated root-level tables. The table number will be the +- * offset at the previous level. It is not possible to +- * concatenate a level-0 root. ++ * While the root table index is the offset from the previous level, ++ * we can't use (P2M_ROOT_LEVEL - 1) because the root level might be ++ * 0. Yet we still want to check if all the unused bits are zeroed. + */ +- ASSERT(P2M_ROOT_LEVEL > 0); +- +- root_table = gfn_x(gfn) >> (level_orders[P2M_ROOT_LEVEL - 1]); +- root_table &= LPAE_ENTRY_MASK; +- ++ root_table = gfn_x(gfn) >> (level_orders[P2M_ROOT_LEVEL] + LPAE_SHIFT); + if ( root_table >= P2M_ROOT_PAGES ) + return NULL; + +-- +2.11.0 + diff --git a/xsa301-4.11-2.patch b/xsa301-4.11-2.patch new file mode 100644 index 0000000..33b6150 --- /dev/null +++ b/xsa301-4.11-2.patch @@ -0,0 +1,92 @@ +From 4426d993b7ee0966fb39531dc5a269ce8493ca97 Mon Sep 17 00:00:00 2001 +From: Julien Grall +Date: Wed, 2 Oct 2019 12:35:59 +0100 +Subject: [PATCH 2/3] xen/arm: p2m: Avoid off-by-one check on + p2m->max_mapped_gfn + +The code base is using inconsistently the field p2m->max_mapped_gfn. +Some of the useres expect that p2m->max_guest_gfn contain the highest +mapped GFN while others expect highest + 1. + +p2m->max_guest_gfn is set as highest + 1, because of that the sanity +check on the GFN in p2m_resolved_translation_fault() and +p2m_get_entry() can be bypassed when GFN == p2m->max_guest_gfn. + +p2m_get_root_pointer(p2m->max_guest_gfn) may return NULL if it is +outside of address range supported and therefore the BUG_ON() could be +hit. + +The current value hold in p2m->max_mapped_gfn is inconsistent with the +expectation of the common code (see domain_get_maximum_gpfn()) and also +the documentation of the field. + +Rather than changing the check in p2m_translation_fault() and +p2m_get_entry(), p2m->max_mapped_gfn is now containing the highest +mapped GFN and the callers assuming "highest + 1" are now adjusted. + +Take the opportunity to use 1UL rather than 1 as page_order could +theoritically big enough to overflow a 32-bit integer. + +Lastly, the documentation of the field max_guest_gfn to reflect how it +is computed. + +This is part of XSA-301. + +Reported-by: Julien Grall +Signed-off-by: Julien Grall +Reviewed-by: Stefano Stabellini +--- + xen/arch/arm/p2m.c | 6 +++--- + xen/include/asm-arm/p2m.h | 5 +---- + 2 files changed, 4 insertions(+), 7 deletions(-) + +diff --git a/xen/arch/arm/p2m.c b/xen/arch/arm/p2m.c +index 3967ee7306..c7e049901d 100644 +--- a/xen/arch/arm/p2m.c ++++ b/xen/arch/arm/p2m.c +@@ -931,7 +931,7 @@ static int __p2m_set_entry(struct p2m_domain *p2m, + p2m_write_pte(entry, pte, p2m->clean_pte); + + p2m->max_mapped_gfn = gfn_max(p2m->max_mapped_gfn, +- gfn_add(sgfn, 1 << page_order)); ++ gfn_add(sgfn, (1UL << page_order) - 1)); + p2m->lowest_mapped_gfn = gfn_min(p2m->lowest_mapped_gfn, sgfn); + } + +@@ -1291,7 +1291,7 @@ int relinquish_p2m_mapping(struct domain *d) + p2m_write_lock(p2m); + + start = p2m->lowest_mapped_gfn; +- end = p2m->max_mapped_gfn; ++ end = gfn_add(p2m->max_mapped_gfn, 1); + + for ( ; gfn_x(start) < gfn_x(end); + start = gfn_next_boundary(start, order) ) +@@ -1356,7 +1356,7 @@ int p2m_cache_flush(struct domain *d, gfn_t start, unsigned long nr) + p2m_read_lock(p2m); + + start = gfn_max(start, p2m->lowest_mapped_gfn); +- end = gfn_min(end, p2m->max_mapped_gfn); ++ end = gfn_min(end, gfn_add(p2m->max_mapped_gfn, 1)); + + for ( ; gfn_x(start) < gfn_x(end); start = next_gfn ) + { +diff --git a/xen/include/asm-arm/p2m.h b/xen/include/asm-arm/p2m.h +index 8823707c17..7f1f7e9109 100644 +--- a/xen/include/asm-arm/p2m.h ++++ b/xen/include/asm-arm/p2m.h +@@ -38,10 +38,7 @@ struct p2m_domain { + /* Current Translation Table Base Register for the p2m */ + uint64_t vttbr; + +- /* +- * Highest guest frame that's ever been mapped in the p2m +- * Only takes into account ram and foreign mapping +- */ ++ /* Highest guest frame that's ever been mapped in the p2m */ + gfn_t max_mapped_gfn; + + /* +-- +2.11.0 + diff --git a/xsa301-4.11-3.patch b/xsa301-4.11-3.patch new file mode 100644 index 0000000..55a701a --- /dev/null +++ b/xsa301-4.11-3.patch @@ -0,0 +1,49 @@ +From 61c73af08b4ede1fc8cfd2cf72661e6c7cfdbeaa Mon Sep 17 00:00:00 2001 +From: Julien Grall +Date: Wed, 2 Oct 2019 10:55:07 +0100 +Subject: [PATCH 3/3] xen/arm: p2m: Don't check the return of + p2m_get_root_pointer() with BUG_ON() + +It turns out that the BUG_ON() was actually reachable with well-crafted +hypercalls. The BUG_ON() is here to prevent catch logical error, so +crashing Xen is a bit over the top. + +While all the holes should now be fixed, it would be better to downgrade +the BUG_ON() to something less fatal to prevent any more DoS. + +The BUG_ON() in p2m_get_entry() is now replaced by ASSERT_UNREACHABLE() +to catch mistake in debug build and return INVALID_MFN for production +build. The interface also requires to set page_order to give an idea of +the size of "hole". So 'level' is now set so we report a hole of size of +the an entry of the root page-table. This stays inline with what happen +when the GFN is higher than p2m->max_mapped_gfn. + +This is part of XSA-301. + +Reported-by: Julien Grall +Signed-off-by: Julien Grall +--- + xen/arch/arm/p2m.c | 7 ++++++- + 1 file changed, 6 insertions(+), 1 deletion(-) + +diff --git a/xen/arch/arm/p2m.c b/xen/arch/arm/p2m.c +index c7e049901d..af3515df42 100644 +--- a/xen/arch/arm/p2m.c ++++ b/xen/arch/arm/p2m.c +@@ -318,7 +318,12 @@ mfn_t p2m_get_entry(struct p2m_domain *p2m, gfn_t gfn, + * the table should always be non-NULL because the gfn is below + * p2m->max_mapped_gfn and the root table pages are always present. + */ +- BUG_ON(table == NULL); ++ if ( !table ) ++ { ++ ASSERT_UNREACHABLE(); ++ level = P2M_ROOT_LEVEL; ++ goto out; ++ } + + for ( level = P2M_ROOT_LEVEL; level < 3; level++ ) + { +-- +2.11.0 + diff --git a/xsa302-4.11-0001-IOMMU-add-missing-HVM-check.patch b/xsa302-4.11-0001-IOMMU-add-missing-HVM-check.patch new file mode 100644 index 0000000..c3d4435 --- /dev/null +++ b/xsa302-4.11-0001-IOMMU-add-missing-HVM-check.patch @@ -0,0 +1,37 @@ +From bbca29f88d9ad9c7e91125a3b5d5f13a23e5801f Mon Sep 17 00:00:00 2001 +From: Jan Beulich +Date: Wed, 2 Oct 2019 13:36:59 +0200 +Subject: [PATCH 1/2] IOMMU: add missing HVM check +MIME-Version: 1.0 +Content-Type: text/plain; charset=UTF-8 +Content-Transfer-Encoding: 8bit + +Fix an unguarded d->arch.hvm access in assign_device(). + +Signed-off-by: Jan Beulich +Reviewed-by: Roger Pau Monné +Acked-by: Andrew Cooper + +(cherry picked from commit 41fd1009cd7416b73d745a77c24b4e8d1a296fe6) +Signed-off-by: Ian Jackson +--- + xen/drivers/passthrough/pci.c | 3 ++- + 1 file changed, 2 insertions(+), 1 deletion(-) + +diff --git a/xen/drivers/passthrough/pci.c b/xen/drivers/passthrough/pci.c +index f51cae7f4e..037aba7c94 100644 +--- a/xen/drivers/passthrough/pci.c ++++ b/xen/drivers/passthrough/pci.c +@@ -1416,7 +1416,8 @@ static int assign_device(struct domain *d, u16 seg, u8 bus, u8 devfn, u32 flag) + /* Prevent device assign if mem paging or mem sharing have been + * enabled for this domain */ + if ( unlikely(!need_iommu(d) && +- (d->arch.hvm_domain.mem_sharing_enabled || ++ ((is_hvm_domain(d) && ++ d->arch.hvm_domain.mem_sharing_enabled) || + vm_event_check_ring(d->vm_event_paging) || + p2m_get_hostp2m(d)->global_logdirty)) ) + return -EXDEV; +-- +2.11.0 + diff --git a/xsa302-4.11-0002-passthrough-quarantine-PCI-devices.patch b/xsa302-4.11-0002-passthrough-quarantine-PCI-devices.patch new file mode 100644 index 0000000..5204c9f --- /dev/null +++ b/xsa302-4.11-0002-passthrough-quarantine-PCI-devices.patch @@ -0,0 +1,498 @@ +From ec99857f59f7f06236f11ca8b0b2303e5e745cc4 Mon Sep 17 00:00:00 2001 +From: Paul Durrant +Date: Mon, 14 Oct 2019 17:52:59 +0100 +Subject: [PATCH 2/2] passthrough: quarantine PCI devices + +When a PCI device is assigned to an untrusted domain, it is possible for +that domain to program the device to DMA to an arbitrary address. The +IOMMU is used to protect the host from malicious DMA by making sure that +the device addresses can only target memory assigned to the guest. However, +when the guest domain is torn down the device is assigned back to dom0, +thus allowing any in-flight DMA to potentially target critical host data. + +This patch introduces a 'quarantine' for PCI devices using dom_io. When +the toolstack makes a device assignable (by binding it to pciback), it +will now also assign it to DOMID_IO and the device will only be assigned +back to dom0 when the device is made unassignable again. Whilst device is +assignable it will only ever transfer between dom_io and guest domains. +dom_io is actually only used as a sentinel domain for quarantining purposes; +it is not configured with any IOMMU mappings. Assignment to dom_io simply +means that the device's initiator (requestor) identifier is not present in +the IOMMU's device table and thus any DMA transactions issued will be +terminated with a fault condition. + +In addition, a fix to assignment handling is made for VT-d. Failure +during the assignment step should not lead to a device still being +associated with its prior owner. Hand the device to DomIO temporarily, +until the assignment step has completed successfully. Remove the PI +hooks from the source domain then earlier as well. + +Failure of the recovery reassign_device_ownership() may not go silent: +There e.g. may still be left over RMRR mappings in the domain assignment +to which has failed, and hence we can't allow that domain to continue +executing. + +NOTE: This patch also includes one printk() cleanup; the + "XEN_DOMCTL_assign_device: " tag is dropped in iommu_do_pci_domctl(), + since similar printk()-s elsewhere also don't log such a tag. + +This is XSA-302. + +Signed-off-by: Paul Durrant +Signed-off-by: Jan Beulich +Signed-off-by: Ian Jackson +--- + tools/libxl/libxl_pci.c | 25 +++++++++++- + xen/arch/x86/mm.c | 2 + + xen/common/domctl.c | 14 ++++++- + xen/drivers/passthrough/amd/pci_amd_iommu.c | 10 ++++- + xen/drivers/passthrough/iommu.c | 9 +++++ + xen/drivers/passthrough/pci.c | 59 ++++++++++++++++++++++------- + xen/drivers/passthrough/vtd/iommu.c | 40 ++++++++++++++++--- + xen/include/xen/pci.h | 3 ++ + 8 files changed, 138 insertions(+), 24 deletions(-) + +diff --git a/tools/libxl/libxl_pci.c b/tools/libxl/libxl_pci.c +index 4755a0c93c..81890a91ac 100644 +--- a/tools/libxl/libxl_pci.c ++++ b/tools/libxl/libxl_pci.c +@@ -754,6 +754,7 @@ static int libxl__device_pci_assignable_add(libxl__gc *gc, + libxl_device_pci *pcidev, + int rebind) + { ++ libxl_ctx *ctx = libxl__gc_owner(gc); + unsigned dom, bus, dev, func; + char *spath, *driver_path = NULL; + int rc; +@@ -779,7 +780,7 @@ static int libxl__device_pci_assignable_add(libxl__gc *gc, + } + if ( rc ) { + LOG(WARN, PCI_BDF" already assigned to pciback", dom, bus, dev, func); +- return 0; ++ goto quarantine; + } + + /* Check to see if there's already a driver that we need to unbind from */ +@@ -810,6 +811,19 @@ static int libxl__device_pci_assignable_add(libxl__gc *gc, + return ERROR_FAIL; + } + ++quarantine: ++ /* ++ * DOMID_IO is just a sentinel domain, without any actual mappings, ++ * so always pass XEN_DOMCTL_DEV_RDM_RELAXED to avoid assignment being ++ * unnecessarily denied. ++ */ ++ rc = xc_assign_device(ctx->xch, DOMID_IO, pcidev_encode_bdf(pcidev), ++ XEN_DOMCTL_DEV_RDM_RELAXED); ++ if ( rc < 0 ) { ++ LOG(ERROR, "failed to quarantine "PCI_BDF, dom, bus, dev, func); ++ return ERROR_FAIL; ++ } ++ + return 0; + } + +@@ -817,9 +831,18 @@ static int libxl__device_pci_assignable_remove(libxl__gc *gc, + libxl_device_pci *pcidev, + int rebind) + { ++ libxl_ctx *ctx = libxl__gc_owner(gc); + int rc; + char *driver_path; + ++ /* De-quarantine */ ++ rc = xc_deassign_device(ctx->xch, DOMID_IO, pcidev_encode_bdf(pcidev)); ++ if ( rc < 0 ) { ++ LOG(ERROR, "failed to de-quarantine "PCI_BDF, pcidev->domain, pcidev->bus, ++ pcidev->dev, pcidev->func); ++ return ERROR_FAIL; ++ } ++ + /* Unbind from pciback */ + if ( (rc=pciback_dev_is_assigned(gc, pcidev)) < 0 ) { + return ERROR_FAIL; +diff --git a/xen/arch/x86/mm.c b/xen/arch/x86/mm.c +index e6a4cb28f8..c1ab57f9a5 100644 +--- a/xen/arch/x86/mm.c ++++ b/xen/arch/x86/mm.c +@@ -295,9 +295,11 @@ void __init arch_init_memory(void) + * Initialise our DOMID_IO domain. + * This domain owns I/O pages that are within the range of the page_info + * array. Mappings occur at the priv of the caller. ++ * Quarantined PCI devices will be associated with this domain. + */ + dom_io = domain_create(DOMID_IO, NULL); + BUG_ON(IS_ERR(dom_io)); ++ INIT_LIST_HEAD(&dom_io->arch.pdev_list); + + /* + * Initialise our COW domain. +diff --git a/xen/common/domctl.c b/xen/common/domctl.c +index 9b7bc083ee..741d774cd1 100644 +--- a/xen/common/domctl.c ++++ b/xen/common/domctl.c +@@ -392,6 +392,16 @@ long do_domctl(XEN_GUEST_HANDLE_PARAM(xen_domctl_t) u_domctl) + + switch ( op->cmd ) + { ++ case XEN_DOMCTL_assign_device: ++ case XEN_DOMCTL_deassign_device: ++ if ( op->domain == DOMID_IO ) ++ { ++ d = dom_io; ++ break; ++ } ++ else if ( op->domain == DOMID_INVALID ) ++ return -ESRCH; ++ /* fall through */ + case XEN_DOMCTL_test_assign_device: + if ( op->domain == DOMID_INVALID ) + { +@@ -413,7 +423,7 @@ long do_domctl(XEN_GUEST_HANDLE_PARAM(xen_domctl_t) u_domctl) + + if ( !domctl_lock_acquire() ) + { +- if ( d ) ++ if ( d && d != dom_io ) + rcu_unlock_domain(d); + return hypercall_create_continuation( + __HYPERVISOR_domctl, "h", u_domctl); +@@ -1148,7 +1158,7 @@ long do_domctl(XEN_GUEST_HANDLE_PARAM(xen_domctl_t) u_domctl) + domctl_lock_release(); + + domctl_out_unlock_domonly: +- if ( d ) ++ if ( d && d != dom_io ) + rcu_unlock_domain(d); + + if ( copyback && __copy_to_guest(u_domctl, op, 1) ) +diff --git a/xen/drivers/passthrough/amd/pci_amd_iommu.c b/xen/drivers/passthrough/amd/pci_amd_iommu.c +index 12d2695b89..ec8baae717 100644 +--- a/xen/drivers/passthrough/amd/pci_amd_iommu.c ++++ b/xen/drivers/passthrough/amd/pci_amd_iommu.c +@@ -118,6 +118,10 @@ static void amd_iommu_setup_domain_device( + u8 bus = pdev->bus; + const struct domain_iommu *hd = dom_iommu(domain); + ++ /* dom_io is used as a sentinel for quarantined devices */ ++ if ( domain == dom_io ) ++ return; ++ + BUG_ON( !hd->arch.root_table || !hd->arch.paging_mode || + !iommu->dev_table.buffer ); + +@@ -305,6 +309,10 @@ void amd_iommu_disable_domain_device(struct domain *domain, + int req_id; + u8 bus = pdev->bus; + ++ /* dom_io is used as a sentinel for quarantined devices */ ++ if ( domain == dom_io ) ++ return; ++ + BUG_ON ( iommu->dev_table.buffer == NULL ); + req_id = get_dma_requestor_id(iommu->seg, PCI_BDF2(bus, devfn)); + dte = iommu->dev_table.buffer + (req_id * IOMMU_DEV_TABLE_ENTRY_SIZE); +@@ -391,7 +399,7 @@ static int amd_iommu_assign_device(struct domain *d, u8 devfn, + ivrs_mappings[req_id].read_permission); + } + +- return reassign_device(hardware_domain, d, devfn, pdev); ++ return reassign_device(pdev->domain, d, devfn, pdev); + } + + static void deallocate_next_page_table(struct page_info *pg, int level) +diff --git a/xen/drivers/passthrough/iommu.c b/xen/drivers/passthrough/iommu.c +index 04b0be37d3..8027d96f1c 100644 +--- a/xen/drivers/passthrough/iommu.c ++++ b/xen/drivers/passthrough/iommu.c +@@ -219,6 +219,9 @@ void iommu_teardown(struct domain *d) + { + const struct domain_iommu *hd = dom_iommu(d); + ++ if ( d == dom_io ) ++ return; ++ + d->need_iommu = 0; + hd->platform_ops->teardown(d); + tasklet_schedule(&iommu_pt_cleanup_tasklet); +@@ -229,6 +232,9 @@ int iommu_construct(struct domain *d) + if ( need_iommu(d) > 0 ) + return 0; + ++ if ( d == dom_io ) ++ return 0; ++ + if ( !iommu_use_hap_pt(d) ) + { + int rc; +@@ -404,6 +410,9 @@ int __init iommu_setup(void) + printk("I/O virtualisation %sabled\n", iommu_enabled ? "en" : "dis"); + if ( iommu_enabled ) + { ++ if ( iommu_domain_init(dom_io) ) ++ panic("Could not set up quarantine\n"); ++ + printk(" - Dom0 mode: %s\n", + iommu_passthrough ? "Passthrough" : + iommu_dom0_strict ? "Strict" : "Relaxed"); +diff --git a/xen/drivers/passthrough/pci.c b/xen/drivers/passthrough/pci.c +index 037aba7c94..fb010a547b 100644 +--- a/xen/drivers/passthrough/pci.c ++++ b/xen/drivers/passthrough/pci.c +@@ -1389,19 +1389,29 @@ static int iommu_remove_device(struct pci_dev *pdev) + return hd->platform_ops->remove_device(pdev->devfn, pci_to_dev(pdev)); + } + +-/* +- * If the device isn't owned by the hardware domain, it means it already +- * has been assigned to other domain, or it doesn't exist. +- */ + static int device_assigned(u16 seg, u8 bus, u8 devfn) + { + struct pci_dev *pdev; ++ int rc = 0; + + pcidevs_lock(); +- pdev = pci_get_pdev_by_domain(hardware_domain, seg, bus, devfn); ++ ++ pdev = pci_get_pdev(seg, bus, devfn); ++ ++ if ( !pdev ) ++ rc = -ENODEV; ++ /* ++ * If the device exists and it is not owned by either the hardware ++ * domain or dom_io then it must be assigned to a guest, or be ++ * hidden (owned by dom_xen). ++ */ ++ else if ( pdev->domain != hardware_domain && ++ pdev->domain != dom_io ) ++ rc = -EBUSY; ++ + pcidevs_unlock(); + +- return pdev ? 0 : -EBUSY; ++ return rc; + } + + static int assign_device(struct domain *d, u16 seg, u8 bus, u8 devfn, u32 flag) +@@ -1415,7 +1425,8 @@ static int assign_device(struct domain *d, u16 seg, u8 bus, u8 devfn, u32 flag) + + /* Prevent device assign if mem paging or mem sharing have been + * enabled for this domain */ +- if ( unlikely(!need_iommu(d) && ++ if ( d != dom_io && ++ unlikely(!need_iommu(d) && + ((is_hvm_domain(d) && + d->arch.hvm_domain.mem_sharing_enabled) || + vm_event_check_ring(d->vm_event_paging) || +@@ -1432,12 +1443,20 @@ static int assign_device(struct domain *d, u16 seg, u8 bus, u8 devfn, u32 flag) + return rc; + } + +- pdev = pci_get_pdev_by_domain(hardware_domain, seg, bus, devfn); ++ pdev = pci_get_pdev(seg, bus, devfn); ++ ++ rc = -ENODEV; + if ( !pdev ) +- { +- rc = pci_get_pdev(seg, bus, devfn) ? -EBUSY : -ENODEV; + goto done; +- } ++ ++ rc = 0; ++ if ( d == pdev->domain ) ++ goto done; ++ ++ rc = -EBUSY; ++ if ( pdev->domain != hardware_domain && ++ pdev->domain != dom_io ) ++ goto done; + + if ( pdev->msix ) + msixtbl_init(d); +@@ -1460,6 +1479,10 @@ static int assign_device(struct domain *d, u16 seg, u8 bus, u8 devfn, u32 flag) + } + + done: ++ /* The device is assigned to dom_io so mark it as quarantined */ ++ if ( !rc && d == dom_io ) ++ pdev->quarantine = true; ++ + if ( !has_arch_pdevs(d) && need_iommu(d) ) + iommu_teardown(d); + pcidevs_unlock(); +@@ -1472,6 +1495,7 @@ int deassign_device(struct domain *d, u16 seg, u8 bus, u8 devfn) + { + const struct domain_iommu *hd = dom_iommu(d); + struct pci_dev *pdev = NULL; ++ struct domain *target; + int ret = 0; + + if ( !iommu_enabled || !hd->platform_ops ) +@@ -1482,12 +1506,16 @@ int deassign_device(struct domain *d, u16 seg, u8 bus, u8 devfn) + if ( !pdev ) + return -ENODEV; + ++ /* De-assignment from dom_io should de-quarantine the device */ ++ target = (pdev->quarantine && pdev->domain != dom_io) ? ++ dom_io : hardware_domain; ++ + while ( pdev->phantom_stride ) + { + devfn += pdev->phantom_stride; + if ( PCI_SLOT(devfn) != PCI_SLOT(pdev->devfn) ) + break; +- ret = hd->platform_ops->reassign_device(d, hardware_domain, devfn, ++ ret = hd->platform_ops->reassign_device(d, target, devfn, + pci_to_dev(pdev)); + if ( !ret ) + continue; +@@ -1498,7 +1526,7 @@ int deassign_device(struct domain *d, u16 seg, u8 bus, u8 devfn) + } + + devfn = pdev->devfn; +- ret = hd->platform_ops->reassign_device(d, hardware_domain, devfn, ++ ret = hd->platform_ops->reassign_device(d, target, devfn, + pci_to_dev(pdev)); + if ( ret ) + { +@@ -1508,6 +1536,9 @@ int deassign_device(struct domain *d, u16 seg, u8 bus, u8 devfn) + return ret; + } + ++ if ( pdev->domain == hardware_domain ) ++ pdev->quarantine = false; ++ + pdev->fault.count = 0; + + if ( !has_arch_pdevs(d) && need_iommu(d) ) +@@ -1686,7 +1717,7 @@ int iommu_do_pci_domctl( + ret = hypercall_create_continuation(__HYPERVISOR_domctl, + "h", u_domctl); + else if ( ret ) +- printk(XENLOG_G_ERR "XEN_DOMCTL_assign_device: " ++ printk(XENLOG_G_ERR + "assign %04x:%02x:%02x.%u to dom%d failed (%d)\n", + seg, bus, PCI_SLOT(devfn), PCI_FUNC(devfn), + d->domain_id, ret); +diff --git a/xen/drivers/passthrough/vtd/iommu.c b/xen/drivers/passthrough/vtd/iommu.c +index 4c719d4ee7..19f7d13013 100644 +--- a/xen/drivers/passthrough/vtd/iommu.c ++++ b/xen/drivers/passthrough/vtd/iommu.c +@@ -1338,6 +1338,10 @@ int domain_context_mapping_one( + int agaw, rc, ret; + bool_t flush_dev_iotlb; + ++ /* dom_io is used as a sentinel for quarantined devices */ ++ if ( domain == dom_io ) ++ return 0; ++ + ASSERT(pcidevs_locked()); + spin_lock(&iommu->lock); + maddr = bus_to_context_maddr(iommu, bus); +@@ -1573,6 +1577,10 @@ int domain_context_unmap_one( + int iommu_domid, rc, ret; + bool_t flush_dev_iotlb; + ++ /* dom_io is used as a sentinel for quarantined devices */ ++ if ( domain == dom_io ) ++ return 0; ++ + ASSERT(pcidevs_locked()); + spin_lock(&iommu->lock); + +@@ -1705,6 +1713,10 @@ static int domain_context_unmap(struct domain *domain, u8 devfn, + goto out; + } + ++ /* dom_io is used as a sentinel for quarantined devices */ ++ if ( domain == dom_io ) ++ goto out; ++ + /* + * if no other devices under the same iommu owned by this domain, + * clear iommu in iommu_bitmap and clear domain_id in domid_bitmp +@@ -2389,6 +2401,15 @@ static int reassign_device_ownership( + if ( ret ) + return ret; + ++ if ( devfn == pdev->devfn ) ++ { ++ list_move(&pdev->domain_list, &dom_io->arch.pdev_list); ++ pdev->domain = dom_io; ++ } ++ ++ if ( !has_arch_pdevs(source) ) ++ vmx_pi_hooks_deassign(source); ++ + if ( !has_arch_pdevs(target) ) + vmx_pi_hooks_assign(target); + +@@ -2407,15 +2428,13 @@ static int reassign_device_ownership( + pdev->domain = target; + } + +- if ( !has_arch_pdevs(source) ) +- vmx_pi_hooks_deassign(source); +- + return ret; + } + + static int intel_iommu_assign_device( + struct domain *d, u8 devfn, struct pci_dev *pdev, u32 flag) + { ++ struct domain *s = pdev->domain; + struct acpi_rmrr_unit *rmrr; + int ret = 0, i; + u16 bdf, seg; +@@ -2458,8 +2477,8 @@ static int intel_iommu_assign_device( + } + } + +- ret = reassign_device_ownership(hardware_domain, d, devfn, pdev); +- if ( ret ) ++ ret = reassign_device_ownership(s, d, devfn, pdev); ++ if ( ret || d == dom_io ) + return ret; + + /* Setup rmrr identity mapping */ +@@ -2472,11 +2491,20 @@ static int intel_iommu_assign_device( + ret = rmrr_identity_mapping(d, 1, rmrr, flag); + if ( ret ) + { +- reassign_device_ownership(d, hardware_domain, devfn, pdev); ++ int rc; ++ ++ rc = reassign_device_ownership(d, s, devfn, pdev); + printk(XENLOG_G_ERR VTDPREFIX + " cannot map reserved region (%"PRIx64",%"PRIx64"] for Dom%d (%d)\n", + rmrr->base_address, rmrr->end_address, + d->domain_id, ret); ++ if ( rc ) ++ { ++ printk(XENLOG_ERR VTDPREFIX ++ " failed to reclaim %04x:%02x:%02x.%u from %pd (%d)\n", ++ seg, bus, PCI_SLOT(devfn), PCI_FUNC(devfn), d, rc); ++ domain_crash(d); ++ } + break; + } + } +diff --git a/xen/include/xen/pci.h b/xen/include/xen/pci.h +index 4cfa774615..066364bdef 100644 +--- a/xen/include/xen/pci.h ++++ b/xen/include/xen/pci.h +@@ -88,6 +88,9 @@ struct pci_dev { + + nodeid_t node; /* NUMA node */ + ++ /* Device to be quarantined, don't automatically re-assign to dom0 */ ++ bool quarantine; ++ + enum pdev_type { + DEV_TYPE_PCI_UNKNOWN, + DEV_TYPE_PCIe_ENDPOINT, +-- +2.11.0 + diff --git a/xsa303-0001-xen-arm32-entry-Split-__DEFINE_ENTRY_TRAP-in-two.patch b/xsa303-0001-xen-arm32-entry-Split-__DEFINE_ENTRY_TRAP-in-two.patch new file mode 100644 index 0000000..afb1096 --- /dev/null +++ b/xsa303-0001-xen-arm32-entry-Split-__DEFINE_ENTRY_TRAP-in-two.patch @@ -0,0 +1,74 @@ +From c8cb33fa64c9ccbfa2a494a9dad2e0a763c09176 Mon Sep 17 00:00:00 2001 +From: Julien Grall +Date: Tue, 1 Oct 2019 13:07:53 +0100 +Subject: [PATCH 1/4] xen/arm32: entry: Split __DEFINE_ENTRY_TRAP in two + +The preprocessing macro __DEFINE_ENTRY_TRAP is used to generate trap +entry function. While the macro is fairly small today, follow-up patches +will increase the size signicantly. + +In general, assembly macros are more readable as they allow you to name +parameters and avoid '\'. So the actual implementation of the trap is +now switched to an assembly macro. + +This is part of XSA-303. + +Reported-by: Julien Grall +Signed-off-by: Julien Grall +Reviewed-by: Stefano Stabellini +Reviewed-by: Andre Przywara +--- + xen/arch/arm/arm32/entry.S | 34 +++++++++++++++++++--------------- + 1 file changed, 19 insertions(+), 15 deletions(-) + +diff --git a/xen/arch/arm/arm32/entry.S b/xen/arch/arm/arm32/entry.S +index 0b4cd19abd..4a762e04f1 100644 +--- a/xen/arch/arm/arm32/entry.S ++++ b/xen/arch/arm/arm32/entry.S +@@ -126,24 +126,28 @@ abort_guest_exit_end: + skip_check: + mov pc, lr + +-/* +- * Macro to define trap entry. The iflags corresponds to the list of +- * interrupts (Asynchronous Abort, IRQ, FIQ) to unmask. +- */ ++ /* ++ * Macro to define trap entry. The iflags corresponds to the list of ++ * interrupts (Asynchronous Abort, IRQ, FIQ) to unmask. ++ */ ++ .macro vector trap, iflags ++ SAVE_ALL ++ cpsie \iflags ++ adr lr, return_from_trap ++ mov r0, sp ++ /* ++ * Save the stack pointer in r11. It will be restored after the ++ * trap has been handled (see return_from_trap). ++ */ ++ mov r11, sp ++ bic sp, #7 /* Align the stack pointer (noop on guest trap) */ ++ b do_trap_\trap ++ .endm ++ + #define __DEFINE_TRAP_ENTRY(trap, iflags) \ + ALIGN; \ + trap_##trap: \ +- SAVE_ALL; \ +- cpsie iflags; \ +- adr lr, return_from_trap; \ +- mov r0, sp; \ +- /* \ +- * Save the stack pointer in r11. It will be restored after the \ +- * trap has been handled (see return_from_trap). \ +- */ \ +- mov r11, sp; \ +- bic sp, #7; /* Align the stack pointer (noop on guest trap) */ \ +- b do_trap_##trap ++ vector trap, iflags + + /* Trap handler which unmask IRQ/Abort, keep FIQ masked */ + #define DEFINE_TRAP_ENTRY(trap) __DEFINE_TRAP_ENTRY(trap, ai) +-- +2.11.0 + diff --git a/xsa303-0002-xen-arm32-entry-Fold-the-macro-SAVE_ALL-in-the-macro.patch b/xsa303-0002-xen-arm32-entry-Fold-the-macro-SAVE_ALL-in-the-macro.patch new file mode 100644 index 0000000..35f9c04 --- /dev/null +++ b/xsa303-0002-xen-arm32-entry-Fold-the-macro-SAVE_ALL-in-the-macro.patch @@ -0,0 +1,97 @@ +From be7379207c83fa74f8a6c22a8ea213f02714776f Mon Sep 17 00:00:00 2001 +From: Julien Grall +Date: Tue, 1 Oct 2019 13:15:48 +0100 +Subject: [PATCH 2/4] xen/arm32: entry: Fold the macro SAVE_ALL in the macro + vector + +Follow-up rework will require the macro vector to distinguish between +a trap from a guest vs while in the hypervisor. + +The macro SAVE_ALL already has code to distinguish between the two and +it is only called by the vector macro. So fold the former into the +latter. This will help to avoid duplicating the check. + +This is part of XSA-303. + +Reported-by: Julien Grall +Signed-off-by: Julien Grall +Reviewed-by: Stefano Stabellini +Reviewed-by: Andre Przywara +--- + xen/arch/arm/arm32/entry.S | 46 +++++++++++++++++++++++----------------------- + 1 file changed, 23 insertions(+), 23 deletions(-) + +diff --git a/xen/arch/arm/arm32/entry.S b/xen/arch/arm/arm32/entry.S +index 4a762e04f1..150cbc0b4b 100644 +--- a/xen/arch/arm/arm32/entry.S ++++ b/xen/arch/arm/arm32/entry.S +@@ -13,27 +13,6 @@ + #define RESTORE_BANKED(mode) \ + RESTORE_ONE_BANKED(SP_##mode) ; RESTORE_ONE_BANKED(LR_##mode) ; RESTORE_ONE_BANKED(SPSR_##mode) + +-#define SAVE_ALL \ +- sub sp, #(UREGS_SP_usr - UREGS_sp); /* SP, LR, SPSR, PC */ \ +- push {r0-r12}; /* Save R0-R12 */ \ +- \ +- mrs r11, ELR_hyp; /* ELR_hyp is return address. */\ +- str r11, [sp, #UREGS_pc]; \ +- \ +- str lr, [sp, #UREGS_lr]; \ +- \ +- add r11, sp, #UREGS_kernel_sizeof+4; \ +- str r11, [sp, #UREGS_sp]; \ +- \ +- mrc CP32(r11, HSR); /* Save exception syndrome */ \ +- str r11, [sp, #UREGS_hsr]; \ +- \ +- mrs r11, SPSR_hyp; \ +- str r11, [sp, #UREGS_cpsr]; \ +- and r11, #PSR_MODE_MASK; \ +- cmp r11, #PSR_MODE_HYP; \ +- blne save_guest_regs +- + save_guest_regs: + #ifdef CONFIG_ARM32_HARDEN_BRANCH_PREDICTOR + /* +@@ -52,7 +31,7 @@ save_guest_regs: + ldr r11, =0xffffffff /* Clobber SP which is only valid for hypervisor frames. */ + str r11, [sp, #UREGS_sp] + SAVE_ONE_BANKED(SP_usr) +- /* LR_usr is the same physical register as lr and is saved in SAVE_ALL */ ++ /* LR_usr is the same physical register as lr and is saved by the caller */ + SAVE_BANKED(svc) + SAVE_BANKED(abt) + SAVE_BANKED(und) +@@ -131,7 +110,28 @@ skip_check: + * interrupts (Asynchronous Abort, IRQ, FIQ) to unmask. + */ + .macro vector trap, iflags +- SAVE_ALL ++ /* Save registers in the stack */ ++ sub sp, #(UREGS_SP_usr - UREGS_sp) /* SP, LR, SPSR, PC */ ++ push {r0-r12} /* Save R0-R12 */ ++ mrs r11, ELR_hyp /* ELR_hyp is return address */ ++ str r11, [sp, #UREGS_pc] ++ ++ str lr, [sp, #UREGS_lr] ++ ++ add r11, sp, #(UREGS_kernel_sizeof + 4) ++ ++ str r11, [sp, #UREGS_sp] ++ ++ mrc CP32(r11, HSR) /* Save exception syndrome */ ++ str r11, [sp, #UREGS_hsr] ++ ++ mrs r11, SPSR_hyp ++ str r11, [sp, #UREGS_cpsr] ++ and r11, #PSR_MODE_MASK ++ cmp r11, #PSR_MODE_HYP ++ blne save_guest_regs ++ ++ /* We are ready to handle the trap, setup the registers and jump. */ + cpsie \iflags + adr lr, return_from_trap + mov r0, sp +-- +2.11.0 + diff --git a/xsa303-0003-xen-arm32-Don-t-blindly-unmask-interrupts-on-trap-wi.patch b/xsa303-0003-xen-arm32-Don-t-blindly-unmask-interrupts-on-trap-wi.patch new file mode 100644 index 0000000..5168452 --- /dev/null +++ b/xsa303-0003-xen-arm32-Don-t-blindly-unmask-interrupts-on-trap-wi.patch @@ -0,0 +1,226 @@ +From 098fe877967870ffda2dfd9629a5fd272f6aacdc Mon Sep 17 00:00:00 2001 +From: Julien Grall +Date: Fri, 11 Oct 2019 17:49:28 +0100 +Subject: [PATCH 3/4] xen/arm32: Don't blindly unmask interrupts on trap + without a change of level + +Exception vectors will unmask interrupts regardless the state of them in +the interrupted context. + +One of the consequences is IRQ will be unmasked when receiving an +undefined instruction exception (used by WARN*) from the hypervisor. +This could result to unexpected behavior such as deadlock (if a lock was +shared with interrupts). + +In a nutshell, interrupts should only be unmasked when it is safe to do. +Xen only unmask IRQ and Abort interrupts, so the logic can stay simple. + +As vectors exceptions may be shared between guest and hypervisor, we now +need to have a different policy for the interrupts. + +On exception from hypervisor, each vector will select the list of +interrupts to inherit from the interrupted context. Any interrupts not +listed will be kept masked. + +On exception from the guest, the Abort and IRQ will be unmasked +depending on the exact vector. + +The interrupts will be kept unmasked when the vector cannot used by +either guest or hypervisor. + +Note that each vector is not anymore preceded by ALIGN. This is fine +because the alignment is already bigger than what we need. + +This is part of XSA-303. + +Reported-by: Julien Grall +Signed-off-by: Julien Grall +Reviewed-by: Stefano Stabellini +Reviewed-by: Andre Przywara +--- + xen/arch/arm/arm32/entry.S | 138 +++++++++++++++++++++++++++++++++++---------- + 1 file changed, 109 insertions(+), 29 deletions(-) + +diff --git a/xen/arch/arm/arm32/entry.S b/xen/arch/arm/arm32/entry.S +index 150cbc0b4b..ec90cca093 100644 +--- a/xen/arch/arm/arm32/entry.S ++++ b/xen/arch/arm/arm32/entry.S +@@ -4,6 +4,17 @@ + #include + #include + ++/* ++ * Short-hands to defined the interrupts (A, I, F) ++ * ++ * _ means the interrupt state will not change ++ * X means the state of interrupt X will change ++ * ++ * To be used with msr cpsr_* only ++ */ ++#define IFLAGS_AIF PSR_ABT_MASK | PSR_IRQ_MASK | PSR_FIQ_MASK ++#define IFLAGS_A_F PSR_ABT_MASK | PSR_FIQ_MASK ++ + #define SAVE_ONE_BANKED(reg) mrs r11, reg; str r11, [sp, #UREGS_##reg] + #define RESTORE_ONE_BANKED(reg) ldr r11, [sp, #UREGS_##reg]; msr reg, r11 + +@@ -106,10 +117,18 @@ skip_check: + mov pc, lr + + /* +- * Macro to define trap entry. The iflags corresponds to the list of +- * interrupts (Asynchronous Abort, IRQ, FIQ) to unmask. ++ * Macro to define a trap entry. ++ * ++ * @guest_iflags: Optional list of interrupts to unmask when ++ * entering from guest context. As this is used with cpsie, ++ * the letter (a, i, f) should be used. ++ * ++ * @hyp_iflags: Optional list of interrupts to inherit when ++ * entering from hypervisor context. Any interrupts not ++ * listed will be kept unchanged. As this is used with cpsr_*, ++ * IFLAGS_* short-hands should be used. + */ +- .macro vector trap, iflags ++ .macro vector trap, guest_iflags=n, hyp_iflags=0 + /* Save registers in the stack */ + sub sp, #(UREGS_SP_usr - UREGS_sp) /* SP, LR, SPSR, PC */ + push {r0-r12} /* Save R0-R12 */ +@@ -127,12 +146,39 @@ skip_check: + + mrs r11, SPSR_hyp + str r11, [sp, #UREGS_cpsr] +- and r11, #PSR_MODE_MASK +- cmp r11, #PSR_MODE_HYP +- blne save_guest_regs + ++ /* ++ * We need to distinguish whether we came from guest or ++ * hypervisor context. ++ */ ++ and r0, r11, #PSR_MODE_MASK ++ cmp r0, #PSR_MODE_HYP ++ ++ bne 1f ++ /* ++ * Trap from the hypervisor ++ * ++ * Inherit the state of the interrupts from the hypervisor ++ * context. For that we need to use SPSR (stored in r11) and ++ * modify CPSR accordingly. ++ * ++ * CPSR = (CPSR & ~hyp_iflags) | (SPSR & hyp_iflags) ++ */ ++ mrs r10, cpsr ++ bic r10, r10, #\hyp_iflags ++ and r11, r11, #\hyp_iflags ++ orr r10, r10, r11 ++ msr cpsr_cx, r10 ++ b 2f ++ ++1: ++ /* Trap from the guest */ ++ bl save_guest_regs ++ .if \guest_iflags != n ++ cpsie \guest_iflags ++ .endif ++2: + /* We are ready to handle the trap, setup the registers and jump. */ +- cpsie \iflags + adr lr, return_from_trap + mov r0, sp + /* +@@ -144,20 +190,6 @@ skip_check: + b do_trap_\trap + .endm + +-#define __DEFINE_TRAP_ENTRY(trap, iflags) \ +- ALIGN; \ +-trap_##trap: \ +- vector trap, iflags +- +-/* Trap handler which unmask IRQ/Abort, keep FIQ masked */ +-#define DEFINE_TRAP_ENTRY(trap) __DEFINE_TRAP_ENTRY(trap, ai) +- +-/* Trap handler which unmask Abort, keep IRQ/FIQ masked */ +-#define DEFINE_TRAP_ENTRY_NOIRQ(trap) __DEFINE_TRAP_ENTRY(trap, a) +- +-/* Trap handler which unmask IRQ, keep Abort/FIQ masked */ +-#define DEFINE_TRAP_ENTRY_NOABORT(trap) __DEFINE_TRAP_ENTRY(trap, i) +- + .align 5 + GLOBAL(hyp_traps_vector) + b trap_reset /* 0x00 - Reset */ +@@ -228,14 +260,62 @@ decode_vectors: + + #endif /* CONFIG_HARDEN_BRANCH_PREDICTOR */ + +-DEFINE_TRAP_ENTRY(reset) +-DEFINE_TRAP_ENTRY(undefined_instruction) +-DEFINE_TRAP_ENTRY(hypervisor_call) +-DEFINE_TRAP_ENTRY(prefetch_abort) +-DEFINE_TRAP_ENTRY(guest_sync) +-DEFINE_TRAP_ENTRY_NOIRQ(irq) +-DEFINE_TRAP_ENTRY_NOIRQ(fiq) +-DEFINE_TRAP_ENTRY_NOABORT(data_abort) ++/* Vector not used by the Hypervisor. */ ++trap_reset: ++ vector reset ++ ++/* ++ * Vector only used by the Hypervisor. ++ * ++ * While the exception can be executed with all the interrupts (e.g. ++ * IRQ) unmasked, the interrupted context may have purposefully masked ++ * some of them. So we want to inherit the state from the interrupted ++ * context. ++ */ ++trap_undefined_instruction: ++ vector undefined_instruction, hyp_iflags=IFLAGS_AIF ++ ++/* We should never reach this trap */ ++trap_hypervisor_call: ++ vector hypervisor_call ++ ++/* ++ * Vector only used by the hypervisor. ++ * ++ * While the exception can be executed with all the interrupts (e.g. ++ * IRQ) unmasked, the interrupted context may have purposefully masked ++ * some of them. So we want to inherit the state from the interrupted ++ * context. ++ */ ++trap_prefetch_abort: ++ vector prefetch_abort, hyp_iflags=IFLAGS_AIF ++ ++/* ++ * Vector only used by the hypervisor. ++ * ++ * Data Abort should be rare and most likely fatal. It is best to not ++ * unmask any interrupts to limit the amount of code that can run before ++ * the Data Abort is treated. ++ */ ++trap_data_abort: ++ vector data_abort ++ ++/* Vector only used by the guest. We can unmask Abort/IRQ. */ ++trap_guest_sync: ++ vector guest_sync, guest_iflags=ai ++ ++ ++/* Vector used by the hypervisor and the guest. */ ++trap_irq: ++ vector irq, guest_iflags=a, hyp_iflags=IFLAGS_A_F ++ ++/* ++ * Vector used by the hypervisor and the guest. ++ * ++ * FIQ are not meant to happen, so we don't unmask any interrupts. ++ */ ++trap_fiq: ++ vector fiq + + return_from_trap: + /* +-- +2.11.0 + diff --git a/xsa303-0004-xen-arm64-Don-t-blindly-unmask-interrupts-on-trap-wi.patch b/xsa303-0004-xen-arm64-Don-t-blindly-unmask-interrupts-on-trap-wi.patch new file mode 100644 index 0000000..106cbf9 --- /dev/null +++ b/xsa303-0004-xen-arm64-Don-t-blindly-unmask-interrupts-on-trap-wi.patch @@ -0,0 +1,114 @@ +From c6d290ce157a044dec417fdda8db71e41a37d744 Mon Sep 17 00:00:00 2001 +From: Julien Grall +Date: Mon, 7 Oct 2019 18:10:56 +0100 +Subject: [PATCH 4/4] xen/arm64: Don't blindly unmask interrupts on trap + without a change of level + +Some of the traps without a change of the level (i.e. hypervisor -> +hypervisor) will unmask interrupts regardless the state of them in the +interrupted context. + +One of the consequences is IRQ will be unmasked when receiving a +synchronous exception (used by WARN*()). This could result to unexpected +behavior such as deadlock (if a lock was shared with interrupts). + +In a nutshell, interrupts should only be unmasked when it is safe to +do. Xen only unmask IRQ and Abort interrupts, so the logic can stay +simple: + - hyp_error: All the interrupts are now kept masked. SError should + be pretty rare and if ever happen then we most likely want to + avoid any other interrupts to be generated. The potential main + "caller" is during virtual SError synchronization on the exit + path from the guest (see check_pending_vserror). + + - hyp_sync: The interrupts state is inherited from the interrupted + context. + + - hyp_irq: All the interrupts but IRQ state are inherited from the + interrupted context. IRQ is kept masked. + +This is part of XSA-303. + +Reported-by: Julien Grall +Signed-off-by: Julien Grall +Reviewed-by: Stefano Stabellini +Reviewed-by: Andre Przywara +--- + xen/arch/arm/arm64/entry.S | 47 ++++++++++++++++++++++++++++++++++++++++++---- + 1 file changed, 43 insertions(+), 4 deletions(-) + +diff --git a/xen/arch/arm/arm64/entry.S b/xen/arch/arm/arm64/entry.S +index 2d9a2713a1..3e41ba65b6 100644 +--- a/xen/arch/arm/arm64/entry.S ++++ b/xen/arch/arm/arm64/entry.S +@@ -188,24 +188,63 @@ hyp_error_invalid: + entry hyp=1 + invalid BAD_ERROR + ++/* ++ * SError received while running in the hypervisor mode. ++ * ++ * Technically, we could unmask the IRQ if it were unmasked in the ++ * interrupted context. However, this require to check the PSTATE. For ++ * simplicity, as SError should be rare and potentially fatal, ++ * all interrupts are kept masked. ++ */ + hyp_error: + entry hyp=1 +- msr daifclr, #2 + mov x0, sp + bl do_trap_hyp_serror + exit hyp=1 + +-/* Traps taken in Current EL with SP_ELx */ ++/* ++ * Synchronous exception received while running in the hypervisor mode. ++ * ++ * While the exception could be executed with all the interrupts (e.g. ++ * IRQ) unmasked, the interrupted context may have purposefully masked ++ * some of them. So we want to inherit the state from the interrupted ++ * context. ++ */ + hyp_sync: + entry hyp=1 +- msr daifclr, #6 ++ ++ /* Inherit interrupts */ ++ mrs x0, SPSR_el2 ++ and x0, x0, #(PSR_DBG_MASK | PSR_ABT_MASK | PSR_IRQ_MASK | PSR_FIQ_MASK) ++ msr daif, x0 ++ + mov x0, sp + bl do_trap_hyp_sync + exit hyp=1 + ++/* ++ * IRQ received while running in the hypervisor mode. ++ * ++ * While the exception could be executed with all the interrupts but IRQ ++ * unmasked, the interrupted context may have purposefully masked some ++ * of them. So we want to inherit the state from the interrupt context ++ * and keep IRQ masked. ++ * ++ * XXX: We may want to consider an ordering between interrupts (e.g. if ++ * SError are masked, then IRQ should be masked too). However, this ++ * would require some rework in some paths (e.g. panic, livepatch) to ++ * ensure the ordering is enforced everywhere. ++ */ + hyp_irq: + entry hyp=1 +- msr daifclr, #4 ++ ++ /* Inherit D, A, F interrupts and keep I masked */ ++ mrs x0, SPSR_el2 ++ mov x1, #(PSR_DBG_MASK | PSR_ABT_MASK | PSR_FIQ_MASK) ++ and x0, x0, x1 ++ orr x0, x0, #PSR_IRQ_MASK ++ msr daif, x0 ++ + mov x0, sp + bl do_trap_irq + exit hyp=1 +-- +2.11.0 + diff --git a/xsa483.patch b/xsa483.patch deleted file mode 100644 index 8ecb2e9..0000000 --- a/xsa483.patch +++ /dev/null @@ -1,30 +0,0 @@ -From: Andrii Sultanov -Subject: tools/oxenstored: Reset quota when resetting permissions - -The quota object contains both limits and the current node usage counts. - -When a domain is torn down, the node data itself is cleaned up but the node -usage counts are not. A later domain reusing the same domid can create fewer -nodes before being deemed to be over quota. - -Reset the count when the node permissions are cleaned up. - -This is XSA-483 / CVE-2026-23556. - -Signed-off-by: Andrii Sultanov -Signed-off-by: Andrew Cooper - -diff --git a/tools/ocaml/xenstored/store.ml b/tools/ocaml/xenstored/store.ml -index 9b8dd2812df0..aa9204ead3ec 100644 ---- a/tools/ocaml/xenstored/store.ml -+++ b/tools/ocaml/xenstored/store.ml -@@ -465,7 +465,8 @@ let reset_permissions store domid = - if perms <> node.perms then - Logging.debug "store|node" "Changed permissions for node %s" (Node.get_name node); - Some { node with Node.perms } -- ) store.root -+ ) store.root; -+ store.quota <- Quota.del store.quota domid - - type ops = { - store: t; diff --git a/xsa484.patch b/xsa484.patch deleted file mode 100644 index 522549e..0000000 --- a/xsa484.patch +++ /dev/null @@ -1,89 +0,0 @@ -From 3d0d19ad17f29c64dde4a7baf392da4fd58f3654 Mon Sep 17 00:00:00 2001 -From: Juergen Gross -Date: Mon, 16 Mar 2026 15:06:11 +0100 -Subject: [PATCH] tools/xenstored: make conn_delete_all_transactions() - idempotent - -conn_delete_all_transactions() should be callable in any context, -resetting ALL transaction related data. - -This includes number of active transactions and the transaction -pointer in struct connection. - -So reset conn->trans to NULL in conn_delete_all_transactions() and -do the cleanup for each transaction in destroy_transaction(). - -This avoids triggering the assert() in conn_delete_all_transactions() -in case e.g. ignore_connection() was called while an operation inside -a transaction was performed, or XS_RESET_WATCHES was called in a -transaction. - -This is XSA-484 / CVE-2026-23557. - -Reported-by: Andrii Sultanov -Fixes: 1f9d04fb021c ("xenstored: allow guest to shutdown all its watches/transactions") -Signed-off-by: Juergen Gross ---- - tools/xenstored/transaction.c | 20 +++++++++----------- - 1 file changed, 9 insertions(+), 11 deletions(-) - -diff --git a/tools/xenstored/transaction.c b/tools/xenstored/transaction.c -index 167cd597fd..0825c48859 100644 ---- a/tools/xenstored/transaction.c -+++ b/tools/xenstored/transaction.c -@@ -432,17 +432,23 @@ static int finalize_transaction(struct connection *conn, - static int destroy_transaction(void *_transaction) - { - struct transaction *trans = _transaction; -+ struct connection *conn = trans->conn; - struct accessed_node *i; - - wrl_ntransactions--; - trace_destroy(trans, "transaction"); - while ((i = list_top(&trans->accessed, struct accessed_node, list))) { - if (i->ta_node) -- db_delete(trans->conn, i->trans_name, NULL); -+ db_delete(conn, i->trans_name, NULL); - list_del(&i->list); - talloc_free(i); - } - -+ list_del(&trans->list); -+ domain_transaction_dec(conn); -+ if (list_empty(&conn->transaction_list)) -+ conn->ta_start_time = 0; -+ - return 0; - } - -@@ -523,10 +529,6 @@ int do_transaction_end(const void *ctx, struct connection *conn, - return ENOENT; - - conn->transaction = NULL; -- list_del(&trans->list); -- domain_transaction_dec(conn); -- if (list_empty(&conn->transaction_list)) -- conn->ta_start_time = 0; - - chk_quota = trans->node_created && domain_is_unprivileged(conn); - -@@ -572,14 +574,10 @@ void conn_delete_all_transactions(struct connection *conn) - struct transaction *trans; - - while ((trans = list_top(&conn->transaction_list, -- struct transaction, list))) { -- list_del(&trans->list); -+ struct transaction, list))) - talloc_free(trans); -- } -- -- assert(conn->transaction == NULL); - -- conn->ta_start_time = 0; -+ conn->transaction = NULL; - } - - int check_transactions(struct hashtable *hash) --- -2.53.0 - diff --git a/xsa486.patch b/xsa486.patch deleted file mode 100644 index 654e957..0000000 --- a/xsa486.patch +++ /dev/null @@ -1,181 +0,0 @@ -From: Jan Beulich -Subject: gnttab: split gnttab_map_frame() - -If a domain tries to map status frames in parallel to switching grant -table version from 2 to 1, the mapping operation may put in place P2M -entries referencing MFNs which gnttab_unpopulate_status_frames() is in the -process of freeing. - -Ideally we would refcount pages when entered into P2M tables, but that's a -significant change. Extend the grant-table-locked region instead in -xenmem_add_to_physmap_one() (being the sole caller of gnttab_map_frame()), -such that a race with gnttab_unpopulate_status_frames() is no longer -possible. - -This is XSA-486 / CVE-2026-23558. - -Fixes: 5ce8fafa947c ("Dynamic grant-table sizing") -Fixes: a98dc13703e0 ("Introduce a grant_entry_v2 structure") -Reported-by: Rafal Wojtczuk -Signed-off-by: Jan Beulich -Reviewed-by: Roger Pau Monné - ---- a/xen/arch/arm/mm.c -+++ b/xen/arch/arm/mm.c -@@ -174,12 +174,10 @@ int xenmem_add_to_physmap_one( - switch ( space ) - { - case XENMAPSPACE_grant_table: -- rc = gnttab_map_frame(d, idx, gfn, &mfn); -+ rc = gnttab_map_frame_begin(d, idx, gfn, &mfn); - if ( rc ) - return rc; - -- /* Need to take care of the reference obtained in gnttab_map_frame(). */ -- page = mfn_to_page(mfn); - t = p2m_ram_rw; - - break; -@@ -281,10 +279,23 @@ int xenmem_add_to_physmap_one( - * to drop the reference we took earlier. In all other cases we need to - * drop any reference we took earlier (perhaps indirectly). - */ -- if ( space == XENMAPSPACE_gmfn_foreign ? rc : page != NULL ) -+ switch ( space ) - { -+ default: -+ if ( page ) -+ put_page(page); -+ break; -+ -+ case XENMAPSPACE_grant_table: -+ gnttab_map_frame_end(d, mfn); -+ break; -+ -+ case XENMAPSPACE_gmfn_foreign: -+ if ( !rc ) -+ break; - ASSERT(page != NULL); - put_page(page); -+ break; - } - - return rc; ---- a/xen/arch/x86/mm/p2m.c -+++ b/xen/arch/x86/mm/p2m.c -@@ -2009,11 +2009,9 @@ int xenmem_add_to_physmap_one( - break; - - case XENMAPSPACE_grant_table: -- rc = gnttab_map_frame(d, idx, gfn, &mfn); -+ rc = gnttab_map_frame_begin(d, idx, gfn, &mfn); - if ( rc ) - return rc; -- /* Need to take care of the reference obtained in gnttab_map_frame(). */ -- page = mfn_to_page(mfn); - break; - - case XENMAPSPACE_gmfn: -@@ -2095,19 +2093,28 @@ int xenmem_add_to_physmap_one( - put_gfn(d, gfn_x(gfn)); - - put_both: -- /* -- * In the XENMAPSPACE_gmfn case, we took a ref of the gfn at the top. -- * We also may need to transfer ownership of the page reference to our -- * caller. -- */ -- if ( space == XENMAPSPACE_gmfn ) -+ switch ( space ) - { -+ case XENMAPSPACE_gmfn: -+ /* -+ * We took a ref of the gfn at the top. We also may need to transfer -+ * ownership of the page reference to our caller. -+ */ - put_gfn(d, gmfn); - if ( !rc && extra.ppage ) - { - *extra.ppage = page; - page = NULL; - } -+ break; -+ -+ case XENMAPSPACE_grant_table: -+ /* -+ * We (gnttab_map_frame_begin()) acquired a lock and took a ref of the -+ * page underlying the MFN at the top. -+ */ -+ gnttab_map_frame_end(d, mfn); -+ break; - } - - if ( page ) ---- a/xen/common/grant_table.c -+++ b/xen/common/grant_table.c -@@ -4250,7 +4250,8 @@ int gnttab_acquire_resource( - return rc; - } - --int gnttab_map_frame(struct domain *d, unsigned long idx, gfn_t gfn, mfn_t *mfn) -+int gnttab_map_frame_begin( -+ struct domain *d, unsigned long idx, gfn_t gfn, mfn_t *mfn) - { - int rc = 0; - struct grant_table *gt = d->grant_table; -@@ -4288,11 +4289,19 @@ int gnttab_map_frame(struct domain *d, u - put_page(pg); - } - -- grant_write_unlock(gt); -+ if ( rc ) -+ grant_write_unlock(d->grant_table); - - return rc; - } - -+void gnttab_map_frame_end(struct domain *d, mfn_t mfn) -+{ -+ put_page(mfn_to_page(mfn)); -+ -+ grant_write_unlock(d->grant_table); -+} -+ - static void gnttab_usage_print(struct domain *rd) - { - int first = 1; ---- a/xen/include/xen/grant_table.h -+++ b/xen/include/xen/grant_table.h -@@ -60,8 +60,13 @@ int gnttab_release_mappings(struct domai - int mem_sharing_gref_to_gfn(struct grant_table *gt, grant_ref_t ref, - gfn_t *gfn, uint16_t *status); - --int gnttab_map_frame(struct domain *d, unsigned long idx, gfn_t gfn, -- mfn_t *mfn); -+/* -+ * These need to be used as a pair, as the first (in the success case) returns -+ * with a lock and page reference held which the second needs to drop. -+ */ -+int gnttab_map_frame_begin(struct domain *d, unsigned long idx, gfn_t gfn, -+ mfn_t *mfn); -+void gnttab_map_frame_end(struct domain *d, mfn_t mfn); - - unsigned int gnttab_resource_max_frames(const struct domain *d, unsigned int id); - -@@ -100,12 +105,14 @@ static inline int mem_sharing_gref_to_gf - return -EINVAL; - } - --static inline int gnttab_map_frame(struct domain *d, unsigned long idx, -- gfn_t gfn, mfn_t *mfn) -+static inline int gnttab_map_frame_begin(struct domain *d, unsigned long idx, -+ gfn_t gfn, mfn_t *mfn) - { - return -EINVAL; - } - -+static inline void gnttab_map_frame_end(struct domain *d, mfn_t mfn) {} -+ - static inline unsigned int gnttab_resource_max_frames( - const struct domain *d, unsigned int id) - { diff --git a/xsa490-4.21.patch b/xsa490-4.21.patch deleted file mode 100644 index 5a560cb..0000000 --- a/xsa490-4.21.patch +++ /dev/null @@ -1,43 +0,0 @@ -From: Andrew Cooper -Subject: x86/amd: Mitigate AMD-SN-7052 - -This is XSA-490 / CVE-2025-54518. - -Signed-off-by: Andrew Cooper -Reviewed-by: Roger Pau Monné - -diff --git a/xen/arch/x86/cpu/amd.c b/xen/arch/x86/cpu/amd.c -index 1bb0766ebf13..b5bf2b732e8f 100644 ---- a/xen/arch/x86/cpu/amd.c -+++ b/xen/arch/x86/cpu/amd.c -@@ -1116,11 +1116,25 @@ static void amd_check_bp_cfg(void) - { - uint64_t val, new = 0; - -- /* -- * AMD Erratum #1485. Set bit 5, as instructed. -- */ -- if (!cpu_has_hypervisor && boot_cpu_data.x86 == 0x19 && is_zen4_uarch()) -- new |= (1 << 5); -+ if (!cpu_has_hypervisor) { -+ /* -+ * AMD Erratum #1485. If SMT is enabled and STIBP disabled, -+ * the CPU may fetch incorrect instruction bytes. -+ * -+ * Set bit 5, as instructed. -+ */ -+ if (boot_cpu_data.x86 == 0x19 && is_zen4_uarch()) -+ new |= (1 << 5); -+ -+ /* -+ * AMD SB-7052. CPU OP Cache corruption, causing instructions -+ * to be executed at a higher privilege. -+ * -+ * Set bit 33, as instructed. -+ */ -+ if (boot_cpu_data.x86 == 0x17 && is_zen2_uarch()) -+ new |= (1UL << 33); -+ } - - /* - * On hardware supporting SRSO_MSR_FIX, activate BP_SPEC_REDUCE by diff --git a/xsa491-4.21.patch b/xsa491-4.21.patch deleted file mode 100644 index d1ebc1a..0000000 --- a/xsa491-4.21.patch +++ /dev/null @@ -1,211 +0,0 @@ -From: Jan Beulich -Subject: x86/HVM: add locking to I/O port translation list traversal - -XEN_DOMCTL_ioport_mapping is usable by DM stubdoms, and hence we can't -assume the list to be left unaltered while the guest (really: the -hypervisor on behalf of the guest) is accessing it. - -This is XSA-491 / CVE-2026-42487. - -Fixes: 192c4dabc344 ("domctl and p2m changes for PCI passthru") -Signed-off-by: Jan Beulich -Reviewed-by: Roger Pau Monné - ---- a/xen/arch/x86/domctl.c -+++ b/xen/arch/x86/domctl.c -@@ -663,6 +663,7 @@ long arch_do_domctl( - "ioport_map:add: dom%d gport=%x mport=%x nr=%x\n", - d->domain_id, fgp, fmp, np); - -+ write_lock(&hvm->g2m_ioport_lock); - list_for_each_entry(g2m_ioport, &hvm->g2m_ioport_list, list) - if (g2m_ioport->mport == fmp ) - { -@@ -684,11 +685,14 @@ long arch_do_domctl( - g2m_ioport->np = np; - list_add_tail(&g2m_ioport->list, &hvm->g2m_ioport_list); - } -+ write_unlock(&hvm->g2m_ioport_lock); - if ( !ret ) - ret = ioports_permit_access(d, fmp, fmp + np - 1); - if ( ret && !found && g2m_ioport ) - { -+ write_lock(&hvm->g2m_ioport_lock); - list_del(&g2m_ioport->list); -+ write_unlock(&hvm->g2m_ioport_lock); - xfree(g2m_ioport); - } - } -@@ -697,6 +701,8 @@ long arch_do_domctl( - printk(XENLOG_G_INFO - "ioport_map:remove: dom%d gport=%x mport=%x nr=%x\n", - d->domain_id, fgp, fmp, np); -+ -+ write_lock(&hvm->g2m_ioport_lock); - list_for_each_entry(g2m_ioport, &hvm->g2m_ioport_list, list) - if ( g2m_ioport->mport == fmp ) - { -@@ -704,6 +710,8 @@ long arch_do_domctl( - xfree(g2m_ioport); - break; - } -+ write_unlock(&hvm->g2m_ioport_lock); -+ - ret = ioports_deny_access(d, fmp, fmp + np - 1); - if ( ret && is_hardware_domain(currd) ) - printk(XENLOG_ERR ---- a/xen/arch/x86/hvm/emulate.c -+++ b/xen/arch/x86/hvm/emulate.c -@@ -160,7 +160,6 @@ void hvmemul_cancel(struct vcpu *v) - hvio->mmio_insn_bytes = 0; - hvio->mmio_access = (struct npfec){}; - hvio->mmio_retry = false; -- hvio->g2m_ioport = NULL; - - hvmemul_cache_disable(v); - } ---- a/xen/arch/x86/hvm/hvm.c -+++ b/xen/arch/x86/hvm/hvm.c -@@ -610,6 +610,7 @@ int hvm_domain_initialise(struct domain - spin_lock_init(&d->arch.hvm.irq_lock); - spin_lock_init(&d->arch.hvm.uc_lock); - spin_lock_init(&d->arch.hvm.write_map.lock); -+ rwlock_init(&d->arch.hvm.g2m_ioport_lock); - rwlock_init(&d->arch.hvm.mmcfg_lock); - INIT_LIST_HEAD(&d->arch.hvm.write_map.list); - INIT_LIST_HEAD(&d->arch.hvm.g2m_ioport_list); ---- a/xen/arch/x86/hvm/io.c -+++ b/xen/arch/x86/hvm/io.c -@@ -143,36 +143,56 @@ bool handle_pio(uint16_t port, unsigned - return true; - } - --static bool cf_check g2m_portio_accept( -- const struct hvm_io_handler *handler, const ioreq_t *p) -+/* NB: Returns with the lock held in the success case. */ -+static const struct g2m_ioport *g2m_portio_find_and_lock(struct hvm_domain *hvm, -+ uint64_t addr, -+ uint32_t size) - { -- struct vcpu *curr = current; -- const struct hvm_domain *hvm = &curr->domain->arch.hvm; -- struct hvm_vcpu_io *hvio = &curr->arch.hvm.hvm_io; -- struct g2m_ioport *g2m_ioport; -- unsigned int start, end; -+ const struct g2m_ioport *g2m_ioport; -+ -+ read_lock(&hvm->g2m_ioport_lock); - - list_for_each_entry( g2m_ioport, &hvm->g2m_ioport_list, list ) - { -- start = g2m_ioport->gport; -- end = start + g2m_ioport->np; -- if ( (p->addr >= start) && (p->addr + p->size <= end) ) -- { -- hvio->g2m_ioport = g2m_ioport; -- return 1; -- } -+ unsigned int start = g2m_ioport->gport; -+ -+ if ( addr >= start && addr + size <= start + g2m_ioport->np ) -+ return g2m_ioport; - } - -- return 0; -+ read_unlock(&hvm->g2m_ioport_lock); -+ -+ return NULL; -+} -+ -+static bool cf_check g2m_portio_accept( -+ const struct hvm_io_handler *handler, const ioreq_t *p) -+{ -+ struct hvm_domain *hvm = ¤t->domain->arch.hvm; -+ const struct g2m_ioport *g2m_ioport = -+ g2m_portio_find_and_lock(hvm, p->addr, p->size); -+ -+ if ( !g2m_ioport ) -+ return false; -+ -+ read_unlock(&hvm->g2m_ioport_lock); -+ -+ return true; - } - - static int cf_check g2m_portio_read( - const struct hvm_io_handler *handler, uint64_t addr, uint32_t size, - uint64_t *data) - { -- struct hvm_vcpu_io *hvio = ¤t->arch.hvm.hvm_io; -- const struct g2m_ioport *g2m_ioport = hvio->g2m_ioport; -- unsigned int mport = (addr - g2m_ioport->gport) + g2m_ioport->mport; -+ struct hvm_domain *hvm = ¤t->domain->arch.hvm; -+ const struct g2m_ioport *g2m_ioport = -+ g2m_portio_find_and_lock(hvm, addr, size); -+ unsigned int mport; -+ -+ if ( !g2m_ioport ) -+ return X86EMUL_RETRY; -+ -+ mport = addr - g2m_ioport->gport + g2m_ioport->mport; - - switch ( size ) - { -@@ -189,6 +209,8 @@ static int cf_check g2m_portio_read( - BUG(); - } - -+ read_unlock(&hvm->g2m_ioport_lock); -+ - return X86EMUL_OKAY; - } - -@@ -196,9 +218,15 @@ static int cf_check g2m_portio_write( - const struct hvm_io_handler *handler, uint64_t addr, uint32_t size, - uint64_t data) - { -- struct hvm_vcpu_io *hvio = ¤t->arch.hvm.hvm_io; -- const struct g2m_ioport *g2m_ioport = hvio->g2m_ioport; -- unsigned int mport = (addr - g2m_ioport->gport) + g2m_ioport->mport; -+ struct hvm_domain *hvm = ¤t->domain->arch.hvm; -+ const struct g2m_ioport *g2m_ioport = -+ g2m_portio_find_and_lock(hvm, addr, size); -+ unsigned int mport; -+ -+ if ( !g2m_ioport ) -+ return X86EMUL_RETRY; -+ -+ mport = addr - g2m_ioport->gport + g2m_ioport->mport; - - switch ( size ) - { -@@ -215,6 +243,8 @@ static int cf_check g2m_portio_write( - BUG(); - } - -+ read_unlock(&hvm->g2m_ioport_lock); -+ - return X86EMUL_OKAY; - } - ---- a/xen/arch/x86/include/asm/hvm/domain.h -+++ b/xen/arch/x86/include/asm/hvm/domain.h -@@ -125,6 +125,7 @@ struct hvm_domain { - - /* List of guest to machine IO ports mapping. */ - struct list_head g2m_ioport_list; -+ rwlock_t g2m_ioport_lock; - - /* List of MMCFG regions trapped by Xen. */ - struct list_head mmcfg_regions; ---- a/xen/arch/x86/include/asm/hvm/vcpu.h -+++ b/xen/arch/x86/include/asm/hvm/vcpu.h -@@ -54,8 +54,6 @@ struct hvm_vcpu_io { - unsigned long msix_unmask_address; - unsigned long msix_snoop_address; - unsigned long msix_snoop_gpa; -- -- const struct g2m_ioport *g2m_ioport; - }; - - struct nestedvcpu { diff --git a/xsa492-4.21-01.patch b/xsa492-4.21-01.patch deleted file mode 100644 index 7244ebd..0000000 --- a/xsa492-4.21-01.patch +++ /dev/null @@ -1,264 +0,0 @@ -From: Jan Beulich -Subject: sched: use sequence counter to enlighten vcpu_runstate_get() - -Subsequently XEN_DOMCTL_getdomaininfo will want to invoke the function -without holding a lock, thus allowing parallel execution of potentially -many instances. As was learned from 228ab9992ffb ("domctl: improve -locking during domain destruction"), reverted by d0887cc6b16e, such -parallelism can result in severe lock contention on any (previously) -inner lock. To avoid taking that risk replace the use of the scheduler -lock in vcpu_runstate_get() by a newly introduced sequence counter. -Convert the "no lock if current" property to "use a local counter -instance", thus guaranteeing the loop to exit after the first iteration. - -Skeleton and commentary of the seqcount implementation based on / -derived from Linux 6.11-rc. - -To have runstate_seq placed next to runstate in struct vcpu, without -introducing a new obvious padding hole, yet while keeping the latter -adjacent to runstate_guest{,_area} as well, move runstate down a little. - -This is part of XSA-492. - -Requested-by: Andrew Cooper -Signed-off-by: Jan Beulich -Signed-off-by: Andrew Cooper -Reviewed-by: Roger Pau Monné -Reviewed-by: Juergen Gross - ---- a/xen/common/sched/core.c -+++ b/xen/common/sched/core.c -@@ -281,13 +281,18 @@ static inline void vcpu_runstate_change( - } - - delta = new_entry_time - v->runstate.state_entry_time; -- if ( delta > 0 ) -+ -+ /* Serialization: ->schedule_lock (see ASSERT() above). */ -+ with_seq_write(&v->runstate_seq) - { -- v->runstate.time[v->runstate.state] += delta; -- v->runstate.state_entry_time = new_entry_time; -- } -+ if ( delta > 0 ) -+ { -+ v->runstate.time[v->runstate.state] += delta; -+ v->runstate.state_entry_time = new_entry_time; -+ } - -- v->runstate.state = new_state; -+ v->runstate.state = new_state; -+ } - } - - void sched_guest_idle(void (*idle) (void), unsigned int cpu) -@@ -307,30 +312,18 @@ void sched_guest_idle(void (*idle) (void - void vcpu_runstate_get(const struct vcpu *v, - struct vcpu_runstate_info *runstate) - { -- spinlock_t *lock; -- s_time_t delta; -- struct sched_unit *unit; -+ struct seqcount seq = SEQCNT_ZERO(); -+ const struct seqcount *s = likely(v == current) ? &seq : &v->runstate_seq; - -- rcu_read_lock(&sched_res_rculock); -- -- /* -- * Be careful in case of an idle vcpu: the assignment to a unit might -- * change even with the scheduling lock held, so be sure to use the -- * correct unit for locking in order to avoid triggering an ASSERT() in -- * the unlock function. -- */ -- unit = is_idle_vcpu(v) ? get_sched_res(v->processor)->sched_unit_idle -- : v->sched_unit; -- lock = likely(v == current) ? NULL : unit_schedule_lock_irq(unit); -- memcpy(runstate, &v->runstate, sizeof(*runstate)); -- delta = NOW() - runstate->state_entry_time; -- if ( delta > 0 ) -- runstate->time[runstate->state] += delta; -- -- if ( unlikely(lock != NULL) ) -- unit_schedule_unlock_irq(lock, unit); -+ until_seq_read(s) -+ { -+ s_time_t delta; - -- rcu_read_unlock(&sched_res_rculock); -+ *runstate = v->runstate; -+ delta = NOW() - runstate->state_entry_time; -+ if ( delta > 0 ) -+ runstate->time[runstate->state] += delta; -+ } - } - - uint64_t get_cpu_idle_time(unsigned int cpu) ---- a/xen/include/xen/sched.h -+++ b/xen/include/xen/sched.h -@@ -16,6 +16,7 @@ - #include - #include - #include -+#include - #include - #include - #include -@@ -198,7 +199,6 @@ struct vcpu - - struct sched_unit *sched_unit; - -- struct vcpu_runstate_info runstate; - #ifndef CONFIG_COMPAT - # define runstate_guest(v) ((v)->runstate_guest) - XEN_GUEST_HANDLE(vcpu_runstate_info_t) runstate_guest; /* guest address */ -@@ -210,6 +210,8 @@ struct vcpu - } runstate_guest; /* guest address */ - #endif - struct guest_area runstate_guest_area; -+ struct vcpu_runstate_info runstate; -+ struct seqcount runstate_seq; - unsigned int new_state; - - /* Has the FPU been initialised? */ ---- /dev/null -+++ b/xen/include/xen/seqcount.h -@@ -0,0 +1,139 @@ -+/* SPDX-License-Identifier: GPL-2.0-only */ -+#ifndef XEN_SEQCOUNT_H -+#define XEN_SEQCOUNT_H -+ -+#include -+#include -+ -+#include -+#include -+ -+/* -+ * Sequence counters (seqcount_t) -+ * -+ * This is the raw counting mechanism, without any writer protection. -+ * -+ * Write side critical sections must be serialized (and non-preemptible). -+ * -+ * If readers can be invoked from interrupt contexts, interrupts must also -+ * be respectively disabled before entering the write section. -+ * -+ * This mechanism can't be used if the protected data contains pointers, -+ * as the writer can invalidate a pointer that a reader is following. -+ */ -+struct seqcount { -+ unsigned int sequence; -+}; -+ -+/* -+ * SEQCNT_ZERO() - initializer for seqcount_t -+ * @name: Name of the struct seqcount instance -+ */ -+#define SEQCNT_ZERO() { .sequence = 0 } -+ -+static inline unsigned int seqprop_sequence(const struct seqcount *s) -+{ -+ return ACCESS_ONCE(s->sequence); -+} -+ -+/* -+ * read_seqcount_begin() - begin a seqcount read critical section -+ * @s: Pointer to struct seqcount -+ * -+ * Return: count to be passed to read_seqcount_retry() -+ */ -+static inline unsigned int _read_seqcount_begin(const struct seqcount *s) -+{ -+ unsigned int seq; -+ -+ while ((seq = seqprop_sequence(s)) & 1) -+ cpu_relax(); -+ -+ smp_rmb(); -+ -+ return seq; -+} -+ -+static always_inline unsigned int read_seqcount_begin(const struct seqcount *s) -+{ -+ unsigned int seq = _read_seqcount_begin(s); -+ -+ block_lock_speculation(); -+ -+ return seq; -+} -+ -+/* -+ * read_seqcount_retry() - end a seqcount read critical section -+ * @s: Pointer to struct seqcount -+ * @start: count, from read_seqcount_begin() -+ * -+ * read_seqcount_retry closes the read critical section of given struct -+ * seqcount. If the critical section was invalid, it must be ignored -+ * (and typically retried). -+ * -+ * Return: true if a read section retry is required, else false -+ */ -+static inline bool _read_seqcount_retry(const struct seqcount *s, -+ unsigned int start) -+{ -+ smp_rmb(); -+ return unlikely(seqprop_sequence(s) != start); -+} -+ -+static always_inline bool read_seqcount_retry(const struct seqcount *s, -+ unsigned int start) -+{ -+ return lock_evaluate_nospec(_read_seqcount_retry(s, start)); -+} -+ -+/* Loops until a consistent count has been observed across the loop body. */ -+#define until_seq_read(seq) \ -+ for ( unsigned int retry_ = 1, count_; \ -+ retry_ && (count_ = read_seqcount_begin(seq), true); \ -+ retry_ = read_seqcount_retry(seq, count_) ) -+ -+/* -+ * write_seqcount_begin() - start a struct seqcount write side critical section -+ * @s: Pointer to struct seqcount -+ * -+ * Context: sequence counter write side sections must be serialized. -+ * If readers can be invoked from interrupt context, interrupts must be -+ * respectively disabled. -+ */ -+static inline void write_seqcount_begin(struct seqcount *s) -+{ -+ add_sized(&s->sequence, 1); -+ smp_wmb(); -+} -+ -+/* -+ * write_seqcount_end() - end a struct seqcount write side critical section -+ * @s: Pointer to seqcount -+ */ -+static inline void write_seqcount_end(struct seqcount *s) -+{ -+ smp_wmb(); -+ add_sized(&s->sequence, 1); -+} -+ -+/* -+ * Not really a loop, but we need write_seqcount_{begin,end}() in the correct -+ * position. -+ */ -+#define with_seq_write(seq) \ -+ for ( bool once_ = true; \ -+ once_ && (write_seqcount_begin(seq), true); \ -+ (write_seqcount_end(seq), once_ = false) ) -+ -+#endif /* XEN_SEQCOUNT_H */ -+ -+/* -+ * Local variables: -+ * mode: C -+ * c-file-style: "BSD" -+ * c-basic-offset: 4 -+ * tab-width: 4 -+ * indent-tabs-mode: nil -+ * End: -+ */ diff --git a/xsa492-4.21-02.patch b/xsa492-4.21-02.patch deleted file mode 100644 index 75ca8ca..0000000 --- a/xsa492-4.21-02.patch +++ /dev/null @@ -1,104 +0,0 @@ -From: Jan Beulich -Subject: domctl: handle XEN_DOMCTL_getdomaininfo without acquiring domctl lock - -getdomaininfo() is not called under consistently the same lock. Thus, -with caller side locking irrelevant, it can as well be called with the -domctl lock not held. (Callers not pausing the domain they want to -retrieve information for already need to be aware that not all of the -data returned can be relied on as being consistent; most data will also -be stale by the time the caller gets to look at it.) - -Move the handling not only ahead of acquiring the lock, but also ahead -of the XSM check, leveraging that the sub-op has its own hook. - -While moving, convert an assignment to an assertion: The domain in -question was determined from the field which previously was "updated". - -This is part of XSA-492. - -Fixes: 5513bd0b4675 ("add xenstore domain flag to hypervisor") -Reported-by: Andrew Cooper -Signed-off-by: Jan Beulich -Reviewed-by: Roger Pau Monné -Acked-by: Daniel P. Smith - ---- a/xen/common/domctl.c -+++ b/xen/common/domctl.c -@@ -318,6 +318,26 @@ long do_domctl(XEN_GUEST_HANDLE_PARAM(xe - break; - } - -+ /* Handle sub-ops not requiring the domctl lock. */ -+ switch ( op->cmd ) -+ { -+ case XEN_DOMCTL_getdomaininfo: -+ ret = xsm_getdomaininfo(XSM_XS_PRIV, d); -+ if ( !ret ) -+ { -+ getdomaininfo(d, &op->u.getdomaininfo); -+ -+ ASSERT(op->domain == op->u.getdomaininfo.domain); -+ copyback = true; -+ } -+ -+ goto domctl_out_unlock_domonly; -+ -+ default: -+ /* Everything else handled further down. */ -+ break; -+ } -+ - ret = xsm_domctl(XSM_OTHER, d, op->cmd, - /* SSIDRef only applicable for cmd == createdomain */ - op->u.createdomain.ssidref); -@@ -516,17 +536,6 @@ long do_domctl(XEN_GUEST_HANDLE_PARAM(xe - copyback = 1; - break; - -- case XEN_DOMCTL_getdomaininfo: -- ret = xsm_getdomaininfo(XSM_XS_PRIV, d); -- if ( ret ) -- break; -- -- getdomaininfo(d, &op->u.getdomaininfo); -- -- op->domain = op->u.getdomaininfo.domain; -- copyback = 1; -- break; -- - case XEN_DOMCTL_getvcpucontext: - { - vcpu_guest_context_u c = { .nat = NULL }; ---- a/xen/include/xsm/dummy.h -+++ b/xen/include/xsm/dummy.h -@@ -172,9 +172,13 @@ static XSM_INLINE int cf_check xsm_domct - case XEN_DOMCTL_bind_pt_irq: - case XEN_DOMCTL_unbind_pt_irq: - return xsm_default_action(XSM_DM_PRIV, current->domain, d); -- case XEN_DOMCTL_getdomaininfo: - case XEN_DOMCTL_get_domain_state: - return xsm_default_action(XSM_XS_PRIV, current->domain, d); -+ -+ case XEN_DOMCTL_getdomaininfo: -+ ASSERT_UNREACHABLE(); -+ return -EILSEQ; -+ - default: - return xsm_default_action(XSM_PRIV, current->domain, d); - } ---- a/xen/xsm/flask/hooks.c -+++ b/xen/xsm/flask/hooks.c -@@ -682,8 +682,12 @@ static int cf_check flask_domctl(struct - */ - return avc_current_has_perm(ssidref, SECCLASS_DOMAIN, DOMAIN__CREATE, NULL); - -- /* These have individual XSM hooks (common/domctl.c) */ -+ /* These have individual XSM hooks and don't make it here. */ - case XEN_DOMCTL_getdomaininfo: -+ ASSERT_UNREACHABLE(); -+ return -EILSEQ; -+ -+ /* These have individual XSM hooks (common/domctl.c) */ - case XEN_DOMCTL_scheduler_op: - case XEN_DOMCTL_irq_permission: - case XEN_DOMCTL_iomem_permission: diff --git a/xsa492-4.21-03.patch b/xsa492-4.21-03.patch deleted file mode 100644 index 5a0db22..0000000 --- a/xsa492-4.21-03.patch +++ /dev/null @@ -1,87 +0,0 @@ -From: Daniel P. Smith -Subject: domctl: protect locking for get_domain_state - -When DOMID_INVALID is passed, the dom exec handler lock is being taken -without any check that the domain is even allowed to take the lock. This -allows for an unauthorized domain to DoS the get_domain_state domctl op. -Move to consider the op effectively being called against the hypervisor. -Thus it is the target of the call being invoked to identify the last -domain with a state change. The subsequent check of whether the source -domain is allowed the state of the last domain to change state is still -relevant. - -This is part of XSA-492. - -Signed-off-by: Daniel P. Smith -Signed-off-by: Jan Beulich -Reviewed-by: Roger Pau Monné - ---- a/tools/flask/policy/modules/xenstore.te -+++ b/tools/flask/policy/modules/xenstore.te -@@ -14,6 +14,7 @@ allow xenstore_t xen_t:xen writeconsole; - # Xenstore queries domaininfo on all domains - allow xenstore_t domain_type:domain getdomaininfo; - allow xenstore_t domain_type:domain2 get_domain_state; -+allow xenstore_t domxen_t:domain2 get_domain_state; - - # As a shortcut, the following 3 rules are used instead of adding a domain_comms - # rule between xenstore_t and every domain type that talks to xenstore ---- a/xen/common/domain.c -+++ b/xen/common/domain.c -@@ -216,12 +216,8 @@ int get_domain_state(struct xen_domctl_g - if ( info->pad0 ) - return -EINVAL; - -- if ( d ) -+ if ( d != dom_xen ) - { -- rc = xsm_get_domain_state(XSM_XS_PRIV, d); -- if ( rc ) -- return rc; -- - set_domain_state_info(info, d); - - return 0; ---- a/xen/common/domctl.c -+++ b/xen/common/domctl.c -@@ -304,13 +304,19 @@ long do_domctl(XEN_GUEST_HANDLE_PARAM(xe - fallthrough; - case XEN_DOMCTL_test_assign_device: - case XEN_DOMCTL_vm_event_op: -- case XEN_DOMCTL_get_domain_state: - if ( op->domain == DOMID_INVALID ) - { - d = NULL; - break; - } - fallthrough; -+ case XEN_DOMCTL_get_domain_state: -+ if ( op->domain == DOMID_INVALID ) -+ { -+ d = dom_xen; -+ break; -+ } -+ fallthrough; - default: - d = rcu_lock_domain_by_id(op->domain); - if ( !d ) -@@ -863,7 +869,9 @@ long do_domctl(XEN_GUEST_HANDLE_PARAM(xe - break; - - case XEN_DOMCTL_get_domain_state: -- ret = get_domain_state(&op->u.get_domain_state, d, &op->domain); -+ ret = xsm_get_domain_state(XSM_XS_PRIV, d); -+ if ( !ret ) -+ ret = get_domain_state(&op->u.get_domain_state, d, &op->domain); - if ( !ret ) - copyback = true; - break; -@@ -876,7 +884,7 @@ long do_domctl(XEN_GUEST_HANDLE_PARAM(xe - domctl_lock_release(); - - domctl_out_unlock_domonly: -- if ( d && d != dom_io ) -+ if ( d && !is_system_domain(d) ) - rcu_unlock_domain(d); - - if ( copyback && __copy_to_guest(u_domctl, op, 1) ) diff --git a/xsa492-4.21-04.patch b/xsa492-4.21-04.patch deleted file mode 100644 index 481ff5d..0000000 --- a/xsa492-4.21-04.patch +++ /dev/null @@ -1,81 +0,0 @@ -From: Jan Beulich -Subject: domctl: handle XEN_DOMCTL_get_domain_state without acquiring domctl lock - -get_domain_state() uses its own locking. Thus, with caller side locking -irrelevant, it can as well be called with the domctl lock not held. - -Move the handling not only ahead of acquiring the lock, but also ahead -of the XSM check, leveraging that the sub-op has its own hook. - -This is part of XSA-492. - -Fixes: 3ad3df1bd0aa ("xen: add new domctl get_domain_state") -Reported-by: Andrew Cooper -Signed-off-by: Jan Beulich -Acked-by: Daniel P. Smith -Reviewed-by: Roger Pau Monné - ---- a/xen/common/domctl.c -+++ b/xen/common/domctl.c -@@ -339,6 +339,14 @@ long do_domctl(XEN_GUEST_HANDLE_PARAM(xe - - goto domctl_out_unlock_domonly; - -+ case XEN_DOMCTL_get_domain_state: -+ ret = xsm_get_domain_state(XSM_XS_PRIV, d); -+ if ( !ret ) -+ ret = get_domain_state(&op->u.get_domain_state, d, &op->domain); -+ if ( !ret ) -+ copyback = true; -+ goto domctl_out_unlock_domonly; -+ - default: - /* Everything else handled further down. */ - break; -@@ -868,14 +876,6 @@ long do_domctl(XEN_GUEST_HANDLE_PARAM(xe - ret = -EOPNOTSUPP; - break; - -- case XEN_DOMCTL_get_domain_state: -- ret = xsm_get_domain_state(XSM_XS_PRIV, d); -- if ( !ret ) -- ret = get_domain_state(&op->u.get_domain_state, d, &op->domain); -- if ( !ret ) -- copyback = true; -- break; -- - default: - ret = arch_do_domctl(op, d, u_domctl); - break; ---- a/xen/include/xsm/dummy.h -+++ b/xen/include/xsm/dummy.h -@@ -172,10 +172,9 @@ static XSM_INLINE int cf_check xsm_domct - case XEN_DOMCTL_bind_pt_irq: - case XEN_DOMCTL_unbind_pt_irq: - return xsm_default_action(XSM_DM_PRIV, current->domain, d); -- case XEN_DOMCTL_get_domain_state: -- return xsm_default_action(XSM_XS_PRIV, current->domain, d); - - case XEN_DOMCTL_getdomaininfo: -+ case XEN_DOMCTL_get_domain_state: - ASSERT_UNREACHABLE(); - return -EILSEQ; - ---- a/xen/xsm/flask/hooks.c -+++ b/xen/xsm/flask/hooks.c -@@ -684,6 +684,7 @@ static int cf_check flask_domctl(struct - - /* These have individual XSM hooks and don't make it here. */ - case XEN_DOMCTL_getdomaininfo: -+ case XEN_DOMCTL_get_domain_state: - ASSERT_UNREACHABLE(); - return -EILSEQ; - -@@ -694,7 +695,6 @@ static int cf_check flask_domctl(struct - case XEN_DOMCTL_memory_mapping: - case XEN_DOMCTL_set_target: - case XEN_DOMCTL_vm_event_op: -- case XEN_DOMCTL_get_domain_state: - - /* These have individual XSM hooks (arch/../domctl.c) */ - case XEN_DOMCTL_bind_pt_irq: diff --git a/xsa492-4.21-05.patch b/xsa492-4.21-05.patch deleted file mode 100644 index cb7beaa..0000000 --- a/xsa492-4.21-05.patch +++ /dev/null @@ -1,156 +0,0 @@ -From: Jan Beulich -Subject: domain: locking for iomem_caps accesses - -In order to be able to pull at least the XEN_DOMCTL_iomem_mapping handling -out of the domctl-locked region, a separate (per-domain) lock is needed to -synchronize in particular with XEN_DOMCTL_iomem_permission. - -Locking is added only as far as domctl-s are affected. Uses presently -outside of the domctl lock may want dealing with subsequently (perhaps -limited to non-__init code). - -This is part of XSA-492. - -Signed-off-by: Jan Beulich -Reviewed-by: Roger Pau Monné - ---- a/xen/common/domain.c -+++ b/xen/common/domain.c -@@ -518,10 +518,15 @@ static int late_hwdom_init(struct domain - * may be modified after this hypercall returns if a more complex - * device model is desired. - */ -+ write_lock(&dom0->caps_lock); - rangeset_swap(d->irq_caps, dom0->irq_caps); - rangeset_swap(d->iomem_caps, dom0->iomem_caps); - #ifdef CONFIG_X86 - rangeset_swap(d->arch.ioport_caps, dom0->arch.ioport_caps); -+#endif -+ write_unlock(&dom0->caps_lock); -+ -+#ifdef CONFIG_X86 - setup_io_bitmap(d); - setup_io_bitmap(dom0); - #endif -@@ -873,6 +878,7 @@ struct domain *domain_create(domid_t dom - rspin_lock_init_prof(d, domain_lock); - rspin_lock_init_prof(d, page_alloc_lock); - spin_lock_init(&d->hypercall_deadlock_mutex); -+ rwlock_init(&d->caps_lock); - INIT_PAGE_LIST_HEAD(&d->page_list); - INIT_PAGE_LIST_HEAD(&d->extra_page_list); - INIT_PAGE_LIST_HEAD(&d->xenpage_list); ---- a/xen/common/domctl.c -+++ b/xen/common/domctl.c -@@ -267,6 +267,35 @@ static struct vnuma_info *vnuma_init(con - return ERR_PTR(ret); - } - -+void iocaps_double_lock(struct domain *d, bool write) -+{ -+ struct domain *currd = current->domain; -+ -+ if ( d->domain_id > currd->domain_id ) -+ read_lock(&currd->caps_lock); -+ -+ if ( write ) -+ write_lock(&d->caps_lock); -+ else -+ read_lock(&d->caps_lock); -+ -+ if ( d->domain_id < currd->domain_id ) -+ read_lock(&currd->caps_lock); -+} -+ -+void iocaps_double_unlock(struct domain *d, bool write) -+{ -+ struct domain *currd = current->domain; -+ -+ if ( d != currd ) -+ read_unlock(&currd->caps_lock); -+ -+ if ( write ) -+ write_unlock(&d->caps_lock); -+ else -+ read_unlock(&d->caps_lock); -+} -+ - static bool is_stable_domctl(uint32_t cmd) - { - return cmd == XEN_DOMCTL_get_domain_state; -@@ -687,6 +716,8 @@ long do_domctl(XEN_GUEST_HANDLE_PARAM(xe - if ( (mfn + nr_mfns - 1) < mfn ) /* wrap? */ - break; - -+ iocaps_double_lock(d, true); -+ - if ( !iomem_access_permitted(current->domain, - mfn, mfn + nr_mfns - 1) || - xsm_iomem_permission(XSM_HOOK, d, mfn, mfn + nr_mfns - 1, allow) ) -@@ -695,6 +726,8 @@ long do_domctl(XEN_GUEST_HANDLE_PARAM(xe - ret = iomem_permit_access(d, mfn, mfn + nr_mfns - 1); - else - ret = iomem_deny_access(d, mfn, mfn + nr_mfns - 1); -+ -+ iocaps_double_unlock(d, true); - break; - } - -@@ -719,19 +752,15 @@ long do_domctl(XEN_GUEST_HANDLE_PARAM(xe - break; - #endif - -+ iocaps_double_lock(d, false); -+ - ret = -EPERM; - if ( !iomem_access_permitted(current->domain, mfn, mfn_end) || -- !iomem_access_permitted(d, mfn, mfn_end) ) -- break; -- -- ret = xsm_iomem_mapping(XSM_HOOK, d, mfn, mfn_end, add); -- if ( ret ) -- break; -- -- if ( !paging_mode_translate(d) ) -- break; -- -- if ( add ) -+ !iomem_access_permitted(d, mfn, mfn_end) || -+ (ret = xsm_iomem_mapping(XSM_HOOK, d, mfn, mfn_end, add)) || -+ !paging_mode_translate(d) ) -+ /* Nothing. */; -+ else if ( add ) - { - printk(XENLOG_G_DEBUG - "memory_map:add: dom%d gfn=%lx mfn=%lx nr=%lx\n", -@@ -755,6 +784,8 @@ long do_domctl(XEN_GUEST_HANDLE_PARAM(xe - "memory_map: error %ld removing dom%d access to [%lx,%lx]\n", - ret, d->domain_id, mfn, mfn_end); - } -+ -+ iocaps_double_unlock(d, false); - break; - } - ---- a/xen/include/xen/iocap.h -+++ b/xen/include/xen/iocap.h -@@ -12,6 +12,9 @@ - #include - #include - -+void iocaps_double_lock(struct domain *d, bool write); -+void iocaps_double_unlock(struct domain *d, bool write); -+ - static inline int iomem_permit_access(struct domain *d, unsigned long s, - unsigned long e) - { ---- a/xen/include/xen/sched.h -+++ b/xen/include/xen/sched.h -@@ -536,6 +536,7 @@ struct domain - #endif - - /* I/O capabilities (access to IRQs and memory-mapped I/O). */ -+ rwlock_t caps_lock; - struct rangeset *iomem_caps; - struct rangeset *irq_caps; - diff --git a/xsa492-4.21-06.patch b/xsa492-4.21-06.patch deleted file mode 100644 index c9e0061..0000000 --- a/xsa492-4.21-06.patch +++ /dev/null @@ -1,84 +0,0 @@ -From: Jan Beulich -Subject: x86/domain: locking for ioport_caps accesses - -In order to be able to pull at least the XEN_DOMCTL_ioport_mapping -handling out of the domctl-locked region, the new separate (per-domain) -lock is used to synchronize in particular with -XEN_DOMCTL_ioport_permission. - -Locking is added only as far as domctl-s are affected. Uses presently -outside of the domctl lock may want dealing with subsequently (perhaps -limited to non-__init code). - -This is part of XSA-492. - -Signed-off-by: Jan Beulich -Reviewed-by: Roger Pau Monné - ---- a/xen/arch/x86/domctl.c -+++ b/xen/arch/x86/domctl.c -@@ -233,6 +233,8 @@ long arch_do_domctl( - unsigned int np = domctl->u.ioport_permission.nr_ports; - int allow = domctl->u.ioport_permission.allow_access; - -+ iocaps_double_lock(d, true); -+ - if ( (fp + np) <= fp || (fp + np) > MAX_IOPORTS ) - ret = -EINVAL; - else if ( !ioports_access_permitted(currd, fp, fp + np - 1) || -@@ -242,6 +244,8 @@ long arch_do_domctl( - ret = ioports_permit_access(d, fp, fp + np - 1); - else - ret = ioports_deny_access(d, fp, fp + np - 1); -+ -+ iocaps_double_unlock(d, true); - break; - } - -@@ -648,16 +652,13 @@ long arch_do_domctl( - break; - } - -- ret = -EPERM; -- if ( !ioports_access_permitted(currd, fmp, fmp + np - 1) ) -- break; -- -- ret = xsm_ioport_mapping(XSM_HOOK, d, fmp, fmp + np - 1, add); -- if ( ret ) -- break; -- - hvm = &d->arch.hvm; -- if ( add ) -+ iocaps_double_lock(d, true); -+ -+ if ( !ioports_access_permitted(currd, fmp, fmp + np - 1) || -+ (ret = xsm_ioport_mapping(XSM_HOOK, d, fmp, fmp + np - 1, add)) ) -+ ret = ret ?: -EPERM; -+ else if ( add ) - { - printk(XENLOG_G_INFO - "ioport_map:add: dom%d gport=%x mport=%x nr=%x\n", -@@ -718,6 +720,8 @@ long arch_do_domctl( - "ioport_map: error %ld denying dom%d access to [%x,%x]\n", - ret, d->domain_id, fmp, fmp + np - 1); - } -+ -+ iocaps_double_unlock(d, true); - break; - } - ---- a/xen/arch/x86/setup.c -+++ b/xen/arch/x86/setup.c -@@ -2339,9 +2339,12 @@ void __hwdom_init setup_io_bitmap(struct - return; - - bitmap_fill(d->arch.hvm.io_bitmap, 0x10000); -+ -+ read_lock(&d->caps_lock); - if ( rangeset_report_ranges(d->arch.ioport_caps, 0, 0x10000, - io_bitmap_cb, d) ) - BUG(); -+ read_unlock(&d->caps_lock); - - /* - * We need to trap 4-byte accesses to 0xcf8 (see admin_io_okay(), diff --git a/xsa492-4.21-07.patch b/xsa492-4.21-07.patch deleted file mode 100644 index e343773..0000000 --- a/xsa492-4.21-07.patch +++ /dev/null @@ -1,202 +0,0 @@ -From: Jan Beulich -Subject: domain: locking for irq_caps accesses - -In order to be able to pull at least the XEN_DOMCTL_{,un}bind_pt_irq -handling out of the domctl-locked region, a separate (per-domain) lock is -needed to synchronize in particular with XEN_DOMCTL_{irq,gsi}_permission. - -Locking is added only as far as domctl-s are affected. Uses presently -outside of the domctl lock may want dealing with subsequently (perhaps -limited to non-__init code). - -This is part of XSA-492. - -Signed-off-by: Jan Beulich -Reviewed-by: Roger Pau Monné -Reviewed-by: Julien Grall - ---- a/xen/arch/arm/domctl.c -+++ b/xen/arch/arm/domctl.c -@@ -76,6 +76,7 @@ long arch_do_domctl(struct xen_domctl *d - case XEN_DOMCTL_bind_pt_irq: - { - int rc; -+ struct domain *currd = current->domain; - struct xen_domctl_bind_pt_irq *bind = &domctl->u.bind_pt_irq; - uint32_t irq = bind->u.spi.spi; - uint32_t virq = bind->machine_irq; -@@ -107,21 +108,26 @@ long arch_do_domctl(struct xen_domctl *d - if ( rc ) - return rc; - -- if ( !irq_access_permitted(current->domain, irq) ) -- return -EPERM; -+ read_lock(&currd->caps_lock); - -- if ( !vgic_reserve_virq(d, virq) ) -- return -EBUSY; -- -- rc = route_irq_to_guest(d, virq, irq, "routed IRQ"); -- if ( rc ) -- vgic_free_virq(d, virq); -+ if ( !irq_access_permitted(currd, irq) ) -+ rc = -EPERM; -+ else if ( !vgic_reserve_virq(d, virq) ) -+ rc = -EBUSY; -+ else -+ { -+ rc = route_irq_to_guest(d, virq, irq, "routed IRQ"); -+ if ( rc ) -+ vgic_free_virq(d, virq); -+ } - -+ read_unlock(&currd->caps_lock); - return rc; - } - case XEN_DOMCTL_unbind_pt_irq: - { - int rc; -+ struct domain *currd = current->domain; - struct xen_domctl_bind_pt_irq *bind = &domctl->u.bind_pt_irq; - uint32_t irq = bind->u.spi.spi; - uint32_t virq = bind->machine_irq; -@@ -138,16 +144,15 @@ long arch_do_domctl(struct xen_domctl *d - if ( rc ) - return rc; - -- if ( !irq_access_permitted(current->domain, irq) ) -- return -EPERM; -- -- rc = release_guest_irq(d, virq); -- if ( rc ) -- return rc; -+ read_lock(&currd->caps_lock); - -- vgic_free_virq(d, virq); -+ if ( !irq_access_permitted(currd, irq) ) -+ rc = -EPERM; -+ else if ( !(rc = release_guest_irq(d, virq)) ) -+ vgic_free_virq(d, virq); - -- return 0; -+ read_unlock(&currd->caps_lock); -+ return rc; - } - - case XEN_DOMCTL_vuart_op: ---- a/xen/arch/x86/domctl.c -+++ b/xen/arch/x86/domctl.c -@@ -267,16 +267,17 @@ long arch_do_domctl( - break; - } - -- ret = -EPERM; -+ iocaps_double_lock(d, true); -+ - if ( !irq_access_permitted(currd, irq) || - xsm_irq_permission(XSM_HOOK, d, irq, flags) ) -- break; -- -- if ( flags ) -+ ret = -EPERM; -+ else if ( flags ) - ret = irq_permit_access(d, irq); - else - ret = irq_deny_access(d, irq); - -+ iocaps_double_unlock(d, true); - break; - } - -@@ -579,20 +580,27 @@ long arch_do_domctl( - break; - - irq = domain_pirq_to_irq(d, bind->machine_irq); -- ret = -EPERM; -- if ( irq <= 0 || !irq_access_permitted(currd, irq) ) -- break; -+ if ( irq <= 0 ) -+ ret = -EPERM; - -- ret = -ESRCH; -- if ( is_iommu_enabled(d) ) -+ read_lock(&currd->caps_lock); -+ -+ if ( !irq_access_permitted(currd, irq) ) -+ ret = -EPERM; -+ else if ( is_iommu_enabled(d) ) - { - pcidevs_lock(); - ret = pt_irq_create_bind(d, bind); - pcidevs_unlock(); -+ -+ if ( ret < 0 ) -+ printk(XENLOG_G_ERR "pt_irq_create_bind failed (%ld) for %pd\n", -+ ret, d); - } -- if ( ret < 0 ) -- printk(XENLOG_G_ERR "pt_irq_create_bind failed (%ld) for dom%d\n", -- ret, d->domain_id); -+ else -+ ret = -ESRCH; -+ -+ read_unlock(&currd->caps_lock); - break; - } - -@@ -605,23 +613,26 @@ long arch_do_domctl( - if ( !is_hvm_domain(d) ) - break; - -- ret = -EPERM; -- if ( irq <= 0 || !irq_access_permitted(currd, irq) ) -- break; -- - ret = xsm_unbind_pt_irq(XSM_HOOK, d, bind); - if ( ret ) - break; - -- if ( is_iommu_enabled(d) ) -+ read_lock(&currd->caps_lock); -+ -+ if ( !irq_access_permitted(currd, irq) ) -+ ret = -EPERM; -+ else if ( is_iommu_enabled(d) ) - { - pcidevs_lock(); - ret = pt_irq_destroy_bind(d, bind); - pcidevs_unlock(); -+ -+ if ( ret < 0 ) -+ printk(XENLOG_G_ERR "pt_irq_destroy_bind failed (%ld) for %pd\n", -+ ret, d); - } -- if ( ret < 0 ) -- printk(XENLOG_G_ERR "pt_irq_destroy_bind failed (%ld) for dom%d\n", -- ret, d->domain_id); -+ -+ read_unlock(&currd->caps_lock); - break; - } - ---- a/xen/common/domctl.c -+++ b/xen/common/domctl.c -@@ -695,6 +695,9 @@ long do_domctl(XEN_GUEST_HANDLE_PARAM(xe - ret = -EINVAL; - break; - } -+ -+ iocaps_double_lock(d, true); -+ - irq = pirq_access_permitted(current->domain, pirq); - if ( !irq || xsm_irq_permission(XSM_HOOK, d, irq, allow) ) - ret = -EPERM; -@@ -702,6 +705,8 @@ long do_domctl(XEN_GUEST_HANDLE_PARAM(xe - ret = irq_permit_access(d, irq); - else - ret = irq_deny_access(d, irq); -+ -+ iocaps_double_unlock(d, true); - break; - } - #endif diff --git a/xsa492-4.21-08.patch b/xsa492-4.21-08.patch deleted file mode 100644 index 4aefbf4..0000000 --- a/xsa492-4.21-08.patch +++ /dev/null @@ -1,85 +0,0 @@ -From: Jan Beulich -Subject: XSM/Flask: split the .iomem_mapping() hook - -It's used twice in entirely different situations. The use in do_domctl() -wants to become an ordinary XSM_DM_PRIV invocation, while the one in vPCI -code need to remain XSM_HOOK (it may plausibly become XSM_TARGET). For -Flask, the same backing function will continue to be used for the time -being. - -This is part of XSA-492. - -Signed-off-by: Jan Beulich -Acked-by: Daniel P. Smith - ---- a/xen/drivers/vpci/header.c -+++ b/xen/drivers/vpci/header.c -@@ -67,7 +67,7 @@ static int cf_check map_range( - return -EPERM; - } - -- rc = xsm_iomem_mapping(XSM_HOOK, map->d, map_mfn, m_end, map->map); -+ rc = xsm_iomem_mapping_vpci(XSM_HOOK, map->d, map_mfn, m_end, map->map); - if ( rc ) - { - printk(XENLOG_G_WARNING ---- a/xen/include/xsm/dummy.h -+++ b/xen/include/xsm/dummy.h -@@ -580,6 +580,13 @@ static XSM_INLINE int cf_check xsm_iomem - return xsm_default_action(action, current->domain, d); - } - -+static XSM_INLINE int cf_check xsm_iomem_mapping_vpci( -+ XSM_DEFAULT_ARG struct domain *d, uint64_t s, uint64_t e, uint8_t allow) -+{ -+ XSM_ASSERT_ACTION(XSM_HOOK); -+ return xsm_default_action(action, current->domain, d); -+} -+ - static XSM_INLINE int cf_check xsm_pci_config_permission( - XSM_DEFAULT_ARG struct domain *d, uint32_t machine_bdf, uint16_t start, - uint16_t end, uint8_t access) ---- a/xen/include/xsm/xsm.h -+++ b/xen/include/xsm/xsm.h -@@ -118,6 +118,8 @@ struct xsm_ops { - uint8_t allow); - int (*iomem_mapping)(struct domain *d, uint64_t s, uint64_t e, - uint8_t allow); -+ int (*iomem_mapping_vpci)(struct domain *d, uint64_t s, uint64_t e, -+ uint8_t allow); - int (*pci_config_permission)(struct domain *d, uint32_t machine_bdf, - uint16_t start, uint16_t end, uint8_t access); - -@@ -523,6 +525,12 @@ static inline int xsm_iomem_mapping( - return alternative_call(xsm_ops.iomem_mapping, d, s, e, allow); - } - -+static inline int xsm_iomem_mapping_vpci( -+ xsm_default_t def, struct domain *d, uint64_t s, uint64_t e, uint8_t allow) -+{ -+ return alternative_call(xsm_ops.iomem_mapping_vpci, d, s, e, allow); -+} -+ - static inline int xsm_pci_config_permission( - xsm_default_t def, struct domain *d, uint32_t machine_bdf, uint16_t start, - uint16_t end, uint8_t access) ---- a/xen/xsm/dummy.c -+++ b/xen/xsm/dummy.c -@@ -76,6 +76,7 @@ static const struct xsm_ops __initconst_ - .irq_permission = xsm_irq_permission, - .iomem_permission = xsm_iomem_permission, - .iomem_mapping = xsm_iomem_mapping, -+ .iomem_mapping_vpci = xsm_iomem_mapping_vpci, - .pci_config_permission = xsm_pci_config_permission, - .get_vnumainfo = xsm_get_vnumainfo, - ---- a/xen/xsm/flask/hooks.c -+++ b/xen/xsm/flask/hooks.c -@@ -1950,6 +1950,7 @@ static const struct xsm_ops __initconst_ - .irq_permission = flask_irq_permission, - .iomem_permission = flask_iomem_permission, - .iomem_mapping = flask_iomem_mapping, -+ .iomem_mapping_vpci = flask_iomem_mapping, - .pci_config_permission = flask_pci_config_permission, - - .resource_plug_core = flask_resource_plug_core, diff --git a/xsa492-4.21-09.patch b/xsa492-4.21-09.patch deleted file mode 100644 index 96e9403..0000000 --- a/xsa492-4.21-09.patch +++ /dev/null @@ -1,194 +0,0 @@ -From: Jan Beulich -Subject: domctl: handle XEN_DOMCTL_memory_mapping without acquiring domctl lock - -With dedicated locking added, the domctl lock isn't required here anymore. -Move the re-purposed dedicated XSM check as early as possible. - -Minimal "modernization": Switch "add" to bool and use %pd in log messages. - -This is part of XSA-492. - -Fixes: fda49f9b3fbb ("Add build option to allow more hypercalls from stubdoms") -Reported-by: Andrew Cooper -Signed-off-by: Jan Beulich -Acked-by: Daniel P. Smith -Reviewed-by: Roger Pau Monné - ---- a/xen/common/domctl.c -+++ b/xen/common/domctl.c -@@ -376,6 +376,66 @@ long do_domctl(XEN_GUEST_HANDLE_PARAM(xe - copyback = true; - goto domctl_out_unlock_domonly; - -+ case XEN_DOMCTL_memory_mapping: -+ { -+ unsigned long gfn = op->u.memory_mapping.first_gfn; -+ unsigned long mfn = op->u.memory_mapping.first_mfn; -+ unsigned long nr_mfns = op->u.memory_mapping.nr_mfns; -+ unsigned long mfn_end = mfn + nr_mfns - 1; -+ bool add = op->u.memory_mapping.add_mapping; -+ -+ ret = -EINVAL; -+ if ( mfn_end < mfn || /* Wrap? */ -+ ((mfn | mfn_end) >> (paddr_bits - PAGE_SHIFT)) || -+ (gfn + nr_mfns - 1) < gfn ) /* Wrap? */ -+ goto domctl_out_unlock_domonly; -+ -+ ret = xsm_iomem_mapping(XSM_DM_PRIV, d, mfn, mfn_end, add); -+ if ( ret || !paging_mode_translate(d) ) -+ goto domctl_out_unlock_domonly; -+ -+#ifndef CONFIG_X86 /* XXX ARM!? */ -+ ret = -E2BIG; -+ /* Must break hypercall up as this could take a while. */ -+ if ( nr_mfns > 64 ) -+ goto domctl_out_unlock_domonly; -+#endif -+ -+ iocaps_double_lock(d, false); -+ -+ ret = -EPERM; -+ if ( !iomem_access_permitted(current->domain, mfn, mfn_end) || -+ !iomem_access_permitted(d, mfn, mfn_end) ) -+ /* Nothing. */; -+ else if ( add ) -+ { -+ printk(XENLOG_G_DEBUG -+ "memory_map:add: %pd gfn=%lx mfn=%lx nr=%lx\n", -+ d, gfn, mfn, nr_mfns); -+ -+ ret = map_mmio_regions(d, _gfn(gfn), nr_mfns, _mfn(mfn)); -+ if ( ret < 0 ) -+ printk(XENLOG_G_WARNING -+ "memory_map:fail: %pd gfn=%lx mfn=%lx nr=%lx ret:%ld\n", -+ d, gfn, mfn, nr_mfns, ret); -+ } -+ else -+ { -+ printk(XENLOG_G_DEBUG -+ "memory_map:remove: %pd gfn=%lx mfn=%lx nr=%lx\n", -+ d, gfn, mfn, nr_mfns); -+ -+ ret = unmap_mmio_regions(d, _gfn(gfn), nr_mfns, _mfn(mfn)); -+ if ( ret < 0 && is_hardware_domain(current->domain) ) -+ printk(XENLOG_ERR -+ "memory_map: error %ld removing %pd access to [%lx,%lx]\n", -+ ret, d, mfn, mfn_end); -+ } -+ -+ iocaps_double_unlock(d, false); -+ goto domctl_out_unlock_domonly; -+ } -+ - default: - /* Everything else handled further down. */ - break; -@@ -736,64 +796,6 @@ long do_domctl(XEN_GUEST_HANDLE_PARAM(xe - break; - } - -- case XEN_DOMCTL_memory_mapping: -- { -- unsigned long gfn = op->u.memory_mapping.first_gfn; -- unsigned long mfn = op->u.memory_mapping.first_mfn; -- unsigned long nr_mfns = op->u.memory_mapping.nr_mfns; -- unsigned long mfn_end = mfn + nr_mfns - 1; -- int add = op->u.memory_mapping.add_mapping; -- -- ret = -EINVAL; -- if ( mfn_end < mfn || /* wrap? */ -- ((mfn | mfn_end) >> (paddr_bits - PAGE_SHIFT)) || -- (gfn + nr_mfns - 1) < gfn ) /* wrap? */ -- break; -- --#ifndef CONFIG_X86 /* XXX ARM!? */ -- ret = -E2BIG; -- /* Must break hypercall up as this could take a while. */ -- if ( nr_mfns > 64 ) -- break; --#endif -- -- iocaps_double_lock(d, false); -- -- ret = -EPERM; -- if ( !iomem_access_permitted(current->domain, mfn, mfn_end) || -- !iomem_access_permitted(d, mfn, mfn_end) || -- (ret = xsm_iomem_mapping(XSM_HOOK, d, mfn, mfn_end, add)) || -- !paging_mode_translate(d) ) -- /* Nothing. */; -- else if ( add ) -- { -- printk(XENLOG_G_DEBUG -- "memory_map:add: dom%d gfn=%lx mfn=%lx nr=%lx\n", -- d->domain_id, gfn, mfn, nr_mfns); -- -- ret = map_mmio_regions(d, _gfn(gfn), nr_mfns, _mfn(mfn)); -- if ( ret < 0 ) -- printk(XENLOG_G_WARNING -- "memory_map:fail: dom%d gfn=%lx mfn=%lx nr=%lx ret:%ld\n", -- d->domain_id, gfn, mfn, nr_mfns, ret); -- } -- else -- { -- printk(XENLOG_G_DEBUG -- "memory_map:remove: dom%d gfn=%lx mfn=%lx nr=%lx\n", -- d->domain_id, gfn, mfn, nr_mfns); -- -- ret = unmap_mmio_regions(d, _gfn(gfn), nr_mfns, _mfn(mfn)); -- if ( ret < 0 && is_hardware_domain(current->domain) ) -- printk(XENLOG_ERR -- "memory_map: error %ld removing dom%d access to [%lx,%lx]\n", -- ret, d->domain_id, mfn, mfn_end); -- } -- -- iocaps_double_unlock(d, false); -- break; -- } -- - case XEN_DOMCTL_settimeoffset: - domain_set_time_offset(d, op->u.settimeoffset.time_offset_seconds); - break; ---- a/xen/include/xsm/dummy.h -+++ b/xen/include/xsm/dummy.h -@@ -168,13 +168,13 @@ static XSM_INLINE int cf_check xsm_domct - switch ( cmd ) - { - case XEN_DOMCTL_ioport_mapping: -- case XEN_DOMCTL_memory_mapping: - case XEN_DOMCTL_bind_pt_irq: - case XEN_DOMCTL_unbind_pt_irq: - return xsm_default_action(XSM_DM_PRIV, current->domain, d); - - case XEN_DOMCTL_getdomaininfo: - case XEN_DOMCTL_get_domain_state: -+ case XEN_DOMCTL_memory_mapping: - ASSERT_UNREACHABLE(); - return -EILSEQ; - -@@ -576,7 +576,7 @@ static XSM_INLINE int cf_check xsm_iomem - static XSM_INLINE int cf_check xsm_iomem_mapping( - XSM_DEFAULT_ARG struct domain *d, uint64_t s, uint64_t e, uint8_t allow) - { -- XSM_ASSERT_ACTION(XSM_HOOK); -+ XSM_ASSERT_ACTION(XSM_DM_PRIV); - return xsm_default_action(action, current->domain, d); - } - ---- a/xen/xsm/flask/hooks.c -+++ b/xen/xsm/flask/hooks.c -@@ -685,6 +685,7 @@ static int cf_check flask_domctl(struct - /* These have individual XSM hooks and don't make it here. */ - case XEN_DOMCTL_getdomaininfo: - case XEN_DOMCTL_get_domain_state: -+ case XEN_DOMCTL_memory_mapping: - ASSERT_UNREACHABLE(); - return -EILSEQ; - -@@ -692,7 +693,6 @@ static int cf_check flask_domctl(struct - case XEN_DOMCTL_scheduler_op: - case XEN_DOMCTL_irq_permission: - case XEN_DOMCTL_iomem_permission: -- case XEN_DOMCTL_memory_mapping: - case XEN_DOMCTL_set_target: - case XEN_DOMCTL_vm_event_op: - diff --git a/xsa492-4.21-10.patch b/xsa492-4.21-10.patch deleted file mode 100644 index 6406a19..0000000 --- a/xsa492-4.21-10.patch +++ /dev/null @@ -1,97 +0,0 @@ -From: Jan Beulich -Subject: domctl: handle XEN_DOMCTL_ioport_mapping without acquiring domctl lock - -With dedicated locking added, the domctl lock isn't required here anymore. -As the handling is in arch-specific code (x86 only), almost no code is -being moved, but a 2nd (extensible to other sub-ops) invocation of -arch_do_domctl() is being added. Move just the re-purposed dedicated XSM -check as early as possible. - -In flask_domctl() don't put #ifdef around the moved case label. - -This is part of XSA-492. - -Fixes: fda49f9b3fbb ("Add build option to allow more hypercalls from stubdoms") -Reported-by: Andrew Cooper -Signed-off-by: Jan Beulich -Reviewed-by: Roger Pau Monné -Acked-by: Daniel P. Smith - ---- a/xen/arch/x86/domctl.c -+++ b/xen/arch/x86/domctl.c -@@ -663,12 +663,15 @@ long arch_do_domctl( - break; - } - -+ ret = xsm_ioport_mapping(XSM_DM_PRIV, d, fmp, fmp + np - 1, add); -+ if ( ret ) -+ break; -+ - hvm = &d->arch.hvm; - iocaps_double_lock(d, true); - -- if ( !ioports_access_permitted(currd, fmp, fmp + np - 1) || -- (ret = xsm_ioport_mapping(XSM_HOOK, d, fmp, fmp + np - 1, add)) ) -- ret = ret ?: -EPERM; -+ if ( !ioports_access_permitted(currd, fmp, fmp + np - 1) ) -+ ret = -EPERM; - else if ( add ) - { - printk(XENLOG_G_INFO ---- a/xen/common/domctl.c -+++ b/xen/common/domctl.c -@@ -436,6 +436,10 @@ long do_domctl(XEN_GUEST_HANDLE_PARAM(xe - goto domctl_out_unlock_domonly; - } - -+ case XEN_DOMCTL_ioport_mapping: -+ ret = arch_do_domctl(op, d, u_domctl); -+ goto domctl_out_unlock_domonly; -+ - default: - /* Everything else handled further down. */ - break; ---- a/xen/include/xsm/dummy.h -+++ b/xen/include/xsm/dummy.h -@@ -167,13 +167,13 @@ static XSM_INLINE int cf_check xsm_domct - XSM_ASSERT_ACTION(XSM_OTHER); - switch ( cmd ) - { -- case XEN_DOMCTL_ioport_mapping: - case XEN_DOMCTL_bind_pt_irq: - case XEN_DOMCTL_unbind_pt_irq: - return xsm_default_action(XSM_DM_PRIV, current->domain, d); - - case XEN_DOMCTL_getdomaininfo: - case XEN_DOMCTL_get_domain_state: -+ case XEN_DOMCTL_ioport_mapping: - case XEN_DOMCTL_memory_mapping: - ASSERT_UNREACHABLE(); - return -EILSEQ; -@@ -772,7 +772,7 @@ static XSM_INLINE int cf_check xsm_iopor - static XSM_INLINE int cf_check xsm_ioport_mapping( - XSM_DEFAULT_ARG struct domain *d, uint32_t s, uint32_t e, uint8_t allow) - { -- XSM_ASSERT_ACTION(XSM_HOOK); -+ XSM_ASSERT_ACTION(XSM_DM_PRIV); - return xsm_default_action(action, current->domain, d); - } - ---- a/xen/xsm/flask/hooks.c -+++ b/xen/xsm/flask/hooks.c -@@ -685,6 +685,7 @@ static int cf_check flask_domctl(struct - /* These have individual XSM hooks and don't make it here. */ - case XEN_DOMCTL_getdomaininfo: - case XEN_DOMCTL_get_domain_state: -+ case XEN_DOMCTL_ioport_mapping: - case XEN_DOMCTL_memory_mapping: - ASSERT_UNREACHABLE(); - return -EILSEQ; -@@ -703,7 +704,6 @@ static int cf_check flask_domctl(struct - /* These have individual XSM hooks (arch/x86/domctl.c) */ - case XEN_DOMCTL_shadow_op: - case XEN_DOMCTL_ioport_permission: -- case XEN_DOMCTL_ioport_mapping: - case XEN_DOMCTL_gsi_permission: - #endif - #ifdef CONFIG_HAS_PASSTHROUGH diff --git a/xsa492-4.21-11.patch b/xsa492-4.21-11.patch deleted file mode 100644 index 647fd5a..0000000 --- a/xsa492-4.21-11.patch +++ /dev/null @@ -1,128 +0,0 @@ -From: Jan Beulich -Subject: domctl: handle XEN_DOMCTL_{,un}bind_pt_irq without acquiring domctl lock - -With dedicated locking added, the domctl lock isn't required here anymore. -(It also already isn't used when pt_irq_{create,destroy}_bind() are -invoked for PVH Dom0.) As the handling is in arch-specific code, no code -is being moved, but the 2nd (extensible to other sub-ops like the ones -here) invocation of arch_do_domctl() is being re-used. - -This is part of XSA-492. - -Fixes: fda49f9b3fbb ("Add build option to allow more hypercalls from stubdoms") -Reported-by: Andrew Cooper -Signed-off-by: Jan Beulich -Reviewed-by: Roger Pau Monné -Acked-by: Daniel P. Smith -Acked-by: Julien Grall - ---- a/xen/arch/arm/domctl.c -+++ b/xen/arch/arm/domctl.c -@@ -104,7 +104,7 @@ long arch_do_domctl(struct xen_domctl *d - if ( rc ) - return rc; - -- rc = xsm_bind_pt_irq(XSM_HOOK, d, bind); -+ rc = xsm_bind_pt_irq(XSM_DM_PRIV, d, bind); - if ( rc ) - return rc; - -@@ -140,7 +140,7 @@ long arch_do_domctl(struct xen_domctl *d - if ( irq != virq ) - return -EINVAL; - -- rc = xsm_unbind_pt_irq(XSM_HOOK, d, bind); -+ rc = xsm_unbind_pt_irq(XSM_DM_PRIV, d, bind); - if ( rc ) - return rc; - ---- a/xen/arch/x86/domctl.c -+++ b/xen/arch/x86/domctl.c -@@ -575,7 +575,7 @@ long arch_do_domctl( - if ( !is_hvm_domain(d) ) - break; - -- ret = xsm_bind_pt_irq(XSM_HOOK, d, bind); -+ ret = xsm_bind_pt_irq(XSM_DM_PRIV, d, bind); - if ( ret ) - break; - -@@ -613,7 +613,7 @@ long arch_do_domctl( - if ( !is_hvm_domain(d) ) - break; - -- ret = xsm_unbind_pt_irq(XSM_HOOK, d, bind); -+ ret = xsm_unbind_pt_irq(XSM_DM_PRIV, d, bind); - if ( ret ) - break; - ---- a/xen/common/domctl.c -+++ b/xen/common/domctl.c -@@ -437,6 +437,8 @@ long do_domctl(XEN_GUEST_HANDLE_PARAM(xe - } - - case XEN_DOMCTL_ioport_mapping: -+ case XEN_DOMCTL_bind_pt_irq: -+ case XEN_DOMCTL_unbind_pt_irq: - ret = arch_do_domctl(op, d, u_domctl); - goto domctl_out_unlock_domonly; - ---- a/xen/include/xsm/dummy.h -+++ b/xen/include/xsm/dummy.h -@@ -168,13 +168,11 @@ static XSM_INLINE int cf_check xsm_domct - switch ( cmd ) - { - case XEN_DOMCTL_bind_pt_irq: -- case XEN_DOMCTL_unbind_pt_irq: -- return xsm_default_action(XSM_DM_PRIV, current->domain, d); -- - case XEN_DOMCTL_getdomaininfo: - case XEN_DOMCTL_get_domain_state: - case XEN_DOMCTL_ioport_mapping: - case XEN_DOMCTL_memory_mapping: -+ case XEN_DOMCTL_unbind_pt_irq: - ASSERT_UNREACHABLE(); - return -EILSEQ; - -@@ -541,14 +539,14 @@ static XSM_INLINE int cf_check xsm_unmap - static XSM_INLINE int cf_check xsm_bind_pt_irq( - XSM_DEFAULT_ARG struct domain *d, struct xen_domctl_bind_pt_irq *bind) - { -- XSM_ASSERT_ACTION(XSM_HOOK); -+ XSM_ASSERT_ACTION(XSM_DM_PRIV); - return xsm_default_action(action, current->domain, d); - } - - static XSM_INLINE int cf_check xsm_unbind_pt_irq( - XSM_DEFAULT_ARG struct domain *d, struct xen_domctl_bind_pt_irq *bind) - { -- XSM_ASSERT_ACTION(XSM_HOOK); -+ XSM_ASSERT_ACTION(XSM_DM_PRIV); - return xsm_default_action(action, current->domain, d); - } - ---- a/xen/xsm/flask/hooks.c -+++ b/xen/xsm/flask/hooks.c -@@ -683,10 +683,12 @@ static int cf_check flask_domctl(struct - return avc_current_has_perm(ssidref, SECCLASS_DOMAIN, DOMAIN__CREATE, NULL); - - /* These have individual XSM hooks and don't make it here. */ -+ case XEN_DOMCTL_bind_pt_irq: - case XEN_DOMCTL_getdomaininfo: - case XEN_DOMCTL_get_domain_state: - case XEN_DOMCTL_ioport_mapping: - case XEN_DOMCTL_memory_mapping: -+ case XEN_DOMCTL_unbind_pt_irq: - ASSERT_UNREACHABLE(); - return -EILSEQ; - -@@ -697,9 +699,6 @@ static int cf_check flask_domctl(struct - case XEN_DOMCTL_set_target: - case XEN_DOMCTL_vm_event_op: - -- /* These have individual XSM hooks (arch/../domctl.c) */ -- case XEN_DOMCTL_bind_pt_irq: -- case XEN_DOMCTL_unbind_pt_irq: - #ifdef CONFIG_X86 - /* These have individual XSM hooks (arch/x86/domctl.c) */ - case XEN_DOMCTL_shadow_op: diff --git a/xsa492-4.21-12.patch b/xsa492-4.21-12.patch deleted file mode 100644 index c19d1e1..0000000 --- a/xsa492-4.21-12.patch +++ /dev/null @@ -1,172 +0,0 @@ -From: Jan Beulich -Subject: domctl: handle XEN_DOMCTL_io{mem,port}_permission without acquiring domctl lock - -With dedicated locking added, the domctl lock isn't required here anymore. -As the I/O port handling is in arch-specific code (x86 only), no code is -being moved, but the 2nd invocation of arch_do_domctl() is re-used. Move -the re-purposed dedicated XSM checks as early as possible. - -This is part of XSA-492. - -Signed-off-by: Jan Beulich -Reviewed-by: Roger Pau Monné -Acked-by: Daniel P. Smith - ---- a/xen/arch/x86/domctl.c -+++ b/xen/arch/x86/domctl.c -@@ -233,12 +233,17 @@ long arch_do_domctl( - unsigned int np = domctl->u.ioport_permission.nr_ports; - int allow = domctl->u.ioport_permission.allow_access; - -+ ret = -EINVAL; -+ if ( (fp + np) <= fp || (fp + np) > MAX_IOPORTS ) -+ break; -+ -+ ret = xsm_ioport_permission(XSM_PRIV, d, fp, fp + np - 1, allow); -+ if ( ret ) -+ break; -+ - iocaps_double_lock(d, true); - -- if ( (fp + np) <= fp || (fp + np) > MAX_IOPORTS ) -- ret = -EINVAL; -- else if ( !ioports_access_permitted(currd, fp, fp + np - 1) || -- xsm_ioport_permission(XSM_HOOK, d, fp, fp + np - 1, allow) ) -+ if ( !ioports_access_permitted(currd, fp, fp + np - 1) ) - ret = -EPERM; - else if ( allow ) - ret = ioports_permit_access(d, fp, fp + np - 1); ---- a/xen/common/domctl.c -+++ b/xen/common/domctl.c -@@ -376,6 +376,34 @@ long do_domctl(XEN_GUEST_HANDLE_PARAM(xe - copyback = true; - goto domctl_out_unlock_domonly; - -+ case XEN_DOMCTL_iomem_permission: -+ { -+ unsigned long mfn = op->u.iomem_permission.first_mfn; -+ unsigned long nr_mfns = op->u.iomem_permission.nr_mfns; -+ bool allow = op->u.iomem_permission.allow_access; -+ -+ ret = -EINVAL; -+ if ( (mfn + nr_mfns - 1) < mfn ) /* Wrap? */ -+ goto domctl_out_unlock_domonly; -+ -+ ret = xsm_iomem_permission(XSM_PRIV, d, mfn, mfn + nr_mfns - 1, allow); -+ if ( ret ) -+ goto domctl_out_unlock_domonly; -+ -+ iocaps_double_lock(d, true); -+ -+ if ( !iomem_access_permitted(current->domain, -+ mfn, mfn + nr_mfns - 1) ) -+ ret = -EPERM; -+ else if ( allow ) -+ ret = iomem_permit_access(d, mfn, mfn + nr_mfns - 1); -+ else -+ ret = iomem_deny_access(d, mfn, mfn + nr_mfns - 1); -+ -+ iocaps_double_unlock(d, true); -+ goto domctl_out_unlock_domonly; -+ } -+ - case XEN_DOMCTL_memory_mapping: - { - unsigned long gfn = op->u.memory_mapping.first_gfn; -@@ -436,6 +464,7 @@ long do_domctl(XEN_GUEST_HANDLE_PARAM(xe - goto domctl_out_unlock_domonly; - } - -+ case XEN_DOMCTL_ioport_permission: - case XEN_DOMCTL_ioport_mapping: - case XEN_DOMCTL_bind_pt_irq: - case XEN_DOMCTL_unbind_pt_irq: -@@ -777,31 +806,6 @@ long do_domctl(XEN_GUEST_HANDLE_PARAM(xe - } - #endif - -- case XEN_DOMCTL_iomem_permission: -- { -- unsigned long mfn = op->u.iomem_permission.first_mfn; -- unsigned long nr_mfns = op->u.iomem_permission.nr_mfns; -- int allow = op->u.iomem_permission.allow_access; -- -- ret = -EINVAL; -- if ( (mfn + nr_mfns - 1) < mfn ) /* wrap? */ -- break; -- -- iocaps_double_lock(d, true); -- -- if ( !iomem_access_permitted(current->domain, -- mfn, mfn + nr_mfns - 1) || -- xsm_iomem_permission(XSM_HOOK, d, mfn, mfn + nr_mfns - 1, allow) ) -- ret = -EPERM; -- else if ( allow ) -- ret = iomem_permit_access(d, mfn, mfn + nr_mfns - 1); -- else -- ret = iomem_deny_access(d, mfn, mfn + nr_mfns - 1); -- -- iocaps_double_unlock(d, true); -- break; -- } -- - case XEN_DOMCTL_settimeoffset: - domain_set_time_offset(d, op->u.settimeoffset.time_offset_seconds); - break; ---- a/xen/include/xsm/dummy.h -+++ b/xen/include/xsm/dummy.h -@@ -170,7 +170,9 @@ static XSM_INLINE int cf_check xsm_domct - case XEN_DOMCTL_bind_pt_irq: - case XEN_DOMCTL_getdomaininfo: - case XEN_DOMCTL_get_domain_state: -+ case XEN_DOMCTL_iomem_permission: - case XEN_DOMCTL_ioport_mapping: -+ case XEN_DOMCTL_ioport_permission: - case XEN_DOMCTL_memory_mapping: - case XEN_DOMCTL_unbind_pt_irq: - ASSERT_UNREACHABLE(); -@@ -567,7 +569,7 @@ static XSM_INLINE int cf_check xsm_irq_p - static XSM_INLINE int cf_check xsm_iomem_permission( - XSM_DEFAULT_ARG struct domain *d, uint64_t s, uint64_t e, uint8_t allow) - { -- XSM_ASSERT_ACTION(XSM_HOOK); -+ XSM_ASSERT_ACTION(XSM_PRIV); - return xsm_default_action(action, current->domain, d); - } - -@@ -763,7 +765,7 @@ static XSM_INLINE int cf_check xsm_priv_ - static XSM_INLINE int cf_check xsm_ioport_permission( - XSM_DEFAULT_ARG struct domain *d, uint32_t s, uint32_t e, uint8_t allow) - { -- XSM_ASSERT_ACTION(XSM_HOOK); -+ XSM_ASSERT_ACTION(XSM_PRIV); - return xsm_default_action(action, current->domain, d); - } - ---- a/xen/xsm/flask/hooks.c -+++ b/xen/xsm/flask/hooks.c -@@ -686,7 +686,9 @@ static int cf_check flask_domctl(struct - case XEN_DOMCTL_bind_pt_irq: - case XEN_DOMCTL_getdomaininfo: - case XEN_DOMCTL_get_domain_state: -+ case XEN_DOMCTL_iomem_permission: - case XEN_DOMCTL_ioport_mapping: -+ case XEN_DOMCTL_ioport_permission: - case XEN_DOMCTL_memory_mapping: - case XEN_DOMCTL_unbind_pt_irq: - ASSERT_UNREACHABLE(); -@@ -695,14 +697,12 @@ static int cf_check flask_domctl(struct - /* These have individual XSM hooks (common/domctl.c) */ - case XEN_DOMCTL_scheduler_op: - case XEN_DOMCTL_irq_permission: -- case XEN_DOMCTL_iomem_permission: - case XEN_DOMCTL_set_target: - case XEN_DOMCTL_vm_event_op: - - #ifdef CONFIG_X86 - /* These have individual XSM hooks (arch/x86/domctl.c) */ - case XEN_DOMCTL_shadow_op: -- case XEN_DOMCTL_ioport_permission: - case XEN_DOMCTL_gsi_permission: - #endif - #ifdef CONFIG_HAS_PASSTHROUGH diff --git a/xsa492-4.21-13.patch b/xsa492-4.21-13.patch deleted file mode 100644 index 91ce1ae..0000000 --- a/xsa492-4.21-13.patch +++ /dev/null @@ -1,163 +0,0 @@ -From: Jan Beulich -Subject: domctl: handle XEN_DOMCTL_{irq,gsi}_permission without acquiring domctl lock - -With dedicated locking added, the domctl lock isn't required here anymore. -As the GSI handling is in arch-specific code (x86 only), no code is being -moved there; the 2nd invocation of arch_do_domctl() is re-used. Move the -re-purposed (XSM_HOOK -> XSM_PRIV, as xsm_domctl() is now bypassed) -dedicated XSM checks as early as possible. - -This is part of XSA-492. - -Signed-off-by: Jan Beulich -Acked-by: Daniel P. Smith -Reviewed-by: Roger Pau Monné - ---- a/xen/arch/x86/domctl.c -+++ b/xen/arch/x86/domctl.c -@@ -272,10 +272,13 @@ long arch_do_domctl( - break; - } - -+ ret = xsm_irq_permission(XSM_PRIV, d, irq, flags); -+ if ( ret ) -+ break; -+ - iocaps_double_lock(d, true); - -- if ( !irq_access_permitted(currd, irq) || -- xsm_irq_permission(XSM_HOOK, d, irq, flags) ) -+ if ( !irq_access_permitted(currd, irq) ) - ret = -EPERM; - else if ( flags ) - ret = irq_permit_access(d, irq); ---- a/xen/common/domctl.c -+++ b/xen/common/domctl.c -@@ -464,8 +464,41 @@ long do_domctl(XEN_GUEST_HANDLE_PARAM(xe - goto domctl_out_unlock_domonly; - } - -+#ifdef CONFIG_HAS_PIRQ -+ case XEN_DOMCTL_irq_permission: -+ { -+ unsigned int pirq = op->u.irq_permission.pirq, irq; -+ bool allow = op->u.irq_permission.allow_access; -+ -+ ret = -EINVAL; -+ if ( pirq >= current->domain->nr_pirqs ) -+ goto domctl_out_unlock_domonly; -+ -+ irq = domain_pirq_to_irq(current->domain, pirq); -+ -+ ret = -EPERM; -+ if ( irq ) -+ ret = xsm_irq_permission(XSM_PRIV, d, irq, allow); -+ if ( ret ) -+ goto domctl_out_unlock_domonly; -+ -+ iocaps_double_lock(d, true); -+ -+ if ( !irq_access_permitted(current->domain, irq) ) -+ ret = -EPERM; -+ else if ( allow ) -+ ret = irq_permit_access(d, irq); -+ else -+ ret = irq_deny_access(d, irq); -+ -+ iocaps_double_unlock(d, true); -+ goto domctl_out_unlock_domonly; -+ } -+#endif -+ - case XEN_DOMCTL_ioport_permission: - case XEN_DOMCTL_ioport_mapping: -+ case XEN_DOMCTL_gsi_permission: - case XEN_DOMCTL_bind_pt_irq: - case XEN_DOMCTL_unbind_pt_irq: - ret = arch_do_domctl(op, d, u_domctl); -@@ -779,33 +812,6 @@ long do_domctl(XEN_GUEST_HANDLE_PARAM(xe - } - break; - --#ifdef CONFIG_HAS_PIRQ -- case XEN_DOMCTL_irq_permission: -- { -- unsigned int pirq = op->u.irq_permission.pirq, irq; -- int allow = op->u.irq_permission.allow_access; -- -- if ( pirq >= current->domain->nr_pirqs ) -- { -- ret = -EINVAL; -- break; -- } -- -- iocaps_double_lock(d, true); -- -- irq = pirq_access_permitted(current->domain, pirq); -- if ( !irq || xsm_irq_permission(XSM_HOOK, d, irq, allow) ) -- ret = -EPERM; -- else if ( allow ) -- ret = irq_permit_access(d, irq); -- else -- ret = irq_deny_access(d, irq); -- -- iocaps_double_unlock(d, true); -- break; -- } --#endif -- - case XEN_DOMCTL_settimeoffset: - domain_set_time_offset(d, op->u.settimeoffset.time_offset_seconds); - break; ---- a/xen/include/xsm/dummy.h -+++ b/xen/include/xsm/dummy.h -@@ -170,9 +170,11 @@ static XSM_INLINE int cf_check xsm_domct - case XEN_DOMCTL_bind_pt_irq: - case XEN_DOMCTL_getdomaininfo: - case XEN_DOMCTL_get_domain_state: -+ case XEN_DOMCTL_gsi_permission: - case XEN_DOMCTL_iomem_permission: - case XEN_DOMCTL_ioport_mapping: - case XEN_DOMCTL_ioport_permission: -+ case XEN_DOMCTL_irq_permission: - case XEN_DOMCTL_memory_mapping: - case XEN_DOMCTL_unbind_pt_irq: - ASSERT_UNREACHABLE(); -@@ -562,7 +564,7 @@ static XSM_INLINE int cf_check xsm_unmap - static XSM_INLINE int cf_check xsm_irq_permission( - XSM_DEFAULT_ARG struct domain *d, int pirq, uint8_t allow) - { -- XSM_ASSERT_ACTION(XSM_HOOK); -+ XSM_ASSERT_ACTION(XSM_PRIV); - return xsm_default_action(action, current->domain, d); - } - ---- a/xen/xsm/flask/hooks.c -+++ b/xen/xsm/flask/hooks.c -@@ -686,9 +686,11 @@ static int cf_check flask_domctl(struct - case XEN_DOMCTL_bind_pt_irq: - case XEN_DOMCTL_getdomaininfo: - case XEN_DOMCTL_get_domain_state: -+ case XEN_DOMCTL_gsi_permission: - case XEN_DOMCTL_iomem_permission: - case XEN_DOMCTL_ioport_mapping: - case XEN_DOMCTL_ioport_permission: -+ case XEN_DOMCTL_irq_permission: - case XEN_DOMCTL_memory_mapping: - case XEN_DOMCTL_unbind_pt_irq: - ASSERT_UNREACHABLE(); -@@ -696,14 +698,12 @@ static int cf_check flask_domctl(struct - - /* These have individual XSM hooks (common/domctl.c) */ - case XEN_DOMCTL_scheduler_op: -- case XEN_DOMCTL_irq_permission: - case XEN_DOMCTL_set_target: - case XEN_DOMCTL_vm_event_op: - - #ifdef CONFIG_X86 - /* These have individual XSM hooks (arch/x86/domctl.c) */ - case XEN_DOMCTL_shadow_op: -- case XEN_DOMCTL_gsi_permission: - #endif - #ifdef CONFIG_HAS_PASSTHROUGH - /* diff --git a/xsa492-4.21-14.patch b/xsa492-4.21-14.patch deleted file mode 100644 index 2b13377..0000000 --- a/xsa492-4.21-14.patch +++ /dev/null @@ -1,179 +0,0 @@ -From: Jan Beulich -Subject: domctl/XSM: drop vm_event_control hook - -Integrate the checking with xsm_domctl(). Care needs to be taken with the -GET_VERSION sub-op, which may be invoked with DOMID_INVALID, and which has -been (and continues to be) bypassing XSM checking. - -Since the latter two parameters were unused, monitor_domctl() invoking the -hook was actually redundant with the earlier xsm_domctl() (as can be seen -nicely from the hunks changing xsm/flask/hooks.c). - -As a positive side effect, permissions are then checked at the same early -point with and without Flask. - -While folding XEN_DOMCTL_monitor_op and XEN_DOMCTL_vm_event_op in -flask_domctl(), also fold in XEN_DOMCTL_set_access_required. - -This is part of XSA-492. - -Signed-off-by: Jan Beulich -Acked-by: Daniel P. Smith -Reviewed-by: Roger Pau Monné - ---- a/xen/common/domctl.c -+++ b/xen/common/domctl.c -@@ -496,6 +496,23 @@ long do_domctl(XEN_GUEST_HANDLE_PARAM(xe - } - #endif - -+ case XEN_DOMCTL_vm_event_op: -+ if ( op->u.vm_event_op.op == XEN_VM_EVENT_GET_VERSION ) -+ { -+ /* No XSM check (and potentially d == NULL) here. */ -+ ret = vm_event_domctl(d, &op->u.vm_event_op); -+ if ( !ret ) -+ copyback = true; -+ goto domctl_out_unlock_domonly; -+ } -+ if ( !d ) -+ { -+ ret = -ESRCH; -+ goto domctl_out_unlock_domonly; -+ } -+ /* Other sub-ops handled further down. */ -+ break; -+ - case XEN_DOMCTL_ioport_permission: - case XEN_DOMCTL_ioport_mapping: - case XEN_DOMCTL_gsi_permission: ---- a/xen/common/monitor.c -+++ b/xen/common/monitor.c -@@ -30,16 +30,11 @@ - - int monitor_domctl(struct domain *d, struct xen_domctl_monitor_op *mop) - { -- int rc; - bool requested_status = false; - - if ( unlikely(current->domain == d) ) /* no domain_pause() */ - return -EPERM; - -- rc = xsm_vm_event_control(XSM_PRIV, d, mop->op, mop->event); -- if ( unlikely(rc) ) -- return rc; -- - switch ( mop->op ) - { - case XEN_DOMCTL_MONITOR_OP_ENABLE: ---- a/xen/common/vm_event.c -+++ b/xen/common/vm_event.c -@@ -603,11 +603,10 @@ int vm_event_domctl(struct domain *d, st - - /* All other subops need to target a real domain. */ - if ( unlikely(d == NULL) ) -- return -ESRCH; -- -- rc = xsm_vm_event_control(XSM_PRIV, d, vec->mode, vec->op); -- if ( rc ) -- return rc; -+ { -+ ASSERT_UNREACHABLE(); -+ return -EILSEQ; -+ } - - if ( unlikely(d == current->domain) ) /* no domain_pause() */ - { ---- a/xen/include/xsm/dummy.h -+++ b/xen/include/xsm/dummy.h -@@ -652,13 +652,6 @@ static XSM_INLINE int cf_check xsm_hvm_a - } - } - --static XSM_INLINE int cf_check xsm_vm_event_control( -- XSM_DEFAULT_ARG struct domain *d, int mode, int op) --{ -- XSM_ASSERT_ACTION(XSM_PRIV); -- return xsm_default_action(action, current->domain, d); --} -- - #ifdef CONFIG_VM_EVENT - static XSM_INLINE int cf_check xsm_mem_access(XSM_DEFAULT_ARG struct domain *d) - { ---- a/xen/include/xsm/xsm.h -+++ b/xen/include/xsm/xsm.h -@@ -157,8 +157,6 @@ struct xsm_ops { - int (*hvm_altp2mhvm_op)(struct domain *d, uint64_t mode, uint32_t op); - int (*get_vnumainfo)(struct domain *d); - -- int (*vm_event_control)(struct domain *d, int mode, int op); -- - #ifdef CONFIG_VM_EVENT - int (*mem_access)(struct domain *d); - #endif -@@ -657,12 +655,6 @@ static inline int xsm_get_vnumainfo(xsm_ - return alternative_call(xsm_ops.get_vnumainfo, d); - } - --static inline int xsm_vm_event_control( -- xsm_default_t def, struct domain *d, int mode, int op) --{ -- return alternative_call(xsm_ops.vm_event_control, d, mode, op); --} -- - #ifdef CONFIG_VM_EVENT - static inline int xsm_mem_access(xsm_default_t def, struct domain *d) - { ---- a/xen/xsm/dummy.c -+++ b/xen/xsm/dummy.c -@@ -116,8 +116,6 @@ static const struct xsm_ops __initconst_ - .remove_from_physmap = xsm_remove_from_physmap, - .map_gmfn_foreign = xsm_map_gmfn_foreign, - -- .vm_event_control = xsm_vm_event_control, -- - #ifdef CONFIG_VM_EVENT - .mem_access = xsm_mem_access, - #endif ---- a/xen/xsm/flask/hooks.c -+++ b/xen/xsm/flask/hooks.c -@@ -699,7 +699,6 @@ static int cf_check flask_domctl(struct - /* These have individual XSM hooks (common/domctl.c) */ - case XEN_DOMCTL_scheduler_op: - case XEN_DOMCTL_set_target: -- case XEN_DOMCTL_vm_event_op: - - #ifdef CONFIG_X86 - /* These have individual XSM hooks (arch/x86/domctl.c) */ -@@ -793,9 +792,8 @@ static int cf_check flask_domctl(struct - return current_has_perm(d, SECCLASS_DOMAIN, DOMAIN__TRIGGER); - - case XEN_DOMCTL_set_access_required: -- return current_has_perm(d, SECCLASS_DOMAIN2, DOMAIN2__VM_EVENT); -- - case XEN_DOMCTL_monitor_op: -+ case XEN_DOMCTL_vm_event_op: - return current_has_perm(d, SECCLASS_DOMAIN2, DOMAIN2__VM_EVENT); - - case XEN_DOMCTL_debug_op: -@@ -1368,11 +1366,6 @@ static int cf_check flask_hvm_altp2mhvm_ - return current_has_perm(d, SECCLASS_HVM, HVM__ALTP2MHVM_OP); - } - --static int cf_check flask_vm_event_control(struct domain *d, int mode, int op) --{ -- return current_has_perm(d, SECCLASS_DOMAIN2, DOMAIN2__VM_EVENT); --} -- - #ifdef CONFIG_VM_EVENT - static int cf_check flask_mem_access(struct domain *d) - { -@@ -1971,8 +1964,6 @@ static const struct xsm_ops __initconst_ - .do_xsm_op = do_flask_op, - .get_vnumainfo = flask_get_vnumainfo, - -- .vm_event_control = flask_vm_event_control, -- - #ifdef CONFIG_VM_EVENT - .mem_access = flask_mem_access, - #endif diff --git a/xsa492-4.21-15.patch b/xsa492-4.21-15.patch deleted file mode 100644 index ac87f3b..0000000 --- a/xsa492-4.21-15.patch +++ /dev/null @@ -1,108 +0,0 @@ -From: Jan Beulich -Subject: domctl/XSM: pass full struct xen_domctl to xsm_domctl() - -Subsequently some sub-ops will want to inspect their sub-sub-ops. Plus -this way we don't need to pass SSIDref separately anymore for -domain_create. - -This is part of XSA-492. - -Signed-off-by: Jan Beulich -Acked-by: Daniel P. Smith - ---- a/xen/arch/x86/mm/paging.c -+++ b/xen/arch/x86/mm/paging.c -@@ -735,7 +735,7 @@ long do_paging_domctl_cont( - if ( d == NULL ) - return -ESRCH; - -- ret = xsm_domctl(XSM_OTHER, d, op.cmd, 0 /* SSIDref not applicable */); -+ ret = xsm_domctl(XSM_OTHER, d, &op); - if ( !ret ) - { - if ( domctl_lock_acquire() ) ---- a/xen/common/domctl.c -+++ b/xen/common/domctl.c -@@ -526,9 +526,7 @@ long do_domctl(XEN_GUEST_HANDLE_PARAM(xe - break; - } - -- ret = xsm_domctl(XSM_OTHER, d, op->cmd, -- /* SSIDRef only applicable for cmd == createdomain */ -- op->u.createdomain.ssidref); -+ ret = xsm_domctl(XSM_OTHER, d, op); - if ( ret ) - goto domctl_out_unlock_domonly; - ---- a/xen/include/xsm/dummy.h -+++ b/xen/include/xsm/dummy.h -@@ -162,10 +162,10 @@ static XSM_INLINE int cf_check xsm_set_t - } - - static XSM_INLINE int cf_check xsm_domctl( -- XSM_DEFAULT_ARG struct domain *d, unsigned int cmd, uint32_t ssidref) -+ XSM_DEFAULT_ARG struct domain *d, struct xen_domctl *op) - { - XSM_ASSERT_ACTION(XSM_OTHER); -- switch ( cmd ) -+ switch ( op->cmd ) - { - case XEN_DOMCTL_bind_pt_irq: - case XEN_DOMCTL_getdomaininfo: ---- a/xen/include/xsm/xsm.h -+++ b/xen/include/xsm/xsm.h -@@ -61,7 +61,7 @@ struct xsm_ops { - int (*sysctl_scheduler_op)(int op); - #endif - int (*set_target)(struct domain *d, struct domain *e); -- int (*domctl)(struct domain *d, unsigned int cmd, uint32_t ssidref); -+ int (*domctl)(struct domain *d, struct xen_domctl *op); - int (*sysctl)(int cmd); - int (*readconsole)(uint32_t clear); - -@@ -260,9 +260,9 @@ static inline int xsm_set_target( - } - - static inline int xsm_domctl(xsm_default_t def, struct domain *d, -- unsigned int cmd, uint32_t ssidref) -+ struct xen_domctl *op) - { -- return alternative_call(xsm_ops.domctl, d, cmd, ssidref); -+ return alternative_call(xsm_ops.domctl, d, op); - } - - static inline int xsm_sysctl(xsm_default_t def, int cmd) ---- a/xen/xsm/flask/hooks.c -+++ b/xen/xsm/flask/hooks.c -@@ -667,10 +667,9 @@ static int cf_check flask_set_target(str - return rc; - } - --static int cf_check flask_domctl(struct domain *d, unsigned int cmd, -- uint32_t ssidref) -+static int cf_check flask_domctl(struct domain *d, struct xen_domctl *op) - { -- switch ( cmd ) -+ switch ( op->cmd ) - { - case XEN_DOMCTL_createdomain: - /* -@@ -680,7 +679,8 @@ static int cf_check flask_domctl(struct - * Note that d is NULL because we haven't even allocated memory for it - * this early in XEN_DOMCTL_createdomain. - */ -- return avc_current_has_perm(ssidref, SECCLASS_DOMAIN, DOMAIN__CREATE, NULL); -+ return avc_current_has_perm(op->u.createdomain.ssidref, SECCLASS_DOMAIN, -+ DOMAIN__CREATE, NULL); - - /* These have individual XSM hooks and don't make it here. */ - case XEN_DOMCTL_bind_pt_irq: -@@ -855,7 +855,7 @@ static int cf_check flask_domctl(struct - return current_has_perm(d, SECCLASS_DOMAIN2, DOMAIN2__SET_LLC_COLORS); - - default: -- return avc_unknown_permission("domctl", cmd); -+ return avc_unknown_permission("domctl", op->cmd); - } - } - diff --git a/xsa492-4.21-16.patch b/xsa492-4.21-16.patch deleted file mode 100644 index 2cb8619..0000000 --- a/xsa492-4.21-16.patch +++ /dev/null @@ -1,112 +0,0 @@ -From: Jan Beulich -Subject: domctl/XSM: drop scheduler_op hook - -Integrate the checking with xsm_domctl(), now that it has the full op -struct passed. As a positive side effect, permissions are then checked at -the same early point with and without Flask. - -This is part of XSA-492. - -Signed-off-by: Jan Beulich -Acked-by: Daniel P. Smith -Reviewed-by: Juergen Gross - ---- a/xen/common/sched/core.c -+++ b/xen/common/sched/core.c -@@ -2074,10 +2074,6 @@ long sched_adjust(struct domain *d, stru - { - long ret; - -- ret = xsm_domctl_scheduler_op(XSM_HOOK, d, op->cmd); -- if ( ret ) -- return ret; -- - if ( op->sched_id != dom_scheduler(d)->sched_id ) - return -EINVAL; - ---- a/xen/include/xsm/dummy.h -+++ b/xen/include/xsm/dummy.h -@@ -141,13 +141,6 @@ static XSM_INLINE int cf_check xsm_getdo - return xsm_default_action(action, current->domain, d); - } - --static XSM_INLINE int cf_check xsm_domctl_scheduler_op( -- XSM_DEFAULT_ARG struct domain *d, int cmd) --{ -- XSM_ASSERT_ACTION(XSM_HOOK); -- return xsm_default_action(action, current->domain, d); --} -- - static XSM_INLINE int cf_check xsm_sysctl_scheduler_op(XSM_DEFAULT_ARG int cmd) - { - XSM_ASSERT_ACTION(XSM_HOOK); ---- a/xen/include/xsm/xsm.h -+++ b/xen/include/xsm/xsm.h -@@ -56,7 +56,6 @@ struct xsm_ops { - struct xen_domctl_getdomaininfo *info); - int (*domain_create)(struct domain *d, uint32_t ssidref); - int (*getdomaininfo)(struct domain *d); -- int (*domctl_scheduler_op)(struct domain *d, int op); - #ifdef CONFIG_SYSCTL - int (*sysctl_scheduler_op)(int op); - #endif -@@ -240,12 +239,6 @@ static inline int xsm_get_domain_state(x - return alternative_call(xsm_ops.get_domain_state, d); - } - --static inline int xsm_domctl_scheduler_op( -- xsm_default_t def, struct domain *d, int cmd) --{ -- return alternative_call(xsm_ops.domctl_scheduler_op, d, cmd); --} -- - #ifdef CONFIG_SYSCTL - static inline int xsm_sysctl_scheduler_op(xsm_default_t def, int cmd) - { ---- a/xen/xsm/dummy.c -+++ b/xen/xsm/dummy.c -@@ -18,7 +18,6 @@ static const struct xsm_ops __initconst_ - .security_domaininfo = xsm_security_domaininfo, - .domain_create = xsm_domain_create, - .getdomaininfo = xsm_getdomaininfo, -- .domctl_scheduler_op = xsm_domctl_scheduler_op, - #ifdef CONFIG_SYSCTL - .sysctl_scheduler_op = xsm_sysctl_scheduler_op, - #endif ---- a/xen/xsm/flask/hooks.c -+++ b/xen/xsm/flask/hooks.c -@@ -609,7 +609,7 @@ static int cf_check flask_getdomaininfo( - return current_has_perm(d, SECCLASS_DOMAIN, DOMAIN__GETDOMAININFO); - } - --static int cf_check flask_domctl_scheduler_op(struct domain *d, int op) -+static int flask_domctl_scheduler_op(struct domain *d, int op) - { - switch ( op ) - { -@@ -697,7 +697,6 @@ static int cf_check flask_domctl(struct - return -EILSEQ; - - /* These have individual XSM hooks (common/domctl.c) */ -- case XEN_DOMCTL_scheduler_op: - case XEN_DOMCTL_set_target: - - #ifdef CONFIG_X86 -@@ -745,6 +744,9 @@ static int cf_check flask_domctl(struct - case XEN_DOMCTL_setdomainhandle: - return current_has_perm(d, SECCLASS_DOMAIN, DOMAIN__SETDOMAINHANDLE); - -+ case XEN_DOMCTL_scheduler_op: -+ return flask_domctl_scheduler_op(d, op->u.scheduler_op.cmd); -+ - case XEN_DOMCTL_set_ext_vcpucontext: - case XEN_DOMCTL_set_vcpu_msrs: - case XEN_DOMCTL_setvcpucontext: -@@ -1884,7 +1886,6 @@ static const struct xsm_ops __initconst_ - .security_domaininfo = flask_security_domaininfo, - .domain_create = flask_domain_create, - .getdomaininfo = flask_getdomaininfo, -- .domctl_scheduler_op = flask_domctl_scheduler_op, - #ifdef CONFIG_SYSCTL - .sysctl_scheduler_op = flask_sysctl_scheduler_op, - #endif diff --git a/xsa492-4.21-17.patch b/xsa492-4.21-17.patch deleted file mode 100644 index 99542df..0000000 --- a/xsa492-4.21-17.patch +++ /dev/null @@ -1,124 +0,0 @@ -From: Jan Beulich -Subject: domctl/XSM: drop shadow_control_op hook - -Integrate the checking with xsm_domctl(), now that it has the full op -struct passed. As a positive side effect, permissions are then checked at -the same early point with and without Flask. - -This is part of XSA-492. - -Signed-off-by: Jan Beulich -Acked-by: Daniel P. Smith - ---- a/xen/arch/x86/mm/paging.c -+++ b/xen/arch/x86/mm/paging.c -@@ -677,10 +677,6 @@ int paging_domctl(struct domain *d, stru - return -EBUSY; - } - -- rc = xsm_shadow_control(XSM_HOOK, d, sc->op); -- if ( rc ) -- return rc; -- - /* Code to handle log-dirty. Note that some log dirty operations - * piggy-back on shadow operations. For example, when - * XEN_DOMCTL_SHADOW_OP_OFF is called, it first checks whether log dirty ---- a/xen/include/xsm/dummy.h -+++ b/xen/include/xsm/dummy.h -@@ -682,13 +682,6 @@ static XSM_INLINE int cf_check xsm_do_mc - return xsm_default_action(action, current->domain, NULL); - } - --static XSM_INLINE int cf_check xsm_shadow_control( -- XSM_DEFAULT_ARG struct domain *d, uint32_t op) --{ -- XSM_ASSERT_ACTION(XSM_HOOK); -- return xsm_default_action(action, current->domain, d); --} -- - static XSM_INLINE int cf_check xsm_mem_sharing_op( - XSM_DEFAULT_ARG struct domain *d, struct domain *cd, int op) - { ---- a/xen/include/xsm/xsm.h -+++ b/xen/include/xsm/xsm.h -@@ -172,7 +172,6 @@ struct xsm_ops { - - #ifdef CONFIG_X86 - int (*do_mca)(void); -- int (*shadow_control)(struct domain *d, uint32_t op); - int (*mem_sharing_op)(struct domain *d, struct domain *cd, int op); - int (*apic)(struct domain *d, int cmd); - int (*machine_memory_map)(void); -@@ -680,12 +679,6 @@ static inline int xsm_do_mca(xsm_default - return alternative_call(xsm_ops.do_mca); - } - --static inline int xsm_shadow_control( -- xsm_default_t def, struct domain *d, uint32_t op) --{ -- return alternative_call(xsm_ops.shadow_control, d, op); --} -- - static inline int xsm_mem_sharing_op( - xsm_default_t def, struct domain *d, struct domain *cd, int op) - { ---- a/xen/xsm/dummy.c -+++ b/xen/xsm/dummy.c -@@ -130,7 +130,6 @@ static const struct xsm_ops __initconst_ - .platform_op = xsm_platform_op, - #ifdef CONFIG_X86 - .do_mca = xsm_do_mca, -- .shadow_control = xsm_shadow_control, - .mem_sharing_op = xsm_mem_sharing_op, - .apic = xsm_apic, - .machine_memory_map = xsm_machine_memory_map, ---- a/xen/xsm/flask/hooks.c -+++ b/xen/xsm/flask/hooks.c -@@ -40,6 +40,7 @@ - - #ifdef CONFIG_X86 - #include -+static int flask_shadow_control(struct domain *d, unsigned int op); - #else - #define pv_shim false - #endif -@@ -699,10 +700,6 @@ static int cf_check flask_domctl(struct - /* These have individual XSM hooks (common/domctl.c) */ - case XEN_DOMCTL_set_target: - --#ifdef CONFIG_X86 -- /* These have individual XSM hooks (arch/x86/domctl.c) */ -- case XEN_DOMCTL_shadow_op: --#endif - #ifdef CONFIG_HAS_PASSTHROUGH - /* - * These have individual XSM hooks -@@ -787,6 +784,11 @@ static int cf_check flask_domctl(struct - case XEN_DOMCTL_get_address_size: - return current_has_perm(d, SECCLASS_DOMAIN, DOMAIN__GETADDRSIZE); - -+#ifdef CONFIG_X86 -+ case XEN_DOMCTL_shadow_op: -+ return flask_shadow_control(d, op->u.shadow_op.op); -+#endif -+ - case XEN_DOMCTL_mem_sharing_op: - return current_has_perm(d, SECCLASS_HVM, HVM__MEM_SHARING); - -@@ -1603,7 +1605,7 @@ static int cf_check flask_do_mca(void) - return domain_has_xen(current->domain, XEN__MCA_OP); - } - --static int cf_check flask_shadow_control(struct domain *d, uint32_t op) -+static int flask_shadow_control(struct domain *d, unsigned int op) - { - uint32_t perm; - -@@ -1999,7 +2001,6 @@ static const struct xsm_ops __initconst_ - .platform_op = flask_platform_op, - #ifdef CONFIG_X86 - .do_mca = flask_do_mca, -- .shadow_control = flask_shadow_control, - .mem_sharing_op = flask_mem_sharing_op, - .apic = flask_apic, - .machine_memory_map = flask_machine_memory_map, diff --git a/xsa492-4.21-18.patch b/xsa492-4.21-18.patch deleted file mode 100644 index 1d82124..0000000 --- a/xsa492-4.21-18.patch +++ /dev/null @@ -1,94 +0,0 @@ -From: Jan Beulich -Subject: domctl: handle XEN_DOMCTL_get_device_group without acquiring domctl lock - -iommu_get_device_group() uses its own locking. Thus, with caller side -locking irrelevant, it can as well be called with the domctl lock not -held. - -Move the handling not only ahead of acquiring the lock, but also ahead -of the XSM check, leveraging that the sub-op has its own hook. - -This is part of XSA-492. - -Signed-off-by: Jan Beulich -Acked-by: Daniel P. Smith -Reviewed-by: Roger Pau Monné - ---- a/xen/common/domctl.c -+++ b/xen/common/domctl.c -@@ -513,6 +513,10 @@ long do_domctl(XEN_GUEST_HANDLE_PARAM(xe - /* Other sub-ops handled further down. */ - break; - -+ case XEN_DOMCTL_get_device_group: -+ ret = iommu_do_domctl(op, d, u_domctl); -+ goto domctl_out_unlock_domonly; -+ - case XEN_DOMCTL_ioport_permission: - case XEN_DOMCTL_ioport_mapping: - case XEN_DOMCTL_gsi_permission: -@@ -918,7 +922,6 @@ long do_domctl(XEN_GUEST_HANDLE_PARAM(xe - case XEN_DOMCTL_assign_device: - case XEN_DOMCTL_test_assign_device: - case XEN_DOMCTL_deassign_device: -- case XEN_DOMCTL_get_device_group: - ret = iommu_do_domctl(op, d, u_domctl); - break; - ---- a/xen/drivers/passthrough/pci.c -+++ b/xen/drivers/passthrough/pci.c -@@ -1620,7 +1620,7 @@ static int iommu_get_device_group( - if ( (pdev->seg != seg) || ((b == bus) && (df == devfn)) ) - continue; - -- if ( xsm_get_device_group(XSM_HOOK, (seg << 16) | (b << 8) | df) ) -+ if ( xsm_get_device_group(XSM_PRIV, (seg << 16) | (b << 8) | df) ) - continue; - - sdev_id = iommu_call(ops, get_device_group_id, seg, b, df); -@@ -1690,7 +1690,7 @@ int iommu_do_pci_domctl( - u32 max_sdevs; - XEN_GUEST_HANDLE_64(uint32) sdevs; - -- ret = xsm_get_device_group(XSM_HOOK, domctl->u.get_device_group.machine_sbdf); -+ ret = xsm_get_device_group(XSM_PRIV, domctl->u.get_device_group.machine_sbdf); - if ( ret ) - break; - ---- a/xen/include/xsm/dummy.h -+++ b/xen/include/xsm/dummy.h -@@ -162,6 +162,7 @@ static XSM_INLINE int cf_check xsm_domct - { - case XEN_DOMCTL_bind_pt_irq: - case XEN_DOMCTL_getdomaininfo: -+ case XEN_DOMCTL_get_device_group: - case XEN_DOMCTL_get_domain_state: - case XEN_DOMCTL_gsi_permission: - case XEN_DOMCTL_iomem_permission: -@@ -401,7 +402,7 @@ static XSM_INLINE int cf_check xsm_get_v - static XSM_INLINE int cf_check xsm_get_device_group( - XSM_DEFAULT_ARG uint32_t machine_bdf) - { -- XSM_ASSERT_ACTION(XSM_HOOK); -+ XSM_ASSERT_ACTION(XSM_PRIV); - return xsm_default_action(action, current->domain, NULL); - } - ---- a/xen/xsm/flask/hooks.c -+++ b/xen/xsm/flask/hooks.c -@@ -686,6 +686,7 @@ static int cf_check flask_domctl(struct - /* These have individual XSM hooks and don't make it here. */ - case XEN_DOMCTL_bind_pt_irq: - case XEN_DOMCTL_getdomaininfo: -+ case XEN_DOMCTL_get_device_group: - case XEN_DOMCTL_get_domain_state: - case XEN_DOMCTL_gsi_permission: - case XEN_DOMCTL_iomem_permission: -@@ -705,7 +706,6 @@ static int cf_check flask_domctl(struct - * These have individual XSM hooks - * (drivers/passthrough/{pci,device_tree.c) - */ -- case XEN_DOMCTL_get_device_group: - case XEN_DOMCTL_test_assign_device: - case XEN_DOMCTL_assign_device: - case XEN_DOMCTL_deassign_device: diff --git a/xsa492-4.21-19.patch b/xsa492-4.21-19.patch deleted file mode 100644 index 54a1117..0000000 --- a/xsa492-4.21-19.patch +++ /dev/null @@ -1,378 +0,0 @@ -From: Jan Beulich -Subject: domctl/XSM: drop {,de}assign_{,dt}device hooks - -Integrate the checking with xsm_domctl(). As a positive side effect, -permissions are then checked at the same early point with and without -Flask. As the DT device path needs fetching earlier (but must not be -double fetched), cache it in a private field of the public interface -struct. - -This is part of XSA-492. - -Signed-off-by: Jan Beulich -Acked-by: Daniel P. Smith -Reviewed-by: Roger Pau Monné - ---- a/xen/common/domctl.c -+++ b/xen/common/domctl.c -@@ -325,6 +325,10 @@ long do_domctl(XEN_GUEST_HANDLE_PARAM(xe - case XEN_DOMCTL_deassign_device: - if ( op->domain == DOMID_IO ) - { -+#ifdef CONFIG_HAS_DEVICE_TREE_DISCOVERY -+ if ( op->u.assign_device.dev == XEN_DOMCTL_DEV_DT ) -+ op->u.assign_device.u.dt.dev = NULL; -+#endif - d = dom_io; - break; - } -@@ -332,6 +336,11 @@ long do_domctl(XEN_GUEST_HANDLE_PARAM(xe - return -ESRCH; - fallthrough; - case XEN_DOMCTL_test_assign_device: -+#ifdef CONFIG_HAS_DEVICE_TREE_DISCOVERY -+ if ( op->u.assign_device.dev == XEN_DOMCTL_DEV_DT ) -+ op->u.assign_device.u.dt.dev = NULL; -+ fallthrough; -+#endif - case XEN_DOMCTL_vm_event_op: - if ( op->domain == DOMID_INVALID ) - { ---- a/xen/drivers/passthrough/device_tree.c -+++ b/xen/drivers/passthrough/device_tree.c -@@ -340,15 +340,15 @@ int iommu_do_dt_domctl(struct xen_domctl - if ( (d && d->is_dying) || domctl->u.assign_device.flags ) - break; - -- ret = dt_find_node_by_gpath(domctl->u.assign_device.u.dt.path, -- domctl->u.assign_device.u.dt.size, -- &dev); -- if ( ret ) -- break; -- -- ret = xsm_assign_dtdevice(XSM_HOOK, d, dt_node_full_name(dev)); -- if ( ret ) -- break; -+ dev = domctl->u.assign_device.u.dt.dev; -+ if ( !dev ) -+ { -+ ret = dt_find_node_by_gpath(domctl->u.assign_device.u.dt.path, -+ domctl->u.assign_device.u.dt.size, -+ &dev); -+ if ( ret ) -+ break; -+ } - - if ( domctl->cmd == XEN_DOMCTL_test_assign_device ) - { -@@ -396,15 +396,15 @@ int iommu_do_dt_domctl(struct xen_domctl - if ( domctl->u.assign_device.flags ) - break; - -- ret = dt_find_node_by_gpath(domctl->u.assign_device.u.dt.path, -- domctl->u.assign_device.u.dt.size, -- &dev); -- if ( ret ) -- break; -- -- ret = xsm_deassign_dtdevice(XSM_HOOK, d, dt_node_full_name(dev)); -- if ( ret ) -- break; -+ dev = domctl->u.assign_device.u.dt.dev; -+ if ( !dev ) -+ { -+ ret = dt_find_node_by_gpath(domctl->u.assign_device.u.dt.path, -+ domctl->u.assign_device.u.dt.size, -+ &dev); -+ if ( ret ) -+ break; -+ } - - if ( d == dom_io ) - { ---- a/xen/drivers/passthrough/pci.c -+++ b/xen/drivers/passthrough/pci.c -@@ -1740,10 +1740,6 @@ int iommu_do_pci_domctl( - - machine_sbdf = domctl->u.assign_device.u.pci.machine_sbdf; - -- ret = xsm_assign_device(XSM_HOOK, d, machine_sbdf); -- if ( ret ) -- break; -- - seg = machine_sbdf >> 16; - bus = PCI_BUS(machine_sbdf); - devfn = PCI_DEVFN(machine_sbdf); -@@ -1785,10 +1781,6 @@ int iommu_do_pci_domctl( - - machine_sbdf = domctl->u.assign_device.u.pci.machine_sbdf; - -- ret = xsm_deassign_device(XSM_HOOK, d, machine_sbdf); -- if ( ret ) -- break; -- - seg = machine_sbdf >> 16; - bus = PCI_BUS(machine_sbdf); - devfn = PCI_DEVFN(machine_sbdf); ---- a/xen/include/public/domctl.h -+++ b/xen/include/public/domctl.h -@@ -575,7 +575,10 @@ struct xen_domctl_assign_device { - } pci; - struct { - uint32_t size; /* Length of the path */ -- XEN_GUEST_HANDLE_64(char) path; /* path to the device tree node */ -+ XEN_GUEST_HANDLE_64(char) path; /* Path to the device tree node */ -+#ifdef __XEN__ -+ struct dt_device_node *dev; /* Resolved device node of the above */ -+#endif - } dt; - } u; - }; ---- a/xen/include/xsm/dummy.h -+++ b/xen/include/xsm/dummy.h -@@ -405,40 +405,8 @@ static XSM_INLINE int cf_check xsm_get_d - XSM_ASSERT_ACTION(XSM_PRIV); - return xsm_default_action(action, current->domain, NULL); - } -- --static XSM_INLINE int cf_check xsm_assign_device( -- XSM_DEFAULT_ARG struct domain *d, uint32_t machine_bdf) --{ -- XSM_ASSERT_ACTION(XSM_HOOK); -- return xsm_default_action(action, current->domain, d); --} -- --static XSM_INLINE int cf_check xsm_deassign_device( -- XSM_DEFAULT_ARG struct domain *d, uint32_t machine_bdf) --{ -- XSM_ASSERT_ACTION(XSM_HOOK); -- return xsm_default_action(action, current->domain, d); --} -- - #endif /* HAS_PASSTHROUGH && HAS_PCI */ - --#if defined(CONFIG_HAS_PASSTHROUGH) && defined(CONFIG_HAS_DEVICE_TREE_DISCOVERY) --static XSM_INLINE int cf_check xsm_assign_dtdevice( -- XSM_DEFAULT_ARG struct domain *d, const char *dtpath) --{ -- XSM_ASSERT_ACTION(XSM_HOOK); -- return xsm_default_action(action, current->domain, d); --} -- --static XSM_INLINE int cf_check xsm_deassign_dtdevice( -- XSM_DEFAULT_ARG struct domain *d, const char *dtpath) --{ -- XSM_ASSERT_ACTION(XSM_HOOK); -- return xsm_default_action(action, current->domain, d); --} -- --#endif /* HAS_PASSTHROUGH && HAS_DEVICE_TREE_DISCOVERY */ -- - static XSM_INLINE int cf_check xsm_resource_plug_core(XSM_DEFAULT_VOID) - { - XSM_ASSERT_ACTION(XSM_HOOK); ---- a/xen/include/xsm/xsm.h -+++ b/xen/include/xsm/xsm.h -@@ -124,13 +124,6 @@ struct xsm_ops { - - #if defined(CONFIG_HAS_PASSTHROUGH) && defined(CONFIG_HAS_PCI) - int (*get_device_group)(uint32_t machine_bdf); -- int (*assign_device)(struct domain *d, uint32_t machine_bdf); -- int (*deassign_device)(struct domain *d, uint32_t machine_bdf); --#endif -- --#if defined(CONFIG_HAS_PASSTHROUGH) && defined(CONFIG_HAS_DEVICE_TREE_DISCOVERY) -- int (*assign_dtdevice)(struct domain *d, const char *dtpath); -- int (*deassign_dtdevice)(struct domain *d, const char *dtpath); - #endif - - int (*resource_plug_core)(void); -@@ -533,35 +526,8 @@ static inline int xsm_get_device_group(x - { - return alternative_call(xsm_ops.get_device_group, machine_bdf); - } -- --static inline int xsm_assign_device( -- xsm_default_t def, struct domain *d, uint32_t machine_bdf) --{ -- return alternative_call(xsm_ops.assign_device, d, machine_bdf); --} -- --static inline int xsm_deassign_device( -- xsm_default_t def, struct domain *d, uint32_t machine_bdf) --{ -- return alternative_call(xsm_ops.deassign_device, d, machine_bdf); --} - #endif /* HAS_PASSTHROUGH && HAS_PCI) */ - --#if defined(CONFIG_HAS_PASSTHROUGH) && defined(CONFIG_HAS_DEVICE_TREE_DISCOVERY) --static inline int xsm_assign_dtdevice( -- xsm_default_t def, struct domain *d, const char *dtpath) --{ -- return alternative_call(xsm_ops.assign_dtdevice, d, dtpath); --} -- --static inline int xsm_deassign_dtdevice( -- xsm_default_t def, struct domain *d, const char *dtpath) --{ -- return alternative_call(xsm_ops.deassign_dtdevice, d, dtpath); --} -- --#endif /* HAS_PASSTHROUGH && HAS_DEVICE_TREE_DISCOVERY */ -- - static inline int xsm_resource_plug_pci(xsm_default_t def, uint32_t machine_bdf) - { - return alternative_call(xsm_ops.resource_plug_pci, machine_bdf); ---- a/xen/xsm/dummy.c -+++ b/xen/xsm/dummy.c -@@ -81,13 +81,6 @@ static const struct xsm_ops __initconst_ - - #if defined(CONFIG_HAS_PASSTHROUGH) && defined(CONFIG_HAS_PCI) - .get_device_group = xsm_get_device_group, -- .assign_device = xsm_assign_device, -- .deassign_device = xsm_deassign_device, --#endif -- --#if defined(CONFIG_HAS_PASSTHROUGH) && defined(CONFIG_HAS_DEVICE_TREE_DISCOVERY) -- .assign_dtdevice = xsm_assign_dtdevice, -- .deassign_dtdevice = xsm_deassign_dtdevice, - #endif - - .resource_plug_core = xsm_resource_plug_core, ---- a/xen/xsm/flask/hooks.c -+++ b/xen/xsm/flask/hooks.c -@@ -45,6 +45,17 @@ static int flask_shadow_control(struct d - #define pv_shim false - #endif - -+#ifdef CONFIG_HAS_PASSTHROUGH -+#ifdef CONFIG_HAS_PCI -+static int flask_assign_device(struct domain *d, unsigned int machine_bdf); -+static int flask_deassign_device(struct domain *d, unsigned int machine_bdf); -+#endif -+#ifdef CONFIG_HAS_DEVICE_TREE_DISCOVERY -+static int flask_assign_dtdevice(struct domain *d, const char *dtpath); -+static int flask_deassign_dtdevice(struct domain *d, const char *dtpath); -+#endif -+#endif /* CONFIG_HAS_PASSTHROUGH */ -+ - static uint32_t domain_sid(const struct domain *dom) - { - struct domain_security_struct *dsec = dom->ssid; -@@ -700,16 +711,6 @@ static int cf_check flask_domctl(struct - - /* These have individual XSM hooks (common/domctl.c) */ - case XEN_DOMCTL_set_target: -- --#ifdef CONFIG_HAS_PASSTHROUGH -- /* -- * These have individual XSM hooks -- * (drivers/passthrough/{pci,device_tree.c) -- */ -- case XEN_DOMCTL_test_assign_device: -- case XEN_DOMCTL_assign_device: -- case XEN_DOMCTL_deassign_device: --#endif - return 0; - - case XEN_DOMCTL_destroydomain: -@@ -789,6 +790,49 @@ static int cf_check flask_domctl(struct - return flask_shadow_control(d, op->u.shadow_op.op); - #endif - -+#ifdef CONFIG_HAS_PASSTHROUGH -+ -+ case XEN_DOMCTL_test_assign_device: -+ case XEN_DOMCTL_assign_device: -+ case XEN_DOMCTL_deassign_device: -+ switch ( op->u.assign_device.dev ) -+ { -+#ifdef CONFIG_HAS_PCI -+ case XEN_DOMCTL_DEV_PCI: -+ return op->cmd != XEN_DOMCTL_deassign_device -+ ? flask_assign_device( -+ d, op->u.assign_device.u.pci.machine_sbdf) -+ : flask_deassign_device( -+ d, op->u.assign_device.u.pci.machine_sbdf); -+#endif -+ -+#ifdef CONFIG_HAS_DEVICE_TREE_DISCOVERY -+ case XEN_DOMCTL_DEV_DT: -+ { -+ struct dt_device_node *dev; -+ int ret = dt_find_node_by_gpath(op->u.assign_device.u.dt.path, -+ op->u.assign_device.u.dt.size, -+ &dev); -+ -+ if ( ret ) -+ return ret; -+ -+ op->u.assign_device.u.dt.dev = dev; -+ -+ return op->cmd != XEN_DOMCTL_deassign_device -+ ? flask_assign_dtdevice(d, dt_node_full_name(dev)) -+ : flask_deassign_dtdevice(d, dt_node_full_name(dev)); -+ } -+#endif -+ -+ default: -+ /* Unknown type. */ -+ break; -+ } -+ return avc_unknown_permission("assign_device", op->cmd); -+ -+#endif /* CONFIG_HAS_PASSTHROUGH */ -+ - case XEN_DOMCTL_mem_sharing_op: - return current_has_perm(d, SECCLASS_HVM, HVM__MEM_SHARING); - -@@ -1416,7 +1460,7 @@ static int flask_test_assign_device(uint - return avc_current_has_perm(rsid, SECCLASS_RESOURCE, RESOURCE__STAT_DEVICE, NULL); - } - --static int cf_check flask_assign_device(struct domain *d, uint32_t machine_bdf) -+static int flask_assign_device(struct domain *d, uint32_t machine_bdf) - { - uint32_t dsid, rsid; - int rc = -EPERM; -@@ -1446,7 +1490,7 @@ static int cf_check flask_assign_device( - return avc_has_perm(dsid, rsid, SECCLASS_RESOURCE, dperm, &ad); - } - --static int cf_check flask_deassign_device( -+static int flask_deassign_device( - struct domain *d, uint32_t machine_bdf) - { - uint32_t rsid; -@@ -1478,7 +1522,7 @@ static int flask_test_assign_dtdevice(co - NULL); - } - --static int cf_check flask_assign_dtdevice(struct domain *d, const char *dtpath) -+static int flask_assign_dtdevice(struct domain *d, const char *dtpath) - { - uint32_t dsid, rsid; - int rc = -EPERM; -@@ -1508,7 +1552,7 @@ static int cf_check flask_assign_dtdevic - return avc_has_perm(dsid, rsid, SECCLASS_RESOURCE, dperm, &ad); - } - --static int cf_check flask_deassign_dtdevice( -+static int flask_deassign_dtdevice( - struct domain *d, const char *dtpath) - { - uint32_t rsid; -@@ -1989,13 +2033,6 @@ static const struct xsm_ops __initconst_ - - #if defined(CONFIG_HAS_PASSTHROUGH) && defined(CONFIG_HAS_PCI) - .get_device_group = flask_get_device_group, -- .assign_device = flask_assign_device, -- .deassign_device = flask_deassign_device, --#endif -- --#if defined(CONFIG_HAS_PASSTHROUGH) && defined(CONFIG_HAS_DEVICE_TREE_DISCOVERY) -- .assign_dtdevice = flask_assign_dtdevice, -- .deassign_dtdevice = flask_deassign_dtdevice, - #endif - - .platform_op = flask_platform_op, diff --git a/xsa492-4.21-20.patch b/xsa492-4.21-20.patch deleted file mode 100644 index bfd10a9..0000000 --- a/xsa492-4.21-20.patch +++ /dev/null @@ -1,123 +0,0 @@ -From: Jan Beulich -Subject: domctl: handle XEN_DOMCTL_set_target without acquiring domctl lock - -The only locking required here is that between checking d->target and -setting it. To avoid the need for an explicit lock, use cmpxchgptr() to -update d->target. - -Move the handling not only ahead of acquiring the lock, but also ahead -of the XSM check, leveraging that the sub-op has its own hook. - -This is part of XSA-492. - -Signed-off-by: Jan Beulich -Acked-by: Daniel P. Smith -Reviewed-by: Roger Pau Monné - ---- a/xen/common/domctl.c -+++ b/xen/common/domctl.c -@@ -505,6 +505,30 @@ long do_domctl(XEN_GUEST_HANDLE_PARAM(xe - } - #endif - -+ case XEN_DOMCTL_set_target: -+ { -+ struct domain *e = get_domain_by_id(op->u.set_target.target); -+ -+ ret = -ESRCH; -+ if ( !e ) -+ goto domctl_out_unlock_domonly; -+ -+ if ( d == e ) -+ ret = -EINVAL; -+ else if ( !is_hvm_domain(e) ) -+ ret = -EOPNOTSUPP; -+ else -+ ret = xsm_set_target(XSM_PRIV, d, e); -+ -+ /* Hold reference on @e until we destroy @d. */ -+ if ( !ret && cmpxchgptr(&d->target, NULL, e) ) -+ ret = -EINVAL; -+ -+ if ( ret ) -+ put_domain(e); -+ goto domctl_out_unlock_domonly; -+ } -+ - case XEN_DOMCTL_vm_event_op: - if ( op->u.vm_event_op.op == XEN_VM_EVENT_GET_VERSION ) - { -@@ -844,36 +868,6 @@ long do_domctl(XEN_GUEST_HANDLE_PARAM(xe - domain_set_time_offset(d, op->u.settimeoffset.time_offset_seconds); - break; - -- case XEN_DOMCTL_set_target: -- { -- struct domain *e; -- -- ret = -ESRCH; -- e = get_domain_by_id(op->u.set_target.target); -- if ( e == NULL ) -- break; -- -- ret = -EINVAL; -- if ( (d == e) || (d->target != NULL) ) -- { -- put_domain(e); -- break; -- } -- -- ret = -EOPNOTSUPP; -- if ( is_hvm_domain(e) ) -- ret = xsm_set_target(XSM_HOOK, d, e); -- if ( ret ) -- { -- put_domain(e); -- break; -- } -- -- /* Hold reference on @e until we destroy @d. */ -- d->target = e; -- break; -- } -- - case XEN_DOMCTL_subscribe: - d->suspend_evtchn = op->u.subscribe.port; - break; ---- a/xen/include/xsm/dummy.h -+++ b/xen/include/xsm/dummy.h -@@ -150,7 +150,7 @@ static XSM_INLINE int cf_check xsm_sysct - static XSM_INLINE int cf_check xsm_set_target( - XSM_DEFAULT_ARG struct domain *d, struct domain *e) - { -- XSM_ASSERT_ACTION(XSM_HOOK); -+ XSM_ASSERT_ACTION(XSM_PRIV); - return xsm_default_action(action, current->domain, NULL); - } - -@@ -170,6 +170,7 @@ static XSM_INLINE int cf_check xsm_domct - case XEN_DOMCTL_ioport_permission: - case XEN_DOMCTL_irq_permission: - case XEN_DOMCTL_memory_mapping: -+ case XEN_DOMCTL_set_target: - case XEN_DOMCTL_unbind_pt_irq: - ASSERT_UNREACHABLE(); - return -EILSEQ; ---- a/xen/xsm/flask/hooks.c -+++ b/xen/xsm/flask/hooks.c -@@ -705,14 +705,11 @@ static int cf_check flask_domctl(struct - case XEN_DOMCTL_ioport_permission: - case XEN_DOMCTL_irq_permission: - case XEN_DOMCTL_memory_mapping: -+ case XEN_DOMCTL_set_target: - case XEN_DOMCTL_unbind_pt_irq: - ASSERT_UNREACHABLE(); - return -EILSEQ; - -- /* These have individual XSM hooks (common/domctl.c) */ -- case XEN_DOMCTL_set_target: -- return 0; -- - case XEN_DOMCTL_destroydomain: - return current_has_perm(d, SECCLASS_DOMAIN, DOMAIN__DESTROY); - diff --git a/xsa493-4.21-01.patch b/xsa493-4.21-01.patch deleted file mode 100644 index c06b9a5..0000000 --- a/xsa493-4.21-01.patch +++ /dev/null @@ -1,311 +0,0 @@ -From 2e21b5301765de353c06081eee953255bf327176 Mon Sep 17 00:00:00 2001 -From: Michal Orzel -Date: Tue, 14 Apr 2026 10:11:24 +0200 -Subject: xen/arm64: flushtlb: Optimize ARM64_WORKAROUND_REPEAT_TLBI - -The ARM64_WORKAROUND_REPEAT_TLBI workaround is used to mitigate several -errata where broadcast TLBI;DSB sequences don't provide all the -architecturally required synchronization. The workaround performs more -work than necessary, and can have significant overhead. This patch -optimizes the workaround, as explained below. - -1. All relevant errata only affect the ordering and/or completion of - memory accesses which have been translated by an invalidated TLB - entry. The actual invalidation of TLB entries is unaffected. - -2. The existing workaround is applied to both broadcast and local TLB - invalidation, whereas for all relevant errata it is only necessary to - apply a workaround for broadcast invalidation. - -3. The existing workaround replaces every TLBI with a TLBI;DSB;TLBI - sequence, whereas for all relevant errata it is only necessary to - execute a single additional TLBI;DSB sequence after any number of - TLBIs are completed by a DSB. - - For example, for a sequence of batched TLBIs: - - TLBI [, ] - TLBI [, ] - TLBI [, ] - DSB ISH - - ... the existing workaround will expand this to: - - TLBI [, ] - DSB ISH // additional - TLBI [, ] // additional - TLBI [, ] - DSB ISH // additional - TLBI [, ] // additional - TLBI [, ] - DSB ISH // additional - TLBI [, ] // additional - DSB ISH - - ... whereas it is sufficient to have: - - TLBI [, ] - TLBI [, ] - TLBI [, ] - DSB ISH - TLBI [, ] // additional - DSB ISH // additional - - Using a single additional TLBI and DSB at the end of the sequence can - have significantly lower overhead as each DSB which completes a TLBI - must synchronize with other PEs in the system, with potential - performance effects both locally and system-wide. - -4. The existing workaround repeats each specific TLBI operation, whereas - for all relevant errata it is sufficient for the additional TLBI to - use *any* operation which will be broadcast, regardless of which - translation regime or stage of translation the operation applies to. - - For example, for a single TLBI: - - TLBI ALLE2IS - DSB ISH - - ... the existing workaround will expand this to: - - TLBI ALLE2IS - DSB ISH - TLBI ALLE2IS // additional - DSB ISH // additional - - ... whereas it is sufficient to have: - - TLBI ALLE2IS - DSB ISH - TLBI VALE1IS, XZR // additional - DSB ISH // additional - - As the additional TLBI doesn't have to match a specific earlier TLBI, - the additional TLBI can be implemented in separate code, with no - memory of the earlier TLBIs. The additional TLBI can also use a - cheaper TLBI operation. - -5. The existing workaround is applied to both Stage-1 and Stage-2 TLB - invalidation, whereas for all relevant errata it is only necessary to - apply a workaround for Stage-1 invalidation. - - Architecturally, TLBI operations which invalidate only Stage-2 - information (e.g. IPAS2E1IS) are not required to invalidate TLB - entries which combine information from Stage-1 and Stage-2 - translation table entries, and consequently may not complete memory - accesses translated by those combined entries. In these cases, - completion of memory accesses is only guaranteed after subsequent - invalidation of Stage-1 information (e.g. VMALLE1IS). - -Rework the workaround logic as follows: - - add TLB_HELPER_LOCAL() to be used for local TLB ops without a - workaround, - - modify TLB_HELPER() workaround to use tlbi vale2is, xzr as a second - TLBI, - - drop TLB_HELPER_VA(). It's used only by __flush_xen_tlb_one_local - which is local and does not need workaround and by - __flush_xen_tlb_one. In the latter case, since it's used in a loop, - we don't need a workaround in the middle. Add __tlb_repeat_sync with - a workaround to be used at the end after DSB and before final ISB, - - TLBI VALE2IS passing XZR is used as an additional TLBI. While there is - an identity mapping there, it's used very rarely. The performance - impact is therefore negligible. If things change in the future, we - can revisit the decision. - -Signed-off-by: Michal Orzel -Reviewed-by: Luca Fancellu -Reviewed-by: Julien Grall -(cherry picked from commit 7c502d7591519135765b8041cbd1c70e56e5a0b9) - -diff --git a/xen/arch/arm/include/asm/arm32/flushtlb.h b/xen/arch/arm/include/asm/arm32/flushtlb.h -index 61c25a318998..5483be08fbbe 100644 ---- a/xen/arch/arm/include/asm/arm32/flushtlb.h -+++ b/xen/arch/arm/include/asm/arm32/flushtlb.h -@@ -57,6 +57,9 @@ static inline void __flush_xen_tlb_one(vaddr_t va) - asm volatile(STORE_CP32(0, TLBIMVAHIS) : : "r" (va) : "memory"); - } - -+/* Only for ARM64_WORKAROUND_REPEAT_TLBI */ -+static inline void __tlb_repeat_sync(void) {} -+ - #endif /* __ASM_ARM_ARM32_FLUSHTLB_H__ */ - /* - * Local variables: -diff --git a/xen/arch/arm/include/asm/arm64/flushtlb.h b/xen/arch/arm/include/asm/arm64/flushtlb.h -index 3b99c11b50d1..1606b26bf28a 100644 ---- a/xen/arch/arm/include/asm/arm64/flushtlb.h -+++ b/xen/arch/arm/include/asm/arm64/flushtlb.h -@@ -12,9 +12,14 @@ - * ARM64_WORKAROUND_REPEAT_TLBI: - * Modification of the translation table for a virtual address might lead to - * read-after-read ordering violation. -- * The workaround repeats TLBI+DSB ISH operation for all the TLB flush -- * operations. While this is strictly not necessary, we don't want to -- * take any risk. -+ * The workaround repeats TLBI+DSB ISH operation for broadcast TLB flush -+ * operations. The workaround is not needed for local operations. -+ * -+ * It is sufficient for the additional TLBI to use *any* operation which will -+ * be broadcast, regardless of which translation regime or stage of translation -+ * the operation applies to. TLBI VALE2IS is used passing XZR. While there is -+ * an identity mapping there, it's only used during suspend/resume, CPU on/off, -+ * so the impact (performance if any) is negligible. - * - * For Xen page-tables the ISB will discard any instructions fetched - * from the old mappings. -@@ -26,69 +31,90 @@ - * Note that for local TLB flush, using non-shareable (nsh) is sufficient - * (see D5-4929 in ARM DDI 0487H.a). Although, the memory barrier in - * for the workaround is left as inner-shareable to match with Linux -- * v6.1-rc8. -+ * v6.19. - */ --#define TLB_HELPER(name, tlbop, sh) \ -+#define TLB_HELPER_LOCAL(name, tlbop) \ - static inline void name(void) \ - { \ - asm_inline volatile ( \ -- "dsb " # sh "st;" \ -+ "dsb nshst;" \ - "tlbi " # tlbop ";" \ -- ALTERNATIVE( \ -- "nop; nop;", \ -- "dsb ish;" \ -- "tlbi " # tlbop ";", \ -- ARM64_WORKAROUND_REPEAT_TLBI, \ -- CONFIG_ARM64_WORKAROUND_REPEAT_TLBI) \ -- "dsb " # sh ";" \ -+ "dsb nsh;" \ - "isb;" \ - : : : "memory"); \ - } - --/* -- * FLush TLB by VA. This will likely be used in a loop, so the caller -- * is responsible to use the appropriate memory barriers before/after -- * the sequence. -- * -- * See above about the ARM64_WORKAROUND_REPEAT_TLBI sequence. -- */ --#define TLB_HELPER_VA(name, tlbop) \ --static inline void name(vaddr_t va) \ --{ \ -- asm_inline volatile ( \ -- "tlbi " # tlbop ", %0;" \ -- ALTERNATIVE( \ -- "nop; nop;", \ -- "dsb ish;" \ -- "tlbi " # tlbop ", %0;", \ -- ARM64_WORKAROUND_REPEAT_TLBI, \ -- CONFIG_ARM64_WORKAROUND_REPEAT_TLBI) \ -- : : "r" (va >> PAGE_SHIFT) : "memory"); \ -+#define TLB_HELPER(name, tlbop) \ -+static inline void name(void) \ -+{ \ -+ asm_inline volatile ( \ -+ "dsb ishst;" \ -+ "tlbi " # tlbop ";" \ -+ ALTERNATIVE( \ -+ "nop; nop;", \ -+ "dsb ish;" \ -+ "tlbi vale2is, xzr;", \ -+ ARM64_WORKAROUND_REPEAT_TLBI, \ -+ CONFIG_ARM64_WORKAROUND_REPEAT_TLBI) \ -+ "dsb ish;" \ -+ "isb;" \ -+ : : : "memory"); \ - } - - /* Flush local TLBs, current VMID only. */ --TLB_HELPER(flush_guest_tlb_local, vmalls12e1, nsh) -+TLB_HELPER_LOCAL(flush_guest_tlb_local, vmalls12e1) - - /* Flush innershareable TLBs, current VMID only */ --TLB_HELPER(flush_guest_tlb, vmalls12e1is, ish) -+TLB_HELPER(flush_guest_tlb, vmalls12e1is) - - /* Flush local TLBs, all VMIDs, non-hypervisor mode */ --TLB_HELPER(flush_all_guests_tlb_local, alle1, nsh) -+TLB_HELPER_LOCAL(flush_all_guests_tlb_local, alle1) - - /* Flush innershareable TLBs, all VMIDs, non-hypervisor mode */ --TLB_HELPER(flush_all_guests_tlb, alle1is, ish) -+TLB_HELPER(flush_all_guests_tlb, alle1is) - - /* Flush all hypervisor mappings from the TLB of the local processor. */ --TLB_HELPER(flush_xen_tlb_local, alle2, nsh) -+TLB_HELPER_LOCAL(flush_xen_tlb_local, alle2) -+ -+#undef TLB_HELPER_LOCAL -+#undef TLB_HELPER -+ -+/* -+ * FLush TLB by VA. This will likely be used in a loop, so the caller -+ * is responsible to use the appropriate memory barriers before/after -+ * the sequence. -+ */ - - /* Flush TLB of local processor for address va. */ --TLB_HELPER_VA(__flush_xen_tlb_one_local, vae2) -+static inline void __flush_xen_tlb_one_local(vaddr_t va) -+{ -+ asm_inline volatile ( -+ "tlbi vae2, %0" : : "r" (va >> PAGE_SHIFT) : "memory"); -+} - - /* Flush TLB of all processors in the inner-shareable domain for address va. */ --TLB_HELPER_VA(__flush_xen_tlb_one, vae2is) -+static inline void __flush_xen_tlb_one(vaddr_t va) -+{ -+ asm_inline volatile ( -+ "tlbi vae2is, %0" : : "r" (va >> PAGE_SHIFT) : "memory"); -+} - --#undef TLB_HELPER --#undef TLB_HELPER_VA -+/* -+ * ARM64_WORKAROUND_REPEAT_TLBI: -+ * For all relevant erratas it is only necessary to execute a single -+ * additional TLBI;DSB sequence after any number of TLBIs are completed by DSB. -+ */ -+static inline void __tlb_repeat_sync(void) -+{ -+ asm_inline volatile ( -+ ALTERNATIVE( -+ "nop; nop;", -+ "tlbi vale2is, xzr;" -+ "dsb ish;", -+ ARM64_WORKAROUND_REPEAT_TLBI, -+ CONFIG_ARM64_WORKAROUND_REPEAT_TLBI) -+ : : : "memory"); -+} - - #endif /* __ASM_ARM_ARM64_FLUSHTLB_H__ */ - /* -diff --git a/xen/arch/arm/include/asm/flushtlb.h b/xen/arch/arm/include/asm/flushtlb.h -index e45fb6d97b02..c292c3c00d29 100644 ---- a/xen/arch/arm/include/asm/flushtlb.h -+++ b/xen/arch/arm/include/asm/flushtlb.h -@@ -65,6 +65,7 @@ static inline void flush_xen_tlb_range_va(vaddr_t va, - va += PAGE_SIZE; - } - dsb(ish); /* Ensure the TLB invalidation has completed */ -+ __tlb_repeat_sync(); - isb(); - } - -diff --git a/xen/arch/arm/include/asm/mmu/layout.h b/xen/arch/arm/include/asm/mmu/layout.h -index 19c0ec63a59a..feafc14ebfda 100644 ---- a/xen/arch/arm/include/asm/mmu/layout.h -+++ b/xen/arch/arm/include/asm/mmu/layout.h -@@ -23,6 +23,10 @@ - * - * Reserved to identity map Xen - * -+ * Note: As part of ARM64_WORKAROUND_REPEAT_TLBI, VA 0 is used for an extra -+ * TLBI operation given its rare use (only identity mapping) and thus -+ * negligible performance impact. -+ * - * 0x00000a0000000000 - 0x00000a7fffffffff (512GB, L0 slot [20]) - * (Relative offsets) - * 0 - 2M Unmapped diff --git a/xsa493-4.21-02.patch b/xsa493-4.21-02.patch deleted file mode 100644 index f80119c..0000000 --- a/xsa493-4.21-02.patch +++ /dev/null @@ -1,71 +0,0 @@ -From 7e70b87512c966248b1e8453d9ac54c643c06f44 Mon Sep 17 00:00:00 2001 -From: Michal Orzel -Date: Fri, 22 May 2026 09:35:55 +0200 -Subject: xen/arm: Sync missing definitions for Arm CPUs with Linux - -Synchronize with Linux kernel 7.0 definitions for the following CPUs: - - Cortex-A76AE, - - Cortex-A78AE, - - Cortex-X1C, - - Cortex-X3, - - Neoverse-V2, - - Cortex-X4, - - Neoverse-V3AE, - - Neoverse-V3, - - Cortex-X925. - -These will be used for errata detection in subsequent patches. - -Signed-off-by: Michal Orzel -Reviewed-by: Julien Grall - -diff --git a/xen/arch/arm/include/asm/processor.h b/xen/arch/arm/include/asm/processor.h -index ec23fd098b63..907778683b08 100644 ---- a/xen/arch/arm/include/asm/processor.h -+++ b/xen/arch/arm/include/asm/processor.h -@@ -89,13 +89,22 @@ - #define ARM_CPU_PART_CORTEX_A76 0xD0B - #define ARM_CPU_PART_NEOVERSE_N1 0xD0C - #define ARM_CPU_PART_CORTEX_A77 0xD0D -+#define ARM_CPU_PART_CORTEX_A76AE 0xD0E - #define ARM_CPU_PART_NEOVERSE_V1 0xD40 - #define ARM_CPU_PART_CORTEX_A78 0xD41 -+#define ARM_CPU_PART_CORTEX_A78AE 0xD42 - #define ARM_CPU_PART_CORTEX_X1 0xD44 - #define ARM_CPU_PART_CORTEX_A710 0xD47 - #define ARM_CPU_PART_CORTEX_X2 0xD48 - #define ARM_CPU_PART_NEOVERSE_N2 0xD49 - #define ARM_CPU_PART_CORTEX_A78C 0xD4B -+#define ARM_CPU_PART_CORTEX_X1C 0xD4C -+#define ARM_CPU_PART_CORTEX_X3 0xD4E -+#define ARM_CPU_PART_NEOVERSE_V2 0xD4F -+#define ARM_CPU_PART_CORTEX_X4 0xD82 -+#define ARM_CPU_PART_NEOVERSE_V3AE 0xD83 -+#define ARM_CPU_PART_NEOVERSE_V3 0xD84 -+#define ARM_CPU_PART_CORTEX_X925 0xD85 - - #define MIDR_CORTEX_A12 MIDR_CPU_MODEL(ARM_CPU_IMP_ARM, ARM_CPU_PART_CORTEX_A12) - #define MIDR_CORTEX_A17 MIDR_CPU_MODEL(ARM_CPU_IMP_ARM, ARM_CPU_PART_CORTEX_A17) -@@ -110,13 +119,22 @@ - #define MIDR_CORTEX_A76 MIDR_CPU_MODEL(ARM_CPU_IMP_ARM, ARM_CPU_PART_CORTEX_A76) - #define MIDR_NEOVERSE_N1 MIDR_CPU_MODEL(ARM_CPU_IMP_ARM, ARM_CPU_PART_NEOVERSE_N1) - #define MIDR_CORTEX_A77 MIDR_CPU_MODEL(ARM_CPU_IMP_ARM, ARM_CPU_PART_CORTEX_A77) -+#define MIDR_CORTEX_A76AE MIDR_CPU_MODEL(ARM_CPU_IMP_ARM, ARM_CPU_PART_CORTEX_A76AE) - #define MIDR_NEOVERSE_V1 MIDR_CPU_MODEL(ARM_CPU_IMP_ARM, ARM_CPU_PART_NEOVERSE_V1) - #define MIDR_CORTEX_A78 MIDR_CPU_MODEL(ARM_CPU_IMP_ARM, ARM_CPU_PART_CORTEX_A78) -+#define MIDR_CORTEX_A78AE MIDR_CPU_MODEL(ARM_CPU_IMP_ARM, ARM_CPU_PART_CORTEX_A78AE) - #define MIDR_CORTEX_X1 MIDR_CPU_MODEL(ARM_CPU_IMP_ARM, ARM_CPU_PART_CORTEX_X1) - #define MIDR_CORTEX_A710 MIDR_CPU_MODEL(ARM_CPU_IMP_ARM, ARM_CPU_PART_CORTEX_A710) - #define MIDR_CORTEX_X2 MIDR_CPU_MODEL(ARM_CPU_IMP_ARM, ARM_CPU_PART_CORTEX_X2) - #define MIDR_NEOVERSE_N2 MIDR_CPU_MODEL(ARM_CPU_IMP_ARM, ARM_CPU_PART_NEOVERSE_N2) - #define MIDR_CORTEX_A78C MIDR_CPU_MODEL(ARM_CPU_IMP_ARM, ARM_CPU_PART_CORTEX_A78C) -+#define MIDR_CORTEX_X1C MIDR_CPU_MODEL(ARM_CPU_IMP_ARM, ARM_CPU_PART_CORTEX_X1C) -+#define MIDR_CORTEX_X3 MIDR_CPU_MODEL(ARM_CPU_IMP_ARM, ARM_CPU_PART_CORTEX_X3) -+#define MIDR_NEOVERSE_V2 MIDR_CPU_MODEL(ARM_CPU_IMP_ARM, ARM_CPU_PART_NEOVERSE_V2) -+#define MIDR_CORTEX_X4 MIDR_CPU_MODEL(ARM_CPU_IMP_ARM, ARM_CPU_PART_CORTEX_X4) -+#define MIDR_NEOVERSE_V3AE MIDR_CPU_MODEL(ARM_CPU_IMP_ARM, ARM_CPU_PART_NEOVERSE_V3AE) -+#define MIDR_NEOVERSE_V3 MIDR_CPU_MODEL(ARM_CPU_IMP_ARM, ARM_CPU_PART_NEOVERSE_V3) -+#define MIDR_CORTEX_X925 MIDR_CPU_MODEL(ARM_CPU_IMP_ARM, ARM_CPU_PART_CORTEX_X925) - - /* MPIDR Multiprocessor Affinity Register */ - #define _MPIDR_UP (30) diff --git a/xsa493-4.21-03.patch b/xsa493-4.21-03.patch deleted file mode 100644 index 86bae68..0000000 --- a/xsa493-4.21-03.patch +++ /dev/null @@ -1,37 +0,0 @@ -From c0f7b40fdbb986b3cf470ed51f3878261e33f9cb Mon Sep 17 00:00:00 2001 -From: Michal Orzel -Date: Fri, 22 May 2026 09:35:56 +0200 -Subject: xen/arm: Add C1-Ultra definitions - -Add processor definitions for C1-Ultra. These will be used for errata -detection in subsequent patches. - -These values can be found in the C1-Ultra TRM: - - https://developer.arm.com/documentation/108014/0100/ - -... in section A.5.1 ("MIDR_EL1, Main ID Register"). - -Signed-off-by: Michal Orzel -Reviewed-by: Julien Grall - -diff --git a/xen/arch/arm/include/asm/processor.h b/xen/arch/arm/include/asm/processor.h -index 907778683b08..72745cca62bc 100644 ---- a/xen/arch/arm/include/asm/processor.h -+++ b/xen/arch/arm/include/asm/processor.h -@@ -105,6 +105,7 @@ - #define ARM_CPU_PART_NEOVERSE_V3AE 0xD83 - #define ARM_CPU_PART_NEOVERSE_V3 0xD84 - #define ARM_CPU_PART_CORTEX_X925 0xD85 -+#define ARM_CPU_PART_C1_ULTRA 0xD8C - - #define MIDR_CORTEX_A12 MIDR_CPU_MODEL(ARM_CPU_IMP_ARM, ARM_CPU_PART_CORTEX_A12) - #define MIDR_CORTEX_A17 MIDR_CPU_MODEL(ARM_CPU_IMP_ARM, ARM_CPU_PART_CORTEX_A17) -@@ -135,6 +136,7 @@ - #define MIDR_NEOVERSE_V3AE MIDR_CPU_MODEL(ARM_CPU_IMP_ARM, ARM_CPU_PART_NEOVERSE_V3AE) - #define MIDR_NEOVERSE_V3 MIDR_CPU_MODEL(ARM_CPU_IMP_ARM, ARM_CPU_PART_NEOVERSE_V3) - #define MIDR_CORTEX_X925 MIDR_CPU_MODEL(ARM_CPU_IMP_ARM, ARM_CPU_PART_CORTEX_X925) -+#define MIDR_C1_ULTRA MIDR_CPU_MODEL(ARM_CPU_IMP_ARM, ARM_CPU_PART_C1_ULTRA) - - /* MPIDR Multiprocessor Affinity Register */ - #define _MPIDR_UP (30) diff --git a/xsa493-4.21-04.patch b/xsa493-4.21-04.patch deleted file mode 100644 index b59ee74..0000000 --- a/xsa493-4.21-04.patch +++ /dev/null @@ -1,37 +0,0 @@ -From 6af67aeca418bffb807424eb3415fab59e581733 Mon Sep 17 00:00:00 2001 -From: Michal Orzel -Date: Fri, 22 May 2026 09:35:57 +0200 -Subject: xen/arm: Add C1-Premium definitions - -Add processor definitions for C1-Premium. These will be used for errata -detection in subsequent patches. - -These values can be found in the C1-Premium TRM: - - https://developer.arm.com/documentation/109416/0100/ - -... in section A.5.1 ("MIDR_EL1, Main ID Register"). - -Signed-off-by: Michal Orzel -Reviewed-by: Julien Grall - -diff --git a/xen/arch/arm/include/asm/processor.h b/xen/arch/arm/include/asm/processor.h -index 72745cca62bc..25c5762c6706 100644 ---- a/xen/arch/arm/include/asm/processor.h -+++ b/xen/arch/arm/include/asm/processor.h -@@ -106,6 +106,7 @@ - #define ARM_CPU_PART_NEOVERSE_V3 0xD84 - #define ARM_CPU_PART_CORTEX_X925 0xD85 - #define ARM_CPU_PART_C1_ULTRA 0xD8C -+#define ARM_CPU_PART_C1_PREMIUM 0xD90 - - #define MIDR_CORTEX_A12 MIDR_CPU_MODEL(ARM_CPU_IMP_ARM, ARM_CPU_PART_CORTEX_A12) - #define MIDR_CORTEX_A17 MIDR_CPU_MODEL(ARM_CPU_IMP_ARM, ARM_CPU_PART_CORTEX_A17) -@@ -137,6 +138,7 @@ - #define MIDR_NEOVERSE_V3 MIDR_CPU_MODEL(ARM_CPU_IMP_ARM, ARM_CPU_PART_NEOVERSE_V3) - #define MIDR_CORTEX_X925 MIDR_CPU_MODEL(ARM_CPU_IMP_ARM, ARM_CPU_PART_CORTEX_X925) - #define MIDR_C1_ULTRA MIDR_CPU_MODEL(ARM_CPU_IMP_ARM, ARM_CPU_PART_C1_ULTRA) -+#define MIDR_C1_PREMIUM MIDR_CPU_MODEL(ARM_CPU_IMP_ARM, ARM_CPU_PART_C1_PREMIUM) - - /* MPIDR Multiprocessor Affinity Register */ - #define _MPIDR_UP (30) diff --git a/xsa494-4.21.patch b/xsa494-4.21.patch deleted file mode 100644 index d52a0f1..0000000 --- a/xsa494-4.21.patch +++ /dev/null @@ -1,404 +0,0 @@ -From 579016a359741044c9076bf0884e1dbab00ab080 Mon Sep 17 00:00:00 2001 -From: Roger Pau Monne -Date: Mon, 16 Mar 2026 11:03:22 +0100 -Subject: [PATCH] x86/mm: accurately track which vCPU page-tables are loaded -MIME-Version: 1.0 -Content-Type: text/plain; charset=UTF-8 -Content-Transfer-Encoding: 8bit - -Neither current nor curr_vcpu per-CPU fields accurately track which -page-tables are loaded. There are corner cases when dealing with shadow -paging failures that switch to the idle vCPU page-tables without changing -current or curr_vcpu per-CPU fields. - -Introduce a new per-CPU field that attempts to track which vCPU page-tables -are loaded. Update such tracking when cr3 is changed, and do so in a -region with interrupts disabled, as to avoid handling interrupts with a -mismatch between the vCPU tracking field and the loaded page-tables. - -As a result of this newly more accurate tracking the mapcache override -functionality can be removed: the dom0 PV builder was the only user of it, -and it's updated here to properly signal which vCPU page-tables are loaded -in the calls to switch_cr3_cr4(). - -Note the EFI page-tables have the Xen owned L4 slots copied from the idle -page-tables, so for the effects of the mapcache the EFI page-tables could -use the idle mapcache if it had one. Pass the idle vCPU in the -switch_cr3_cr4() call that switches to the runtime EFI page-tables. - -There are known issues with the use of mapcache in NMI context.  This patch -does not alter the behaviour. - -This is CVE-2026-42488 / XSA-494. - -Fixes: fb0ff49fe9f7 ("x86/shadow: defer releasing of PV's top-level shadow reference") -Signed-off-by: Roger Pau Monné -Acked-by: Andrew Cooper ---- - xen/arch/x86/domain_page.c | 48 ++++++++++++---------------- - xen/arch/x86/flushtlb.c | 5 ++- - xen/arch/x86/include/asm/domain.h | 1 - - xen/arch/x86/include/asm/flushtlb.h | 2 +- - xen/arch/x86/include/asm/processor.h | 3 ++ - xen/arch/x86/mm.c | 4 +-- - xen/arch/x86/pv/dom0_build.c | 12 +++---- - xen/arch/x86/pv/domain.c | 13 ++++++-- - xen/arch/x86/smpboot.c | 1 + - xen/common/efi/common-stub.c | 5 --- - xen/common/efi/runtime.c | 21 +++++------- - xen/include/xen/efi.h | 1 - - 12 files changed, 54 insertions(+), 62 deletions(-) - -diff --git a/xen/arch/x86/domain_page.c b/xen/arch/x86/domain_page.c -index eac5e3304fb8..72c00194f315 100644 ---- a/xen/arch/x86/domain_page.c -+++ b/xen/arch/x86/domain_page.c -@@ -18,48 +18,40 @@ - #include - #include - --static DEFINE_PER_CPU(struct vcpu *, override); -- - static inline struct vcpu *mapcache_current_vcpu(void) - { -- /* In the common case we use the mapcache of the running VCPU. */ -- struct vcpu *v = this_cpu(override) ?: current; -- -- /* -- * When current isn't properly set up yet, this is equivalent to -- * running in an idle vCPU (callers must check for NULL). -- */ -- if ( !v ) -- return NULL; -+ struct vcpu *v = this_cpu(pgtable_vcpu); -+ struct vcpu *curr = current; - - /* -- * When using efi runtime page tables, we have the equivalent of the idle -- * domain's page tables but current may point at another domain's VCPU. -- * Return NULL as though current is not properly set up yet. -+ * During early boot pgtable_vcpu is not set, callers must handle NULL. -+ * Non-PV domains don't have a mapcache, the directmap covers all physical -+ * address space. - */ -- if ( efi_rs_using_pgtables() ) -+ if ( !v || !is_pv_vcpu(v) ) - return NULL; - - /* -- * If guest_table is NULL, and we are running a paravirtualised guest, -- * then it means we are running on the idle domain's page table and must -- * therefore use its mapcache. -+ * If we are in a lazy context-switch state from a PV vCPU do a full switch -+ * to the idle vCPU now, otherwise an incoming FLUSH_VCPU_STATE IPI would -+ * change the page tables under our feet an invalidate any in-use mapcache -+ * entries. - */ -- if ( unlikely(pagetable_is_null(v->arch.guest_table)) && is_pv_vcpu(v) ) -+ if ( unlikely(this_cpu(curr_vcpu) != curr) ) - { -- /* If we really are idling, perform lazy context switch now. */ -- if ( (v = idle_vcpu[smp_processor_id()]) == current ) -- sync_local_execstate(); -+ ASSERT(curr == idle_vcpu[smp_processor_id()]); -+ sync_local_execstate(); - /* We must now be running on the idle page table. */ - ASSERT(cr3_pa(read_cr3()) == __pa(idle_pg_table)); - } - -- return v; --} -- --void __init mapcache_override_current(struct vcpu *v) --{ -- this_cpu(override) = v; -+ /* -+ * At this point we can guarantee Xen is not in lazy context switch: either -+ * the code above will have synced the state, or an incoming -+ * FLUSH_VCPU_STATE IPI has done so behind our back. Use ACCESS_ONCE to -+ * ensure the compiler never returns the locally cached pgtable_vcpu value. -+ */ -+ return ACCESS_ONCE(this_cpu(pgtable_vcpu)); - } - - #define mapcache_l2_entry(e) ((e) >> PAGETABLE_ORDER) -diff --git a/xen/arch/x86/flushtlb.c b/xen/arch/x86/flushtlb.c -index 09e676c151fa..928bca66b433 100644 ---- a/xen/arch/x86/flushtlb.c -+++ b/xen/arch/x86/flushtlb.c -@@ -111,7 +111,9 @@ static void do_tlb_flush(void) - local_irq_restore(flags); - } - --void switch_cr3_cr4(unsigned long cr3, unsigned long cr4) -+DEFINE_PER_CPU(struct vcpu *, pgtable_vcpu); -+ -+void switch_cr3_cr4(struct vcpu *v, unsigned long cr3, unsigned long cr4) - { - unsigned long flags, old_cr4; - u32 t = 0; -@@ -155,6 +157,7 @@ void switch_cr3_cr4(unsigned long cr3, unsigned long cr4) - if ( (old_cr4 & X86_CR4_PCIDE) > (cr4 & X86_CR4_PCIDE) ) - cr3 |= X86_CR3_NOFLUSH; - write_cr3(cr3); -+ this_cpu(pgtable_vcpu) = v; - - if ( old_cr4 != cr4 ) - write_cr4(cr4); -diff --git a/xen/arch/x86/include/asm/domain.h b/xen/arch/x86/include/asm/domain.h -index 828f42c3e448..10d2b9fe2546 100644 ---- a/xen/arch/x86/include/asm/domain.h -+++ b/xen/arch/x86/include/asm/domain.h -@@ -75,7 +75,6 @@ struct mapcache_domain { - - int mapcache_domain_init(struct domain *d); - int mapcache_vcpu_init(struct vcpu *v); --void mapcache_override_current(struct vcpu *v); - - /* x86/64: toggle guest between kernel and user modes. */ - void toggle_guest_mode(struct vcpu *v); -diff --git a/xen/arch/x86/include/asm/flushtlb.h b/xen/arch/x86/include/asm/flushtlb.h -index 7bcbca2b7f31..345677eb72ae 100644 ---- a/xen/arch/x86/include/asm/flushtlb.h -+++ b/xen/arch/x86/include/asm/flushtlb.h -@@ -104,7 +104,7 @@ static inline void invlpg(const void *p) - } - - /* Write pagetable base and implicitly tick the tlbflush clock. */ --void switch_cr3_cr4(unsigned long cr3, unsigned long cr4); -+void switch_cr3_cr4(struct vcpu *v, unsigned long cr3, unsigned long cr4); - - /* flush_* flag fields: */ - /* -diff --git a/xen/arch/x86/include/asm/processor.h b/xen/arch/x86/include/asm/processor.h -index 2e087c625770..d2cacdfedb74 100644 ---- a/xen/arch/x86/include/asm/processor.h -+++ b/xen/arch/x86/include/asm/processor.h -@@ -328,6 +328,9 @@ DECLARE_PER_CPU(struct tss_page, tss_page); - - DECLARE_PER_CPU(root_pgentry_t *, root_pgt); - -+/* vCPU of the currently loaded page-tables. */ -+DECLARE_PER_CPU(struct vcpu *, pgtable_vcpu); -+ - extern void write_ptbase(struct vcpu *v); - - /* PAUSE (encoding: REP NOP) is a good thing to insert into busy-wait loops. */ -diff --git a/xen/arch/x86/mm.c b/xen/arch/x86/mm.c -index 2b23bf2e7a75..d02c9862d387 100644 ---- a/xen/arch/x86/mm.c -+++ b/xen/arch/x86/mm.c -@@ -535,7 +535,7 @@ void write_ptbase(struct vcpu *v) - cpu_info->pv_cr3 = __pa(this_cpu(root_pgt)); - if ( new_cr4 & X86_CR4_PCIDE ) - cpu_info->pv_cr3 |= get_pcid_bits(v, true); -- switch_cr3_cr4(v->arch.cr3, new_cr4); -+ switch_cr3_cr4(v, v->arch.cr3, new_cr4); - } - else - { -@@ -543,7 +543,7 @@ void write_ptbase(struct vcpu *v) - cpu_info->use_pv_cr3 = false; - cpu_info->xen_cr3 = 0; - /* switch_cr3_cr4() serializes. */ -- switch_cr3_cr4(v->arch.cr3, new_cr4); -+ switch_cr3_cr4(v, v->arch.cr3, new_cr4); - cpu_info->pv_cr3 = 0; - } - } -diff --git a/xen/arch/x86/pv/dom0_build.c b/xen/arch/x86/pv/dom0_build.c -index 37729091dfaa..42bc530c0f0d 100644 ---- a/xen/arch/x86/pv/dom0_build.c -+++ b/xen/arch/x86/pv/dom0_build.c -@@ -828,8 +828,7 @@ static int __init dom0_construct(const struct boot_domain *bd) - update_cr3(v); - - /* We run on dom0's page tables for the final part of the build process. */ -- switch_cr3_cr4(cr3_pa(v->arch.cr3), read_cr4()); -- mapcache_override_current(v); -+ switch_cr3_cr4(v, cr3_pa(v->arch.cr3), read_cr4()); - - /* Copy the OS image and free temporary buffer. */ - elf.dest_base = (void*)vkern_start; -@@ -838,8 +837,7 @@ static int __init dom0_construct(const struct boot_domain *bd) - rc = elf_load_binary(&elf); - if ( rc < 0 ) - { -- mapcache_override_current(NULL); -- switch_cr3_cr4(current->arch.cr3, read_cr4()); -+ switch_cr3_cr4(current, current->arch.cr3, read_cr4()); - printk("Failed to load the kernel binary\n"); - goto out; - } -@@ -850,8 +848,7 @@ static int __init dom0_construct(const struct boot_domain *bd) - if ( (parms.virt_hypercall < v_start) || - (parms.virt_hypercall >= v_end) ) - { -- mapcache_override_current(NULL); -- switch_cr3_cr4(current->arch.cr3, read_cr4()); -+ switch_cr3_cr4(current, current->arch.cr3, read_cr4()); - printk("Invalid HYPERCALL_PAGE field in ELF notes.\n"); - return -EINVAL; - } -@@ -992,8 +989,7 @@ static int __init dom0_construct(const struct boot_domain *bd) - #endif - - /* Return to idle domain's page tables. */ -- mapcache_override_current(NULL); -- switch_cr3_cr4(current->arch.cr3, read_cr4()); -+ switch_cr3_cr4(current, current->arch.cr3, read_cr4()); - - update_domain_wallclock_time(d); - -diff --git a/xen/arch/x86/pv/domain.c b/xen/arch/x86/pv/domain.c -index ef4f442e7332..d9e52f5f88f3 100644 ---- a/xen/arch/x86/pv/domain.c -+++ b/xen/arch/x86/pv/domain.c -@@ -451,6 +451,8 @@ static void _toggle_guest_pt(struct vcpu *v) - pagetable_t old_shadow; - unsigned long cr3; - -+ ASSERT(local_irq_is_enabled()); -+ - v->arch.flags ^= TF_kernel_mode; - guest_update = v->arch.flags & TF_kernel_mode; - old_shadow = update_cr3(v); -@@ -473,15 +475,22 @@ static void _toggle_guest_pt(struct vcpu *v) - { - cr3 &= ~X86_CR3_NOFLUSH; - -+ local_irq_disable(); - if ( unlikely(mfn_eq(pagetable_get_mfn(old_shadow), - maddr_to_mfn(cr3))) ) - { -- cr3 = idle_vcpu[v->processor]->arch.cr3; - /* Also suppress runstate/time area updates below. */ - guest_update = false; -+ -+ cr3 = idle_vcpu[v->processor]->arch.cr3; -+ this_cpu(pgtable_vcpu) = idle_vcpu[v->processor]; - } -+ -+ write_cr3(cr3); -+ local_irq_enable(); - } -- write_cr3(cr3); -+ else -+ write_cr3(cr3); - - if ( !pagetable_is_null(old_shadow) ) - shadow_put_top_level(v->domain, old_shadow); -diff --git a/xen/arch/x86/smpboot.c b/xen/arch/x86/smpboot.c -index 27628800a821..b37feab3bef4 100644 ---- a/xen/arch/x86/smpboot.c -+++ b/xen/arch/x86/smpboot.c -@@ -1063,6 +1063,7 @@ static int cpu_smpboot_alloc(unsigned int cpu) - - info->current_vcpu = idle_vcpu[cpu]; /* set_current() */ - per_cpu(curr_vcpu, cpu) = idle_vcpu[cpu]; -+ per_cpu(pgtable_vcpu, cpu) = idle_vcpu[cpu]; - - gdt = per_cpu(gdt, cpu) ?: alloc_xenheap_pages(0, memflags); - if ( gdt == NULL ) -diff --git a/xen/common/efi/common-stub.c b/xen/common/efi/common-stub.c -index 77f138a6c574..7b12005bea3f 100644 ---- a/xen/common/efi/common-stub.c -+++ b/xen/common/efi/common-stub.c -@@ -7,11 +7,6 @@ bool efi_enabled(unsigned int feature) - return false; - } - --bool efi_rs_using_pgtables(void) --{ -- return false; --} -- - unsigned long efi_get_time(void) - { - BUG(); -diff --git a/xen/common/efi/runtime.c b/xen/common/efi/runtime.c -index 30d649ca5c1b..feb09acf754c 100644 ---- a/xen/common/efi/runtime.c -+++ b/xen/common/efi/runtime.c -@@ -49,7 +49,6 @@ const CHAR16 *__read_mostly efi_fw_vendor; - const EFI_RUNTIME_SERVICES *__read_mostly efi_rs; - #ifndef CONFIG_ARM /* TODO - disabled until implemented on ARM */ - static DEFINE_SPINLOCK(efi_rs_lock); --static unsigned int efi_rs_on_cpu = NR_CPUS; - #endif - - UINTN __read_mostly efi_memmap_size; -@@ -92,6 +91,11 @@ struct efi_rs_state efi_rs_enter(void) - if ( mfn_eq(efi_l4_mfn, INVALID_MFN) ) - return state; - -+ /* -+ * If in lazy idle context switch state sync now to avoid an incoming -+ * FLUSH_VCPU_STATE IPI changing the loaded page-tables. -+ */ -+ sync_local_execstate(); - state.cr3 = read_cr3(); - save_fpu_enable(); - asm volatile ( "fnclex; fldcw %0" :: "m" (fcw) ); -@@ -99,8 +103,6 @@ struct efi_rs_state efi_rs_enter(void) - - spin_lock(&efi_rs_lock); - -- efi_rs_on_cpu = smp_processor_id(); -- - /* prevent fixup_page_fault() from doing anything */ - irq_enter(); - -@@ -115,7 +117,8 @@ struct efi_rs_state efi_rs_enter(void) - lgdt(&gdt_desc); - } - -- switch_cr3_cr4(mfn_to_maddr(efi_l4_mfn), read_cr4()); -+ switch_cr3_cr4(idle_vcpu[smp_processor_id()], mfn_to_maddr(efi_l4_mfn), -+ read_cr4()); - - /* - * At the time of writing (2022), no UEFI firwmare is CET-IBT compatible. -@@ -143,7 +146,7 @@ void efi_rs_leave(struct efi_rs_state *state) - if ( state->msr_s_cet ) - wrmsrl(MSR_S_CET, state->msr_s_cet); - -- switch_cr3_cr4(state->cr3, read_cr4()); -+ switch_cr3_cr4(curr, state->cr3, read_cr4()); - if ( is_pv_vcpu(curr) && !is_idle_vcpu(curr) ) - { - struct desc_ptr gdt_desc = { -@@ -154,18 +157,10 @@ void efi_rs_leave(struct efi_rs_state *state) - lgdt(&gdt_desc); - } - irq_exit(); -- efi_rs_on_cpu = NR_CPUS; - spin_unlock(&efi_rs_lock); - vcpu_restore_fpu_nonlazy(curr, true); - } - --bool efi_rs_using_pgtables(void) --{ -- return !mfn_eq(efi_l4_mfn, INVALID_MFN) && -- (smp_processor_id() == efi_rs_on_cpu) && -- (read_cr3() == mfn_to_maddr(efi_l4_mfn)); --} -- - unsigned long efi_get_time(void) - { - EFI_TIME time; -diff --git a/xen/include/xen/efi.h b/xen/include/xen/efi.h -index 723cb8085270..9953197ee553 100644 ---- a/xen/include/xen/efi.h -+++ b/xen/include/xen/efi.h -@@ -40,7 +40,6 @@ extern bool efi_secure_boot; - - void efi_init_memory(void); - bool efi_boot_mem_unused(unsigned long *start, unsigned long *end); --bool efi_rs_using_pgtables(void); - unsigned long efi_get_time(void); - void efi_halt_system(void); - void efi_reset_system(bool warm); --- -2.53.0 -