xv6, line by line
kernel/virtio_disk.c

kernel/virtio_disk.c

C · 333 lines · annotated 100% · kernel · upstream

About this file

The disk driver. xv6’s disk is a virtio block device that QEMU provides, backed by the file fs.img on your computer. This file is the only code in xv6 that talks to it.

It has three entry points:

  • virtio_disk_init, called once from main at boot: resets the device, agrees on features, and sets up the virtqueue (a descriptor table and two rings in shared memory) that carries requests.
  • virtio_disk_rw, called by the buffer cache (bread and bwrite): reads or writes one 1024-byte block. It describes the request in three descriptors, puts it on the available ring, rings the device’s doorbell, and then sleeps until the request is done.
  • virtio_disk_intr, called from devintr when the disk interrupts: finds finished requests on the used ring and wakes the processes waiting for them.

The device copies data straight between the disk and the buffer’s memory (DMA (direct memory access)), so the CPU never copies the block itself. One spinlock, vdisk_lock, protects all of the driver’s state.

Read before: kernel/virtio.h (the register offsets and structures). Read next: kernel/bio.c, the only caller of virtio_disk_rw.

1//
2// driver for qemu's virtio disk device.
3// uses qemu's mmio interface to virtio.
4//
5// qemu ... -drive file=fs.img,if=none,format=raw,id=x0 -device virtio-blk-device,drive=x0,bus=virtio-mmio-bus.0
6//
8#include "types.h"
9#include "riscv.h"
10#include "defs.h"
11#include "param.h"
15#include "fs.h"
16#include "buf.h"
17#include "virtio.h"
19// the address of virtio mmio register r.
20#define R(r) ((volatile uint32 *)(VIRTIO0 + (r)))
22static struct disk {
23 // a set (not a ring) of DMA descriptors, with which the
24 // driver tells the device where to read and write individual
25 // disk operations. there are NUM descriptors.
26 // most commands consist of a "chain" (a linked list) of a couple of
27 // these descriptors.
28 struct virtq_desc *desc;
30 // a ring in which the driver writes descriptor numbers
31 // that the driver would like the device to process. it only
32 // includes the head descriptor of each chain. the ring has
33 // NUM elements.
36 // a ring in which the device writes descriptor numbers that
37 // the device has finished processing (just the head of each chain).
38 // there are NUM used ring entries.
39 struct virtq_used *used;
41 // our own book-keeping.
42 char free[NUM]; // is a descriptor free?
43 uint16 used_idx; // we've looked this far in used[2..NUM].
45 // track info about in-flight operations,
46 // for use when completion interrupt arrives.
47 // indexed by first descriptor index of chain.
48 struct {
49 struct buf *b;
50 char status;
51 } info[NUM];
53 // disk command headers.
54 // one-for-one with descriptors, for convenience.
61void
66 initlock(&disk.vdisk_lock, "virtio_disk");
68 if (*R(VIRTIO_MMIO_MAGIC_VALUE) != 0x74726976 ||
70 *R(VIRTIO_MMIO_VENDOR_ID) != 0x554d4551) {
71 panic("could not find virtio disk");
72 }
74 // reset device
77 // set ACKNOWLEDGE status bit
81 // set DRIVER status bit
85 // negotiate features
97 // tell device that feature negotiation is complete.
101 // re-read status to ensure FEATURES_OK is set.
104 panic("virtio disk FEATURES_OK unset");
106 // initialize queue 0.
109 // ensure queue 0 is not in use.
111 panic("virtio disk should not be ready");
113 // check maximum queue size.
115 if (max == 0)
116 panic("virtio disk has no queue 0");
117 if (max < NUM)
118 panic("virtio disk max queue too short");
120 // allocate and zero queue memory.
124 if (!disk.desc || !disk.avail || !disk.used)
125 panic("virtio disk kalloc");
130 // set queue size.
133 // write physical addresses.
141 // queue is ready.
144 // all NUM descriptors start out unused.
145 for (int i = 0; i < NUM; i++)
146 disk.free[i] = 1;
148 // tell device we're completely ready.
152 // plic.c and trap.c arrange for interrupts from VIRTIO0_IRQ.
155// find a free descriptor, mark it non-free, return its index.
156static int
159 for (int i = 0; i < NUM; i++) {
160 if (disk.free[i]) {
161 disk.free[i] = 0;
162 return i;
163 }
164 }
165 return -1;
168// mark a descriptor as free.
169static void
172 if (i >= NUM)
173 panic("free_desc 1");
174 if (disk.free[i])
175 panic("free_desc 2");
177 disk.desc[i].len = 0;
180 disk.free[i] = 1;
184// free a chain of descriptors.
185static void
188 while (1) {
190 int nxt = disk.desc[i].next;
193 i = nxt;
194 else
195 break;
196 }
199// allocate three descriptors (they need not be contiguous).
200// disk transfers always use three descriptors.
201static int
204 for (int i = 0; i < 3; i++) {
206 if (idx[i] < 0) {
207 for (int j = 0; j < i; j++)
209 return -1;
210 }
211 }
212 return 0;
215void
218 uint64 sector = b->blockno * (BSIZE / 512);
222 // the spec's Section 5.2 says that legacy block operations use
223 // three descriptors: one for type/reserved/sector, one for the
224 // data, one for a 1-byte status result.
226 // allocate the three descriptors.
227 int idx[3];
228 while (1) {
229 if (alloc3_desc(idx) == 0) {
230 break;
231 }
236 }
238 // format the three descriptors.
239 // qemu's virtio-blk.c reads them.
241 struct virtio_blk_req *buf0 = &disk.ops[idx[0]];
243 if (write)
244 buf0->type = VIRTIO_BLK_T_OUT; // write the disk
245 else
246 buf0->type = VIRTIO_BLK_T_IN; // read the disk
251 disk.desc[idx[0]].len = sizeof(struct virtio_blk_req);
253 disk.desc[idx[0]].next = idx[1];
257 if (write)
258 disk.desc[idx[1]].flags = 0; // device reads b->data
259 else
260 disk.desc[idx[1]].flags = VRING_DESC_F_WRITE; // device writes b->data
262 disk.desc[idx[1]].next = idx[2];
264 disk.info[idx[0]].status = 0xff; // device writes 0 on success
266 disk.desc[idx[2]].len = 1;
267 disk.desc[idx[2]].flags = VRING_DESC_F_WRITE; // device writes the status
268 disk.desc[idx[2]].next = 0;
270 // record struct buf for virtio_disk_intr().
271 b->disk = 1;
272 disk.info[idx[0]].b = b;
274 // tell the device the first index in our chain of descriptors.
279 // tell the device another avail ring entry is available.
280 disk.avail->idx += 1; // not % NUM ...
284 *R(VIRTIO_MMIO_QUEUE_NOTIFY) = 0; // value is queue number
286 // Wait for virtio_disk_intr() to say request has finished.
287 while (b->disk == 1) {
292 }
294 disk.info[idx[0]].b = 0;
300void
305 // the device won't raise another interrupt until we tell it
306 // we've seen this interrupt, which the following line does.
307 // this may race with the device writing new entries to
308 // the "used" ring, in which case we may process the new
309 // completion entries in this interrupt, and have nothing to do
310 // in the next interrupt, which is harmless.
315 // the device increments disk.used->idx when it
316 // adds an entry to the used ring.
318 while (disk.used_idx != disk.used->idx) {
322 if (disk.info[id].status != 0)
323 panic("virtio_disk_intr status");
325 struct buf *b = disk.info[id].b;
326 b->disk = 0; // disk is done with buf
330 }