93729c1796
configure sa(4) to request no I/O splitting by default. For tape devices, the user needs to be able to clearly understand what blocksize is actually being used when writing to a tape device. The previous behavior of physio(9) was that it would split up any I/O that was too large for the device, or too large to fit into MAXPHYS. This means that if, for instance, the user wrote a 1MB block to a tape device, and MAXPHYS was 128KB, the 1MB write would be split into 8 128K chunks. This would be done without informing the user. This has suboptimal effects, especially when trying to communicate status to the user. In the event of an error writing to a tape (e.g. physical end of tape) in the middle of a 1MB block that has been split into 8 pieces, the user could have the first two 128K pieces written successfully, the third returned with an error, and the last 5 returned with 0 bytes written. If the user is using a standard write(2) system call, all he will see is the ENOSPC error. He won't have a clue how much actually got written. (With a writev(2) system call, he should be able to determine how much got written in addition to the error.) The solution is to prevent physio(9) from splitting the I/O. The new cdev flag, SI_NOSPLIT, tells physio that the driver does not want I/O to be split beforehand. Although the sa(4) driver now enables SI_NOSPLIT by default, that can be disabled by two loader tunables for now. It will not be configurable starting in FreeBSD 11.0. kern.cam.sa.allow_io_split allows the user to configure I/O splitting for all sa(4) driver instances. kern.cam.sa.%d.allow_io_split allows the user to configure I/O splitting for a specific sa(4) instance. There are also now three sa(4) driver sysctl variables that let the users see some sa(4) driver values. kern.cam.sa.%d.allow_io_split shows whether I/O splitting is turned on. kern.cam.sa.%d.maxio shows the maximum I/O size allowed by kernel configuration parameters (e.g. MAXPHYS, DFLTPHYS) and the capabilities of the controller. kern.cam.sa.%d.cpi_maxio shows the maximum I/O size supported by the controller. Note that a better long term solution would be to implement support for chaining buffers, so that that MAXPHYS is no longer a limiting factor for I/O size to tape and disk devices. At that point, the controller and the tape drive would become the limiting factors. sys/conf.h: Add a new cdev flag, SI_NOSPLIT, that allows a driver to tell physio not to split up I/O. sys/param.h: Bump __FreeBSD_version to 1000049 for the addition of the SI_NOSPLIT cdev flag. kern_physio.c: If the SI_NOSPLIT flag is set on the cdev, return any I/O that is larger than si_iosize_max or MAXPHYS, has more than one segment, or would have to be split because of misalignment with EFBIG. (File too large). In the event of an error, print a console message to give the user a clue about what happened. scsi_sa.c: Set the SI_NOSPLIT cdev flag on the devices created for the sa(4) driver by default. Add tunables to control whether we allow I/O splitting in physio(9). Explain in the comments that allowing I/O splitting will be deprecated for the sa(4) driver in FreeBSD 11.0. Add sysctl variables to display the maximum I/O size we can do (which could be further limited by read block limits) and the maximum I/O size that the controller can do. Limit our maximum I/O size (recorded in the cdev's si_iosize_max) by MAXPHYS. This isn't strictly necessary, because physio(9) will limit it to MAXPHYS, but it will provide some clarity for the application. Record the controller's maximum I/O size reported in the Path Inquiry CCB. sa.4: Document the block size behavior, and explain that the option of allowing physio(9) to split the I/O will disappear in FreeBSD 11.0. Sponsored by: Spectra Logic
174 lines
4.6 KiB
C
174 lines
4.6 KiB
C
/*-
|
|
* Copyright (c) 1994 John S. Dyson
|
|
* All rights reserved.
|
|
*
|
|
* Redistribution and use in source and binary forms, with or without
|
|
* modification, are permitted provided that the following conditions
|
|
* are met:
|
|
* 1. Redistributions of source code must retain the above copyright
|
|
* notice immediately at the beginning of the file, without modification,
|
|
* this list of conditions, and the following disclaimer.
|
|
* 2. Redistributions in binary form must reproduce the above copyright
|
|
* notice, this list of conditions and the following disclaimer in the
|
|
* documentation and/or other materials provided with the distribution.
|
|
* 3. Absolutely no warranty of function or purpose is made by the author
|
|
* John S. Dyson.
|
|
* 4. Modifications may be freely made to this file if the above conditions
|
|
* are met.
|
|
*/
|
|
|
|
#include <sys/cdefs.h>
|
|
__FBSDID("$FreeBSD$");
|
|
|
|
#include <sys/param.h>
|
|
#include <sys/systm.h>
|
|
#include <sys/bio.h>
|
|
#include <sys/buf.h>
|
|
#include <sys/conf.h>
|
|
#include <sys/proc.h>
|
|
#include <sys/uio.h>
|
|
|
|
#include <vm/vm.h>
|
|
#include <vm/vm_extern.h>
|
|
|
|
int
|
|
physio(struct cdev *dev, struct uio *uio, int ioflag)
|
|
{
|
|
struct buf *bp;
|
|
struct cdevsw *csw;
|
|
caddr_t sa;
|
|
u_int iolen;
|
|
int error, i, mapped;
|
|
|
|
/* Keep the process UPAGES from being swapped. XXX: why ? */
|
|
PHOLD(curproc);
|
|
|
|
bp = getpbuf(NULL);
|
|
sa = bp->b_data;
|
|
error = 0;
|
|
|
|
/* XXX: sanity check */
|
|
if(dev->si_iosize_max < PAGE_SIZE) {
|
|
printf("WARNING: %s si_iosize_max=%d, using DFLTPHYS.\n",
|
|
devtoname(dev), dev->si_iosize_max);
|
|
dev->si_iosize_max = DFLTPHYS;
|
|
}
|
|
|
|
/*
|
|
* If the driver does not want I/O to be split, that means that we
|
|
* need to reject any requests that will not fit into one buffer.
|
|
*/
|
|
if ((dev->si_flags & SI_NOSPLIT) &&
|
|
((uio->uio_resid > dev->si_iosize_max) ||
|
|
(uio->uio_resid > MAXPHYS) ||
|
|
(uio->uio_iovcnt > 1))) {
|
|
/*
|
|
* Tell the user why his I/O was rejected.
|
|
*/
|
|
if (uio->uio_resid > dev->si_iosize_max)
|
|
printf("%s: request size %zd > si_iosize_max=%d, "
|
|
"cannot split request\n", devtoname(dev),
|
|
uio->uio_resid, dev->si_iosize_max);
|
|
|
|
if (uio->uio_resid > MAXPHYS)
|
|
printf("%s: request size %zd > MAXPHYS=%d, "
|
|
"cannot split request\n", devtoname(dev),
|
|
uio->uio_resid, MAXPHYS);
|
|
|
|
if (uio->uio_iovcnt > 1)
|
|
printf("%s: request vectors=%d > 1, "
|
|
"cannot split request\n", devtoname(dev),
|
|
uio->uio_iovcnt);
|
|
|
|
error = EFBIG;
|
|
goto doerror;
|
|
}
|
|
|
|
for (i = 0; i < uio->uio_iovcnt; i++) {
|
|
while (uio->uio_iov[i].iov_len) {
|
|
bp->b_flags = 0;
|
|
if (uio->uio_rw == UIO_READ) {
|
|
bp->b_iocmd = BIO_READ;
|
|
curthread->td_ru.ru_inblock++;
|
|
} else {
|
|
bp->b_iocmd = BIO_WRITE;
|
|
curthread->td_ru.ru_oublock++;
|
|
}
|
|
bp->b_iodone = bdone;
|
|
bp->b_data = uio->uio_iov[i].iov_base;
|
|
bp->b_bcount = uio->uio_iov[i].iov_len;
|
|
bp->b_offset = uio->uio_offset;
|
|
bp->b_iooffset = uio->uio_offset;
|
|
bp->b_saveaddr = sa;
|
|
|
|
/* Don't exceed drivers iosize limit */
|
|
if (bp->b_bcount > dev->si_iosize_max)
|
|
bp->b_bcount = dev->si_iosize_max;
|
|
|
|
/*
|
|
* Make sure the pbuf can map the request
|
|
* XXX: The pbuf has kvasize = MAXPHYS so a request
|
|
* XXX: larger than MAXPHYS - PAGE_SIZE must be
|
|
* XXX: page aligned or it will be fragmented.
|
|
*/
|
|
iolen = ((vm_offset_t) bp->b_data) & PAGE_MASK;
|
|
if ((bp->b_bcount + iolen) > bp->b_kvasize) {
|
|
/*
|
|
* This device does not want I/O to be split.
|
|
*/
|
|
if (dev->si_flags & SI_NOSPLIT) {
|
|
printf("%s: request ptr %#jx is not "
|
|
"on a page boundary, cannot split "
|
|
"request\n", devtoname(dev),
|
|
(uintmax_t)bp->b_data);
|
|
error = EFBIG;
|
|
goto doerror;
|
|
}
|
|
bp->b_bcount = bp->b_kvasize;
|
|
if (iolen != 0)
|
|
bp->b_bcount -= PAGE_SIZE;
|
|
}
|
|
bp->b_bufsize = bp->b_bcount;
|
|
|
|
bp->b_blkno = btodb(bp->b_offset);
|
|
|
|
csw = dev->si_devsw;
|
|
if (uio->uio_segflg == UIO_USERSPACE) {
|
|
if (dev->si_flags & SI_UNMAPPED)
|
|
mapped = 0;
|
|
else
|
|
mapped = 1;
|
|
if (vmapbuf(bp, mapped) < 0) {
|
|
error = EFAULT;
|
|
goto doerror;
|
|
}
|
|
}
|
|
|
|
dev_strategy_csw(dev, csw, bp);
|
|
if (uio->uio_rw == UIO_READ)
|
|
bwait(bp, PRIBIO, "physrd");
|
|
else
|
|
bwait(bp, PRIBIO, "physwr");
|
|
|
|
if (uio->uio_segflg == UIO_USERSPACE)
|
|
vunmapbuf(bp);
|
|
iolen = bp->b_bcount - bp->b_resid;
|
|
if (iolen == 0 && !(bp->b_ioflags & BIO_ERROR))
|
|
goto doerror; /* EOF */
|
|
uio->uio_iov[i].iov_len -= iolen;
|
|
uio->uio_iov[i].iov_base =
|
|
(char *)uio->uio_iov[i].iov_base + iolen;
|
|
uio->uio_resid -= iolen;
|
|
uio->uio_offset += iolen;
|
|
if( bp->b_ioflags & BIO_ERROR) {
|
|
error = bp->b_error;
|
|
goto doerror;
|
|
}
|
|
}
|
|
}
|
|
doerror:
|
|
relpbuf(bp, NULL);
|
|
PRELE(curproc);
|
|
return (error);
|
|
}
|