diff --git a/messages/tape_linux_sg/root.txt b/messages/tape_linux_sg/root.txt index 981fdc1b..e4f11b3e 100644 --- a/messages/tape_linux_sg/root.txt +++ b/messages/tape_linux_sg/root.txt @@ -138,6 +138,7 @@ root:table { 30295I:string { "Have unstable TUR response, start over (Cur = %d, Prev = %d)." } 30296I:string { "Capturing a stable TUR at line %d." } 30297W:string { "Cannot retrieve drive dump: failed to communicate with drive. Tried (%d) times." } + 30298W:string { "Retrying write operation due Power On-Reset event reached. Tried (%d) times." } 30392D:string { "Backend %s %s." } 30393D:string { "Backend %s: %d %s." } diff --git a/src/tape_drivers/linux/lin_tape/lin_tape_ibmtape.c b/src/tape_drivers/linux/lin_tape/lin_tape_ibmtape.c index f67e54e9..edc420db 100644 --- a/src/tape_drivers/linux/lin_tape/lin_tape_ibmtape.c +++ b/src/tape_drivers/linux/lin_tape/lin_tape_ibmtape.c @@ -1439,15 +1439,11 @@ int lin_tape_ibmtape_read(void *device, char *buf, size_t count, struct tc_posit #define WRITE_RETRY (-LINUX_MAX_BLOCK_SIZE) -static inline int _handle_block_allocation_failure(void *device, struct tc_position *pos, int *retry) +static inline int _resolve_position_after_io_cmd_failure(void *device, struct tc_position *pos) { int ret = 0; struct tc_position tmp_pos = {0, 0}; - /* Sleep 3 secs to wait garbage correction in kernel side and retry */ - ltfsmsg(LTFS_WARN, 30440W, ++(*retry)); - sleep(3); - ret = lin_tape_ibmtape_readpos(device, &tmp_pos); if (ret == DEVICE_GOOD && pos->partition == tmp_pos.partition) { if (pos->block == tmp_pos.block) { @@ -1547,7 +1543,12 @@ int lin_tape_ibmtape_write(void *device, const char *buf, size_t count, struct t rc = DEVICE_GOOD; } } else if (errno == ENOMEM && retry < MAX_WRITE_RETRY) { - rc = _handle_block_allocation_failure(device, pos, &retry); + ltfsmsg(LTFS_WARN, 30440W, ++retry); + struct timespec delay_ts; + delay_ts.tv_sec = 3; + delay_ts.tv_nsec = 0; + nanosleep(&delay_ts, NULL); // Wait for kernel GC + rc = _resolve_position_after_io_cmd_failure(device, pos); if (rc == WRITE_RETRY) { errno = 0; goto write_start; @@ -1573,7 +1574,11 @@ int lin_tape_ibmtape_write(void *device, const char *buf, size_t count, struct t if (retry < MAX_WRITE_RETRY && ((current_errno == EIO && rc == -EDEV_NO_SENSE ) || (rc == -EDEV_CONFIGURE_CHANGED) || (rc == -EDEV_TIME_STAMP_CHANGED))) { - rc = _handle_block_allocation_failure(device, pos, &retry); + struct timespec delay_ts; + delay_ts.tv_sec = 5; + delay_ts.tv_nsec = 0; + nanosleep(&delay_ts, NULL); + rc = _resolve_position_after_io_cmd_failure(device, pos); if (rc == WRITE_RETRY) { errno = 0; goto write_start; diff --git a/src/tape_drivers/linux/sg/sg_tape.c b/src/tape_drivers/linux/sg/sg_tape.c index 8a13493e..3d43e897 100644 --- a/src/tape_drivers/linux/sg/sg_tape.c +++ b/src/tape_drivers/linux/sg/sg_tape.c @@ -54,6 +54,7 @@ #include #include +#include "libltfs/ltfs_error.h" #include "ltfs_copyright.h" #include "libltfs/ltfslogging.h" #include "libltfs/fs.h" @@ -104,6 +105,7 @@ struct sg_global_data global_data; #define MAX_RETRY (100) #define MAX_TAKE_DUMP_ATTEMPTS (10) +#define POR_MAX_RETRIES (3) /* Forward references (For keep function order to struct tape_ops) */ int sg_readpos(void *device, struct tc_position *pos); @@ -604,7 +606,7 @@ int _raw_tur(const int fd) #define _clear_por(p) _clear_por_raw((p)->dev.fd); -void _clear_por_raw(const int fd) +int _clear_por_raw(const int fd) { int i = 0, ret = -1; @@ -624,6 +626,7 @@ void _clear_por_raw(const int fd) } i++; } + return ret; } #define _get_stable_tur_response(p) _get_stable_tur_response_raw((p)->dev.fd) @@ -1861,16 +1864,11 @@ static int _cdb_read(void *device, char *buf, size_t size, bool sili) return length; } -static inline int _handle_block_allocation_failure(void *device, struct tc_position *pos, - int *retry, char *op) +static inline int _resolve_position_after_io_cmd_failure(void *device, struct tc_position *pos, char *op) { int ret = 0; struct tc_position tmp_pos = {0, 0}; - /* Sleep 3 secs to wait garbage correction in kernel side and retry */ - ltfsmsg(LTFS_WARN, 30277W, ++(*retry)); - sleep(3); - ret = sg_readpos(device, &tmp_pos); if (ret == DEVICE_GOOD && pos->partition == tmp_pos.partition) { if (pos->block == tmp_pos.block) { @@ -1984,7 +1982,12 @@ int sg_read(void *device, char *buf, size_t size, priv->use_sili = false; ret = _cdb_read(device, buf, datacount, unusual_size); } else if (ret == -EDEV_BUFFER_ALLOCATE_ERROR && retry_count < MAX_RETRY) { - ret = _handle_block_allocation_failure(device, pos, &retry_count, "read"); + ltfsmsg(LTFS_WARN, 30277W, ++retry_count); + struct timespec delay_ts; + delay_ts.tv_sec = 3; + delay_ts.tv_nsec = 0; + nanosleep(&delay_ts, NULL); // Wait for kernel GC + ret = _resolve_position_after_io_cmd_failure(device, pos, "read"); if (ret == -EDEV_RETRY) goto start_read; } @@ -2093,12 +2096,13 @@ static int _cdb_write(void *device, uint8_t *buf, size_t size, bool *ew, bool *p int sg_write(void *device, const char *buf, size_t count, struct tc_position *pos) { - int ret, ret_fo; + int ret, ret_fo, ret_write = -1; bool ew = false, pew = false; struct sg_data *priv = (struct sg_data*)device; struct tc_position cur_pos; size_t datacount = count; - int retry_count = 0; + int retry_count = 0, por_retry_count = 0; + struct timespec delay_ts = {0}; ltfs_profiler_add_entry(priv->profiler, NULL, TAPEBEND_REQ_ENTER(REQ_TC_WRITE)); @@ -2128,7 +2132,7 @@ int sg_write(void *device, const char *buf, size_t count, struct tc_position *po } start_write: - ret = _cdb_write(device, (uint8_t *)buf, datacount, &ew, &pew); + ret_write = ret = _cdb_write(device, (uint8_t *)buf, datacount, &ew, &pew); if (ret == DEVICE_GOOD) { pos->block++; pos->early_warning = ew; @@ -2146,9 +2150,30 @@ int sg_write(void *device, const char *buf, size_t count, struct tc_position *po ret = -EDEV_POR_OR_BUS_RESET; } } else if (ret == -EDEV_BUFFER_ALLOCATE_ERROR && retry_count < MAX_RETRY) { - ret = _handle_block_allocation_failure(device, pos, &retry_count, "write"); + ltfsmsg(LTFS_WARN, 30277W, ++retry_count); + delay_ts.tv_sec = 3; + delay_ts.tv_nsec = 0; + nanosleep(&delay_ts, NULL); // Wait for kernel GC + ret = _resolve_position_after_io_cmd_failure(device, pos, "write"); if (ret == -EDEV_RETRY) goto start_write; + } else if (ret == -EDEV_HOST_ERROR && por_retry_count < POR_MAX_RETRIES) { + por_retry_count++; + delay_ts.tv_sec = 5; + delay_ts.tv_nsec = 0; + nanosleep(&delay_ts, NULL); + ret = _clear_por(priv); + if (ret == DEVICE_GOOD) { + int handle_ret = _resolve_position_after_io_cmd_failure(device, pos, "write"); + /* If the original command did not reach the driver, or it reached it but after failing there is a position mismatch; retry */ + if (handle_ret == -EDEV_RETRY) { + ltfsmsg(LTFS_WARN, 30298W, por_retry_count); + goto start_write; + } + } else { + // If we could not clear the POR status, just return the _cdb_write() return value + ret = ret_write; + } } ltfs_profiler_add_entry(priv->profiler, NULL, TAPEBEND_REQ_EXIT(REQ_TC_WRITE)); diff --git a/src/tape_drivers/netbsd/scsipi-ibmtape/scsipi_ibmtape.c b/src/tape_drivers/netbsd/scsipi-ibmtape/scsipi_ibmtape.c index 0ea8df55..8cbe3000 100644 --- a/src/tape_drivers/netbsd/scsipi-ibmtape/scsipi_ibmtape.c +++ b/src/tape_drivers/netbsd/scsipi-ibmtape/scsipi_ibmtape.c @@ -96,6 +96,7 @@ struct scsipi_ibmtape_global_data global_data; #define TU_DEFAULT_TIMEOUT (60) #define MAX_RETRY (100) +#define POR_MAX_RETRIES (3) /* Forward references (For keep function order to struct tape_ops) */ int scsipi_ibmtape_readpos(void *device, struct tc_position *pos); @@ -569,7 +570,7 @@ int _raw_tur(const int fd) #define _clear_por(p) _clear_por_raw((p)->dev.fd); -void _clear_por_raw(const int fd) +int _clear_por_raw(const int fd) { int i = 0, ret = -1; @@ -589,6 +590,7 @@ void _clear_por_raw(const int fd) } i++; } + return ret; } /* Forward reference */ @@ -1425,16 +1427,11 @@ static int _cdb_read(void *device, char *buf, size_t size, bool sili) return length; } -static inline int _handle_block_allocation_failure(void *device, struct tc_position *pos, - int *retry, char *op) +static inline int _resolve_position_after_io_cmd_failure(void *device, struct tc_position *pos, char *op) { int ret = 0; struct tc_position tmp_pos = {0, 0}; - /* Sleep 3 secs to wait garbage correction in kernel side and retry */ - ltfsmsg(LTFS_WARN, 30277W, ++(*retry)); - sleep(3); - ret = scsipi_ibmtape_readpos(device, &tmp_pos); if (ret == DEVICE_GOOD && pos->partition == tmp_pos.partition) { if (pos->block == tmp_pos.block) { @@ -1548,7 +1545,12 @@ int scsipi_ibmtape_read(void *device, char *buf, size_t size, priv->use_sili = false; ret = _cdb_read(device, buf, datacount, unusual_size); } else if (ret == -EDEV_BUFFER_ALLOCATE_ERROR && retry_count < MAX_RETRY) { - ret = _handle_block_allocation_failure(device, pos, &retry_count, "read"); + ltfsmsg(LTFS_WARN, 30277W, ++(*retry)); + struct timespec delay_ts; + delay_ts.tv_sec = 3; + delay_ts.tv_nsec = 0; + nanosleep(&delay_ts, NULL); // Wait for kernel GC + ret = _resolve_position_after_io_cmd_failure(device, pos, "read"); if (ret == -EDEV_RETRY) goto start_read; } @@ -1651,12 +1653,12 @@ static int _cdb_write(void *device, uint8_t *buf, size_t size, bool *ew, bool *p int scsipi_ibmtape_write(void *device, const char *buf, size_t count, struct tc_position *pos) { - int ret, ret_fo; + int ret, ret_fo, ret_write = -1; bool ew = false, pew = false; struct scsipi_ibmtape_data *priv = (struct scsipi_ibmtape_data*)device; struct tc_position cur_pos; size_t datacount = count; - int retry_count = 0; + int retry_count = 0, por_retry_count = 0; ltfs_profiler_add_entry(priv->profiler, NULL, TAPEBEND_REQ_ENTER(REQ_TC_WRITE)); @@ -1686,7 +1688,7 @@ int scsipi_ibmtape_write(void *device, const char *buf, size_t count, struct tc_ } start_write: - ret = _cdb_write(device, (uint8_t *)buf, datacount, &ew, &pew); + ret_write = ret = _cdb_write(device, (uint8_t *)buf, datacount, &ew, &pew); if (ret == DEVICE_GOOD) { pos->block++; pos->early_warning = ew; @@ -1704,9 +1706,32 @@ int scsipi_ibmtape_write(void *device, const char *buf, size_t count, struct tc_ ret = -EDEV_POR_OR_BUS_RESET; } } else if (ret == -EDEV_BUFFER_ALLOCATE_ERROR && retry_count < MAX_RETRY) { - ret = _handle_block_allocation_failure(device, pos, &retry_count, "write"); + ltfsmsg(LTFS_WARN, 30277W, ++(*retry)); + struct timespec delay_ts; + delay_ts.tv_sec = 3; + delay_ts.tv_nsec = 0; + nanosleep(&delay_ts, NULL); // Wait for kernel GC + ret = _resolve_position_after_io_cmd_failure(device, pos, "write"); if (ret == -EDEV_RETRY) goto start_write; + } else if (ret == -EDEV_HOST_ERROR && por_retry_count < POR_MAX_RETRIES) { + por_retry_count++; + struct timespec delay_ts; + delay_ts.tv_sec = 5; + delay_ts.tv_nsec = 0; + nanosleep(&delay_ts, NULL); + ret = _clear_por(priv); + if (ret == DEVICE_GOOD) { + int handle_ret = _resolve_position_after_io_cmd_failure(device, pos, "write"); + // If the original command did not reach the driver, or it reached it but after failing there is block mismatch; retry + if (handle_ret == -EDEV_RETRY) { + ltfsmsg(LTFS_WARN, 30298W, por_retry_count); + goto start_write; + } + } else { + // If we could not clear the POR status, just return the _cdb_write() return value + ret = ret_write; + } } ltfs_profiler_add_entry(priv->profiler, NULL, TAPEBEND_REQ_EXIT(REQ_TC_WRITE));