From a355bc9994fd519e737214fea831c034504dc12d Mon Sep 17 00:00:00 2001 From: Jorge Acosta Date: Mon, 27 Jul 2026 16:19:54 -0600 Subject: [PATCH 1/8] fix: Retrying write operations only after repositioning the drive. fix: Fixing compilation fix: Retrying write operations only after repositioning the drive --- src/tape_drivers/linux/sg/sg_tape.c | 11 +++++++++-- 1 file changed, 9 insertions(+), 2 deletions(-) diff --git a/src/tape_drivers/linux/sg/sg_tape.c b/src/tape_drivers/linux/sg/sg_tape.c index 8a13493e..edcf0988 100644 --- a/src/tape_drivers/linux/sg/sg_tape.c +++ b/src/tape_drivers/linux/sg/sg_tape.c @@ -104,6 +104,7 @@ struct sg_global_data global_data; #define MAX_RETRY (100) #define MAX_TAKE_DUMP_ATTEMPTS (10) +#define SOFT_ERROR_MAX_RETRIES (3) /* Forward references (For keep function order to struct tape_ops) */ int sg_readpos(void *device, struct tc_position *pos); @@ -2098,7 +2099,8 @@ int sg_write(void *device, const char *buf, size_t count, struct tc_position *po struct sg_data *priv = (struct sg_data*)device; struct tc_position cur_pos; size_t datacount = count; - int retry_count = 0; + int reconnect_retry_count = 0, soft_error_retry_count = 0; + int TBL_SLEEP_SECS[SOFT_ERROR_MAX_RETRIES] = {45, 60, 75}; // Hardcoded exponentially increasing sleep time ltfs_profiler_add_entry(priv->profiler, NULL, TAPEBEND_REQ_ENTER(REQ_TC_WRITE)); @@ -2145,7 +2147,12 @@ int sg_write(void *device, const char *buf, size_t count, struct tc_position *po } else ret = -EDEV_POR_OR_BUS_RESET; } - } else if (ret == -EDEV_BUFFER_ALLOCATE_ERROR && retry_count < MAX_RETRY) { + } else if (ret == -EDEV_BUFFER_ALLOCATE_ERROR && reconnect_retry_count < MAX_RETRY) { + ret = _handle_block_allocation_failure(device, pos, &reconnect_retry_count, "write"); + if (ret == -EDEV_RETRY) + goto start_write; + } else if (ret == -EDEV_HOST_ERROR && soft_error_retry_count < SOFT_ERROR_MAX_RETRIES) { + sleep(TBL_SLEEP_SECS[soft_error_retry_count]); ret = _handle_block_allocation_failure(device, pos, &retry_count, "write"); if (ret == -EDEV_RETRY) goto start_write; From ac41205e0b4899c3cced6043d81df56301c29580 Mon Sep 17 00:00:00 2001 From: mcardenas Date: Wed, 29 Jul 2026 16:16:11 -0600 Subject: [PATCH 2/8] fix: retry operations _clear_por changes - clear the previous power on reset status and retry - retry in both ibmtape netbsd driver and the sg one - make _clear_por return an int fix: minor changes chore: revert separation of por and allocation issues feat: make block failure generic --- .../linux/lin_tape/lin_tape_ibmtape.c | 13 +++--- src/tape_drivers/linux/sg/sg_tape.c | 40 ++++++++++--------- .../netbsd/scsipi-ibmtape/scsipi_ibmtape.c | 14 ++++++- 3 files changed, 40 insertions(+), 27 deletions(-) diff --git a/src/tape_drivers/linux/lin_tape/lin_tape_ibmtape.c b/src/tape_drivers/linux/lin_tape/lin_tape_ibmtape.c index f67e54e9..546ff702 100644 --- a/src/tape_drivers/linux/lin_tape/lin_tape_ibmtape.c +++ b/src/tape_drivers/linux/lin_tape/lin_tape_ibmtape.c @@ -1439,15 +1439,11 @@ int lin_tape_ibmtape_read(void *device, char *buf, size_t count, struct tc_posit #define WRITE_RETRY (-LINUX_MAX_BLOCK_SIZE) -static inline int _handle_block_allocation_failure(void *device, struct tc_position *pos, int *retry) +static inline int _handle_block_write_failure(void *device, struct tc_position *pos) { int ret = 0; struct tc_position tmp_pos = {0, 0}; - /* Sleep 3 secs to wait garbage correction in kernel side and retry */ - ltfsmsg(LTFS_WARN, 30440W, ++(*retry)); - sleep(3); - ret = lin_tape_ibmtape_readpos(device, &tmp_pos); if (ret == DEVICE_GOOD && pos->partition == tmp_pos.partition) { if (pos->block == tmp_pos.block) { @@ -1547,7 +1543,9 @@ int lin_tape_ibmtape_write(void *device, const char *buf, size_t count, struct t rc = DEVICE_GOOD; } } else if (errno == ENOMEM && retry < MAX_WRITE_RETRY) { - rc = _handle_block_allocation_failure(device, pos, &retry); + sleep(3); // Wait for kernel GC + ltfsmsg(LTFS_WARN, 30440W, ++retry); + rc = _handle_block_write_failure(device, pos); if (rc == WRITE_RETRY) { errno = 0; goto write_start; @@ -1573,7 +1571,8 @@ int lin_tape_ibmtape_write(void *device, const char *buf, size_t count, struct t if (retry < MAX_WRITE_RETRY && ((current_errno == EIO && rc == -EDEV_NO_SENSE ) || (rc == -EDEV_CONFIGURE_CHANGED) || (rc == -EDEV_TIME_STAMP_CHANGED))) { - rc = _handle_block_allocation_failure(device, pos, &retry); + sleep(5); + rc = _handle_block_write_failure(device, pos); if (rc == WRITE_RETRY) { errno = 0; goto write_start; diff --git a/src/tape_drivers/linux/sg/sg_tape.c b/src/tape_drivers/linux/sg/sg_tape.c index edcf0988..f736ab80 100644 --- a/src/tape_drivers/linux/sg/sg_tape.c +++ b/src/tape_drivers/linux/sg/sg_tape.c @@ -54,6 +54,7 @@ #include #include +#include "libltfs/ltfs_error.h" #include "ltfs_copyright.h" #include "libltfs/ltfslogging.h" #include "libltfs/fs.h" @@ -104,7 +105,7 @@ struct sg_global_data global_data; #define MAX_RETRY (100) #define MAX_TAKE_DUMP_ATTEMPTS (10) -#define SOFT_ERROR_MAX_RETRIES (3) +#define POR_MAX_RETRIES (3) /* Forward references (For keep function order to struct tape_ops) */ int sg_readpos(void *device, struct tc_position *pos); @@ -605,7 +606,7 @@ int _raw_tur(const int fd) #define _clear_por(p) _clear_por_raw((p)->dev.fd); -void _clear_por_raw(const int fd) +int _clear_por_raw(const int fd) { int i = 0, ret = -1; @@ -625,6 +626,7 @@ void _clear_por_raw(const int fd) } i++; } + return ret; } #define _get_stable_tur_response(p) _get_stable_tur_response_raw((p)->dev.fd) @@ -1862,16 +1864,11 @@ static int _cdb_read(void *device, char *buf, size_t size, bool sili) return length; } -static inline int _handle_block_allocation_failure(void *device, struct tc_position *pos, - int *retry, char *op) +static inline int _handle_block_write_failure(void *device, struct tc_position *pos, char *op) { int ret = 0; struct tc_position tmp_pos = {0, 0}; - /* Sleep 3 secs to wait garbage correction in kernel side and retry */ - ltfsmsg(LTFS_WARN, 30277W, ++(*retry)); - sleep(3); - ret = sg_readpos(device, &tmp_pos); if (ret == DEVICE_GOOD && pos->partition == tmp_pos.partition) { if (pos->block == tmp_pos.block) { @@ -1985,7 +1982,9 @@ int sg_read(void *device, char *buf, size_t size, priv->use_sili = false; ret = _cdb_read(device, buf, datacount, unusual_size); } else if (ret == -EDEV_BUFFER_ALLOCATE_ERROR && retry_count < MAX_RETRY) { - ret = _handle_block_allocation_failure(device, pos, &retry_count, "read"); + sleep(3); // Wait for kernel GC + ltfsmsg(LTFS_WARN, 30277W, ++retry_count); + ret = _handle_block_write_failure(device, pos, "read"); if (ret == -EDEV_RETRY) goto start_read; } @@ -2099,8 +2098,7 @@ int sg_write(void *device, const char *buf, size_t count, struct tc_position *po struct sg_data *priv = (struct sg_data*)device; struct tc_position cur_pos; size_t datacount = count; - int reconnect_retry_count = 0, soft_error_retry_count = 0; - int TBL_SLEEP_SECS[SOFT_ERROR_MAX_RETRIES] = {45, 60, 75}; // Hardcoded exponentially increasing sleep time + int retry_count = 0, por_retry_count = 0; ltfs_profiler_add_entry(priv->profiler, NULL, TAPEBEND_REQ_ENTER(REQ_TC_WRITE)); @@ -2147,15 +2145,21 @@ int sg_write(void *device, const char *buf, size_t count, struct tc_position *po } else ret = -EDEV_POR_OR_BUS_RESET; } - } else if (ret == -EDEV_BUFFER_ALLOCATE_ERROR && reconnect_retry_count < MAX_RETRY) { - ret = _handle_block_allocation_failure(device, pos, &reconnect_retry_count, "write"); - if (ret == -EDEV_RETRY) - goto start_write; - } else if (ret == -EDEV_HOST_ERROR && soft_error_retry_count < SOFT_ERROR_MAX_RETRIES) { - sleep(TBL_SLEEP_SECS[soft_error_retry_count]); - ret = _handle_block_allocation_failure(device, pos, &retry_count, "write"); + } else if (ret == -EDEV_BUFFER_ALLOCATE_ERROR && retry_count < MAX_RETRY) { + sleep(3); // Wait for kernel GC + ltfsmsg(LTFS_WARN, 30277W, ++retry_count); + ret = _handle_block_write_failure(device, pos, "write"); if (ret == -EDEV_RETRY) goto start_write; + } else if (ret == -EDEV_HOST_ERROR && por_retry_count < POR_MAX_RETRIES) { + por_retry_count++; + sleep(5); + ret = _clear_por(priv); + if (ret == DEVICE_GOOD) { + ret = _handle_block_write_failure(device, pos, "write"); + if (ret == -EDEV_RETRY) + goto start_write; + } } ltfs_profiler_add_entry(priv->profiler, NULL, TAPEBEND_REQ_EXIT(REQ_TC_WRITE)); diff --git a/src/tape_drivers/netbsd/scsipi-ibmtape/scsipi_ibmtape.c b/src/tape_drivers/netbsd/scsipi-ibmtape/scsipi_ibmtape.c index 0ea8df55..153c2f14 100644 --- a/src/tape_drivers/netbsd/scsipi-ibmtape/scsipi_ibmtape.c +++ b/src/tape_drivers/netbsd/scsipi-ibmtape/scsipi_ibmtape.c @@ -96,6 +96,7 @@ struct scsipi_ibmtape_global_data global_data; #define TU_DEFAULT_TIMEOUT (60) #define MAX_RETRY (100) +#define POR_MAX_RETRIES (3) /* Forward references (For keep function order to struct tape_ops) */ int scsipi_ibmtape_readpos(void *device, struct tc_position *pos); @@ -569,7 +570,7 @@ int _raw_tur(const int fd) #define _clear_por(p) _clear_por_raw((p)->dev.fd); -void _clear_por_raw(const int fd) +int _clear_por_raw(const int fd) { int i = 0, ret = -1; @@ -589,6 +590,7 @@ void _clear_por_raw(const int fd) } i++; } + return ret; } /* Forward reference */ @@ -1656,7 +1658,7 @@ int scsipi_ibmtape_write(void *device, const char *buf, size_t count, struct tc_ struct scsipi_ibmtape_data *priv = (struct scsipi_ibmtape_data*)device; struct tc_position cur_pos; size_t datacount = count; - int retry_count = 0; + int retry_count = 0, por_retry_count = 0; ltfs_profiler_add_entry(priv->profiler, NULL, TAPEBEND_REQ_ENTER(REQ_TC_WRITE)); @@ -1707,6 +1709,14 @@ int scsipi_ibmtape_write(void *device, const char *buf, size_t count, struct tc_ ret = _handle_block_allocation_failure(device, pos, &retry_count, "write"); if (ret == -EDEV_RETRY) goto start_write; + } else if (ret == -EDEV_HOST_ERROR && por_retry_count < POR_MAX_RETRIES) { + sleep(5); + ret = _clear_por(priv); + if (ret == DEVICE_GOOD) { + ret = _handle_block_allocation_failure(device, pos, &por_retry_count, "write"); + if (ret == -EDEV_RETRY) + goto start_write; + } } ltfs_profiler_add_entry(priv->profiler, NULL, TAPEBEND_REQ_EXIT(REQ_TC_WRITE)); From 922c67dd022f236ecaf0f97ce81c624de24cbea5 Mon Sep 17 00:00:00 2001 From: syaoraang Date: Fri, 31 Jul 2026 13:49:05 -0600 Subject: [PATCH 3/8] chore: Fixing identation fix: Avoiding hidding cdb_write return code fix: Using correct return value to check write output --- .../linux/lin_tape/lin_tape_ibmtape.c | 6 ++-- src/tape_drivers/linux/sg/sg_tape.c | 32 +++++++++---------- 2 files changed, 19 insertions(+), 19 deletions(-) diff --git a/src/tape_drivers/linux/lin_tape/lin_tape_ibmtape.c b/src/tape_drivers/linux/lin_tape/lin_tape_ibmtape.c index 546ff702..57ee6298 100644 --- a/src/tape_drivers/linux/lin_tape/lin_tape_ibmtape.c +++ b/src/tape_drivers/linux/lin_tape/lin_tape_ibmtape.c @@ -1543,8 +1543,8 @@ int lin_tape_ibmtape_write(void *device, const char *buf, size_t count, struct t rc = DEVICE_GOOD; } } else if (errno == ENOMEM && retry < MAX_WRITE_RETRY) { - sleep(3); // Wait for kernel GC - ltfsmsg(LTFS_WARN, 30440W, ++retry); + sleep(3); // Wait for kernel GC + ltfsmsg(LTFS_WARN, 30440W, ++retry); rc = _handle_block_write_failure(device, pos); if (rc == WRITE_RETRY) { errno = 0; @@ -1571,7 +1571,7 @@ int lin_tape_ibmtape_write(void *device, const char *buf, size_t count, struct t if (retry < MAX_WRITE_RETRY && ((current_errno == EIO && rc == -EDEV_NO_SENSE ) || (rc == -EDEV_CONFIGURE_CHANGED) || (rc == -EDEV_TIME_STAMP_CHANGED))) { - sleep(5); + sleep(5); rc = _handle_block_write_failure(device, pos); if (rc == WRITE_RETRY) { errno = 0; diff --git a/src/tape_drivers/linux/sg/sg_tape.c b/src/tape_drivers/linux/sg/sg_tape.c index f736ab80..22b420b8 100644 --- a/src/tape_drivers/linux/sg/sg_tape.c +++ b/src/tape_drivers/linux/sg/sg_tape.c @@ -1982,8 +1982,8 @@ int sg_read(void *device, char *buf, size_t size, priv->use_sili = false; ret = _cdb_read(device, buf, datacount, unusual_size); } else if (ret == -EDEV_BUFFER_ALLOCATE_ERROR && retry_count < MAX_RETRY) { - sleep(3); // Wait for kernel GC - ltfsmsg(LTFS_WARN, 30277W, ++retry_count); + sleep(3); // Wait for kernel GC + ltfsmsg(LTFS_WARN, 30277W, ++retry_count); ret = _handle_block_write_failure(device, pos, "read"); if (ret == -EDEV_RETRY) goto start_read; @@ -2093,7 +2093,7 @@ static int _cdb_write(void *device, uint8_t *buf, size_t size, bool *ew, bool *p int sg_write(void *device, const char *buf, size_t count, struct tc_position *pos) { - int ret, ret_fo; + int ret, ret_fo, ret_write = -1; bool ew = false, pew = false; struct sg_data *priv = (struct sg_data*)device; struct tc_position cur_pos; @@ -2128,12 +2128,12 @@ int sg_write(void *device, const char *buf, size_t count, struct tc_position *po } start_write: - ret = _cdb_write(device, (uint8_t *)buf, datacount, &ew, &pew); - if (ret == DEVICE_GOOD) { + ret_write = _cdb_write(device, (uint8_t *)buf, datacount, &ew, &pew); + if (ret_write == DEVICE_GOOD) { pos->block++; pos->early_warning = ew; pos->programmable_early_warning = pew; - } else if (ret == -EDEV_NEED_FAILOVER) { + } else if (ret_write == -EDEV_NEED_FAILOVER) { ret_fo = sg_readpos(device, &cur_pos); if (!ret_fo) { if (pos->partition == cur_pos.partition @@ -2141,30 +2141,30 @@ int sg_write(void *device, const char *buf, size_t count, struct tc_position *po pos->block++; pos->early_warning = cur_pos.early_warning; pos->programmable_early_warning = cur_pos.programmable_early_warning; - ret = DEVICE_GOOD; + ret = ret_write = DEVICE_GOOD; } else ret = -EDEV_POR_OR_BUS_RESET; } - } else if (ret == -EDEV_BUFFER_ALLOCATE_ERROR && retry_count < MAX_RETRY) { - sleep(3); // Wait for kernel GC - ltfsmsg(LTFS_WARN, 30277W, ++retry_count); + } else if (ret_write == -EDEV_BUFFER_ALLOCATE_ERROR && retry_count < MAX_RETRY) { + sleep(3); // Wait for kernel GC + ltfsmsg(LTFS_WARN, 30277W, ++retry_count); ret = _handle_block_write_failure(device, pos, "write"); if (ret == -EDEV_RETRY) goto start_write; - } else if (ret == -EDEV_HOST_ERROR && por_retry_count < POR_MAX_RETRIES) { - por_retry_count++; + } else if (ret_write == -EDEV_HOST_ERROR && por_retry_count < POR_MAX_RETRIES) { + por_retry_count++; sleep(5); ret = _clear_por(priv); if (ret == DEVICE_GOOD) { - ret = _handle_block_write_failure(device, pos, "write"); - if (ret == -EDEV_RETRY) - goto start_write; + ret = _handle_block_write_failure(device, pos, "write"); + if (ret == -EDEV_RETRY) + goto start_write; } } ltfs_profiler_add_entry(priv->profiler, NULL, TAPEBEND_REQ_EXIT(REQ_TC_WRITE)); - return ret; + return ret_write; } int sg_writefm(void *device, size_t count, struct tc_position *pos, bool immed) From a10c057cade8653d8b5951ed4f7be4650a579755 Mon Sep 17 00:00:00 2001 From: mcardenas Date: Wed, 5 Aug 2026 11:18:43 -0600 Subject: [PATCH 4/8] fix: ret and netbsd driver fix: --- .../linux/lin_tape/lin_tape_ibmtape.c | 2 +- src/tape_drivers/linux/sg/sg_tape.c | 28 ++++++++++--------- .../netbsd/scsipi-ibmtape/scsipi_ibmtape.c | 26 +++++++++-------- 3 files changed, 30 insertions(+), 26 deletions(-) diff --git a/src/tape_drivers/linux/lin_tape/lin_tape_ibmtape.c b/src/tape_drivers/linux/lin_tape/lin_tape_ibmtape.c index 57ee6298..a8bc33f3 100644 --- a/src/tape_drivers/linux/lin_tape/lin_tape_ibmtape.c +++ b/src/tape_drivers/linux/lin_tape/lin_tape_ibmtape.c @@ -1543,8 +1543,8 @@ int lin_tape_ibmtape_write(void *device, const char *buf, size_t count, struct t rc = DEVICE_GOOD; } } else if (errno == ENOMEM && retry < MAX_WRITE_RETRY) { - sleep(3); // Wait for kernel GC ltfsmsg(LTFS_WARN, 30440W, ++retry); + sleep(3); // Wait for kernel GC rc = _handle_block_write_failure(device, pos); if (rc == WRITE_RETRY) { errno = 0; diff --git a/src/tape_drivers/linux/sg/sg_tape.c b/src/tape_drivers/linux/sg/sg_tape.c index 22b420b8..49573b15 100644 --- a/src/tape_drivers/linux/sg/sg_tape.c +++ b/src/tape_drivers/linux/sg/sg_tape.c @@ -1982,8 +1982,8 @@ int sg_read(void *device, char *buf, size_t size, priv->use_sili = false; ret = _cdb_read(device, buf, datacount, unusual_size); } else if (ret == -EDEV_BUFFER_ALLOCATE_ERROR && retry_count < MAX_RETRY) { - sleep(3); // Wait for kernel GC ltfsmsg(LTFS_WARN, 30277W, ++retry_count); + sleep(3); // Wait for kernel GC ret = _handle_block_write_failure(device, pos, "read"); if (ret == -EDEV_RETRY) goto start_read; @@ -2128,12 +2128,12 @@ int sg_write(void *device, const char *buf, size_t count, struct tc_position *po } start_write: - ret_write = _cdb_write(device, (uint8_t *)buf, datacount, &ew, &pew); - if (ret_write == DEVICE_GOOD) { + ret_write = ret = _cdb_write(device, (uint8_t *)buf, datacount, &ew, &pew); + if (ret == DEVICE_GOOD) { pos->block++; pos->early_warning = ew; pos->programmable_early_warning = pew; - } else if (ret_write == -EDEV_NEED_FAILOVER) { + } else if (ret == -EDEV_NEED_FAILOVER) { ret_fo = sg_readpos(device, &cur_pos); if (!ret_fo) { if (pos->partition == cur_pos.partition @@ -2141,30 +2141,32 @@ int sg_write(void *device, const char *buf, size_t count, struct tc_position *po pos->block++; pos->early_warning = cur_pos.early_warning; pos->programmable_early_warning = cur_pos.programmable_early_warning; - ret = ret_write = DEVICE_GOOD; + ret = DEVICE_GOOD; } else ret = -EDEV_POR_OR_BUS_RESET; } - } else if (ret_write == -EDEV_BUFFER_ALLOCATE_ERROR && retry_count < MAX_RETRY) { - sleep(3); // Wait for kernel GC + } else if (ret == -EDEV_BUFFER_ALLOCATE_ERROR && retry_count < MAX_RETRY) { ltfsmsg(LTFS_WARN, 30277W, ++retry_count); + sleep(3); // Wait for kernel GC ret = _handle_block_write_failure(device, pos, "write"); if (ret == -EDEV_RETRY) goto start_write; - } else if (ret_write == -EDEV_HOST_ERROR && por_retry_count < POR_MAX_RETRIES) { + } else if (ret == -EDEV_HOST_ERROR && por_retry_count < POR_MAX_RETRIES) { por_retry_count++; sleep(5); ret = _clear_por(priv); if (ret == DEVICE_GOOD) { - ret = _handle_block_write_failure(device, pos, "write"); - if (ret == -EDEV_RETRY) - goto start_write; - } + ret = _handle_block_write_failure(device, pos, "write"); + if (ret == -EDEV_RETRY) + goto start_write; + ret = DEVICE_GOOD; + } else + ret = ret_write; } ltfs_profiler_add_entry(priv->profiler, NULL, TAPEBEND_REQ_EXIT(REQ_TC_WRITE)); - return ret_write; + return ret; } int sg_writefm(void *device, size_t count, struct tc_position *pos, bool immed) diff --git a/src/tape_drivers/netbsd/scsipi-ibmtape/scsipi_ibmtape.c b/src/tape_drivers/netbsd/scsipi-ibmtape/scsipi_ibmtape.c index 153c2f14..1f5d6e92 100644 --- a/src/tape_drivers/netbsd/scsipi-ibmtape/scsipi_ibmtape.c +++ b/src/tape_drivers/netbsd/scsipi-ibmtape/scsipi_ibmtape.c @@ -1427,16 +1427,11 @@ static int _cdb_read(void *device, char *buf, size_t size, bool sili) return length; } -static inline int _handle_block_allocation_failure(void *device, struct tc_position *pos, - int *retry, char *op) +static inline int _handle_block_write_failure(void *device, struct tc_position *pos, char *op) { int ret = 0; struct tc_position tmp_pos = {0, 0}; - /* Sleep 3 secs to wait garbage correction in kernel side and retry */ - ltfsmsg(LTFS_WARN, 30277W, ++(*retry)); - sleep(3); - ret = scsipi_ibmtape_readpos(device, &tmp_pos); if (ret == DEVICE_GOOD && pos->partition == tmp_pos.partition) { if (pos->block == tmp_pos.block) { @@ -1550,7 +1545,9 @@ int scsipi_ibmtape_read(void *device, char *buf, size_t size, priv->use_sili = false; ret = _cdb_read(device, buf, datacount, unusual_size); } else if (ret == -EDEV_BUFFER_ALLOCATE_ERROR && retry_count < MAX_RETRY) { - ret = _handle_block_allocation_failure(device, pos, &retry_count, "read"); + ltfsmsg(LTFS_WARN, 30277W, ++(*retry)); + sleep(3); // Wait for kernel GC + ret = _handle_block_write_failure(device, pos, "read"); if (ret == -EDEV_RETRY) goto start_read; } @@ -1653,7 +1650,7 @@ static int _cdb_write(void *device, uint8_t *buf, size_t size, bool *ew, bool *p int scsipi_ibmtape_write(void *device, const char *buf, size_t count, struct tc_position *pos) { - int ret, ret_fo; + int ret, ret_fo, ret_write = -1; bool ew = false, pew = false; struct scsipi_ibmtape_data *priv = (struct scsipi_ibmtape_data*)device; struct tc_position cur_pos; @@ -1706,17 +1703,22 @@ int scsipi_ibmtape_write(void *device, const char *buf, size_t count, struct tc_ ret = -EDEV_POR_OR_BUS_RESET; } } else if (ret == -EDEV_BUFFER_ALLOCATE_ERROR && retry_count < MAX_RETRY) { - ret = _handle_block_allocation_failure(device, pos, &retry_count, "write"); + ltfsmsg(LTFS_WARN, 30277W, ++(*retry)); + sleep(3); // Wait for kernel GC + ret = _handle_block_write_failure(device, pos, "write"); if (ret == -EDEV_RETRY) goto start_write; } else if (ret == -EDEV_HOST_ERROR && por_retry_count < POR_MAX_RETRIES) { + por_retry_count++; sleep(5); ret = _clear_por(priv); if (ret == DEVICE_GOOD) { - ret = _handle_block_allocation_failure(device, pos, &por_retry_count, "write"); - if (ret == -EDEV_RETRY) + ret = _handle_block_write_failure(device, pos, "write"); + if (ret == -EDEV_RETRY) goto start_write; - } + ret = DEVICE_GOOD; + } else + ret = ret_write; } ltfs_profiler_add_entry(priv->profiler, NULL, TAPEBEND_REQ_EXIT(REQ_TC_WRITE)); From 8421c70f7f73ed802cfa6e020c4388fe2f72b618 Mon Sep 17 00:00:00 2001 From: Jorge Acosta Date: Wed, 12 Aug 2026 11:12:12 -0600 Subject: [PATCH 5/8] fix: Attending Missael's requests fix: Attending Missael's requests fix: Attending Missael's requests fix: write soft error handling (POR and transport failures) Fix: start_block misalignment caused by SIGSTOP (#552) fix: start_block misalignment caused by position mismatch #11 fix: Fixing compilation fix: Fixing compilation chore: Attending Japan team comments * Handling better the return code that was forcefully set as DEVICE_GOOD * Renaming _handle_block_write_failure() to _resolve_position_after_io_cmd_failure() to reduce naming concerns and yet don't repeat code. We don't take inconsideration the EDEV_BUFFER_ALLOCATE_ERROR when called by -EDEV_HOST_ERROR since it doesn't affect that path. fix: Avoiding shadowing the _cdb_write() return value if _clear_por() return value is not DEVICE_GOOD. Fixing in scsipi_ibmtape.c file an issue where the ret_write wasn't changed never but returned anyways. chore: Adding log to mention the retry due the POR chore: Attending Japan team comments on POR issue PR --- messages/tape_linux_sg/root.txt | 1 + .../linux/lin_tape/lin_tape_ibmtape.c | 16 +++++--- src/tape_drivers/linux/sg/sg_tape.c | 35 +++++++++++------ .../netbsd/scsipi-ibmtape/scsipi_ibmtape.c | 38 ++++++++++++------- 4 files changed, 60 insertions(+), 30 deletions(-) diff --git a/messages/tape_linux_sg/root.txt b/messages/tape_linux_sg/root.txt index 981fdc1b..e4f11b3e 100644 --- a/messages/tape_linux_sg/root.txt +++ b/messages/tape_linux_sg/root.txt @@ -138,6 +138,7 @@ root:table { 30295I:string { "Have unstable TUR response, start over (Cur = %d, Prev = %d)." } 30296I:string { "Capturing a stable TUR at line %d." } 30297W:string { "Cannot retrieve drive dump: failed to communicate with drive. Tried (%d) times." } + 30298W:string { "Retrying write operation due Power On-Reset event reached. Tried (%d) times." } 30392D:string { "Backend %s %s." } 30393D:string { "Backend %s: %d %s." } diff --git a/src/tape_drivers/linux/lin_tape/lin_tape_ibmtape.c b/src/tape_drivers/linux/lin_tape/lin_tape_ibmtape.c index a8bc33f3..edc420db 100644 --- a/src/tape_drivers/linux/lin_tape/lin_tape_ibmtape.c +++ b/src/tape_drivers/linux/lin_tape/lin_tape_ibmtape.c @@ -1439,7 +1439,7 @@ int lin_tape_ibmtape_read(void *device, char *buf, size_t count, struct tc_posit #define WRITE_RETRY (-LINUX_MAX_BLOCK_SIZE) -static inline int _handle_block_write_failure(void *device, struct tc_position *pos) +static inline int _resolve_position_after_io_cmd_failure(void *device, struct tc_position *pos) { int ret = 0; struct tc_position tmp_pos = {0, 0}; @@ -1544,8 +1544,11 @@ int lin_tape_ibmtape_write(void *device, const char *buf, size_t count, struct t } } else if (errno == ENOMEM && retry < MAX_WRITE_RETRY) { ltfsmsg(LTFS_WARN, 30440W, ++retry); - sleep(3); // Wait for kernel GC - rc = _handle_block_write_failure(device, pos); + struct timespec delay_ts; + delay_ts.tv_sec = 3; + delay_ts.tv_nsec = 0; + nanosleep(&delay_ts, NULL); // Wait for kernel GC + rc = _resolve_position_after_io_cmd_failure(device, pos); if (rc == WRITE_RETRY) { errno = 0; goto write_start; @@ -1571,8 +1574,11 @@ int lin_tape_ibmtape_write(void *device, const char *buf, size_t count, struct t if (retry < MAX_WRITE_RETRY && ((current_errno == EIO && rc == -EDEV_NO_SENSE ) || (rc == -EDEV_CONFIGURE_CHANGED) || (rc == -EDEV_TIME_STAMP_CHANGED))) { - sleep(5); - rc = _handle_block_write_failure(device, pos); + struct timespec delay_ts; + delay_ts.tv_sec = 5; + delay_ts.tv_nsec = 0; + nanosleep(&delay_ts, NULL); + rc = _resolve_position_after_io_cmd_failure(device, pos); if (rc == WRITE_RETRY) { errno = 0; goto write_start; diff --git a/src/tape_drivers/linux/sg/sg_tape.c b/src/tape_drivers/linux/sg/sg_tape.c index 49573b15..de405bd9 100644 --- a/src/tape_drivers/linux/sg/sg_tape.c +++ b/src/tape_drivers/linux/sg/sg_tape.c @@ -1864,7 +1864,7 @@ static int _cdb_read(void *device, char *buf, size_t size, bool sili) return length; } -static inline int _handle_block_write_failure(void *device, struct tc_position *pos, char *op) +static inline int _resolve_position_after_io_cmd_failure(void *device, struct tc_position *pos, char *op) { int ret = 0; struct tc_position tmp_pos = {0, 0}; @@ -1983,8 +1983,11 @@ int sg_read(void *device, char *buf, size_t size, ret = _cdb_read(device, buf, datacount, unusual_size); } else if (ret == -EDEV_BUFFER_ALLOCATE_ERROR && retry_count < MAX_RETRY) { ltfsmsg(LTFS_WARN, 30277W, ++retry_count); - sleep(3); // Wait for kernel GC - ret = _handle_block_write_failure(device, pos, "read"); + struct timespec delay_ts; + delay_ts.tv_sec = 3; + delay_ts.tv_nsec = 0; + nanosleep(&delay_ts, NULL); // Wait for kernel GC + ret = _resolve_position_after_io_cmd_failure(device, pos, "read"); if (ret == -EDEV_RETRY) goto start_read; } @@ -2099,6 +2102,7 @@ int sg_write(void *device, const char *buf, size_t count, struct tc_position *po struct tc_position cur_pos; size_t datacount = count; int retry_count = 0, por_retry_count = 0; + struct timespec delay_ts = {0}; ltfs_profiler_add_entry(priv->profiler, NULL, TAPEBEND_REQ_ENTER(REQ_TC_WRITE)); @@ -2147,21 +2151,28 @@ int sg_write(void *device, const char *buf, size_t count, struct tc_position *po } } else if (ret == -EDEV_BUFFER_ALLOCATE_ERROR && retry_count < MAX_RETRY) { ltfsmsg(LTFS_WARN, 30277W, ++retry_count); - sleep(3); // Wait for kernel GC - ret = _handle_block_write_failure(device, pos, "write"); + delay_ts.tv_sec = 3; + delay_ts.tv_nsec = 0; + nanosleep(&delay_ts, NULL); // Wait for kernel GC + ret = _resolve_position_after_io_cmd_failure(device, pos, "write"); if (ret == -EDEV_RETRY) goto start_write; } else if (ret == -EDEV_HOST_ERROR && por_retry_count < POR_MAX_RETRIES) { por_retry_count++; - sleep(5); + delay_ts.tv_sec = 5; + delay_ts.tv_nsec = 0; + nanosleep(&delay_ts, NULL); ret = _clear_por(priv); if (ret == DEVICE_GOOD) { - ret = _handle_block_write_failure(device, pos, "write"); - if (ret == -EDEV_RETRY) - goto start_write; - ret = DEVICE_GOOD; - } else - ret = ret_write; + int handle_ret = _resolve_position_after_io_cmd_failure(device, pos, "write"); + /* If the original command did not reach the driver, or it reached it but after failing there is block mismatch; retry */ + if (handle_ret == -EDEV_RETRY) { + ltfsmsg(LTFS_WARN, 30298W, por_retry_count); + goto start_write; + } + } + // If we could not clear the POR status, just return the _cdb_write() return value + ret = ret_write; } ltfs_profiler_add_entry(priv->profiler, NULL, TAPEBEND_REQ_EXIT(REQ_TC_WRITE)); diff --git a/src/tape_drivers/netbsd/scsipi-ibmtape/scsipi_ibmtape.c b/src/tape_drivers/netbsd/scsipi-ibmtape/scsipi_ibmtape.c index 1f5d6e92..67131844 100644 --- a/src/tape_drivers/netbsd/scsipi-ibmtape/scsipi_ibmtape.c +++ b/src/tape_drivers/netbsd/scsipi-ibmtape/scsipi_ibmtape.c @@ -1427,7 +1427,7 @@ static int _cdb_read(void *device, char *buf, size_t size, bool sili) return length; } -static inline int _handle_block_write_failure(void *device, struct tc_position *pos, char *op) +static inline int _resolve_position_after_io_cmd_failure(void *device, struct tc_position *pos, char *op) { int ret = 0; struct tc_position tmp_pos = {0, 0}; @@ -1546,8 +1546,11 @@ int scsipi_ibmtape_read(void *device, char *buf, size_t size, ret = _cdb_read(device, buf, datacount, unusual_size); } else if (ret == -EDEV_BUFFER_ALLOCATE_ERROR && retry_count < MAX_RETRY) { ltfsmsg(LTFS_WARN, 30277W, ++(*retry)); - sleep(3); // Wait for kernel GC - ret = _handle_block_write_failure(device, pos, "read"); + struct timespec delay_ts; + delay_ts.tv_sec = 3; + delay_ts.tv_nsec = 0; + nanosleep(&delay_ts, NULL); // Wait for kernel GC + ret = _resolve_position_after_io_cmd_failure(device, pos, "read"); if (ret == -EDEV_RETRY) goto start_read; } @@ -1685,7 +1688,7 @@ int scsipi_ibmtape_write(void *device, const char *buf, size_t count, struct tc_ } start_write: - ret = _cdb_write(device, (uint8_t *)buf, datacount, &ew, &pew); + ret_write = ret = _cdb_write(device, (uint8_t *)buf, datacount, &ew, &pew); if (ret == DEVICE_GOOD) { pos->block++; pos->early_warning = ew; @@ -1704,21 +1707,30 @@ int scsipi_ibmtape_write(void *device, const char *buf, size_t count, struct tc_ } } else if (ret == -EDEV_BUFFER_ALLOCATE_ERROR && retry_count < MAX_RETRY) { ltfsmsg(LTFS_WARN, 30277W, ++(*retry)); - sleep(3); // Wait for kernel GC - ret = _handle_block_write_failure(device, pos, "write"); + struct timespec delay_ts; + delay_ts.tv_sec = 3; + delay_ts.tv_nsec = 0; + nanosleep(&delay_ts, NULL); // Wait for kernel GC + ret = _resolve_position_after_io_cmd_failure(device, pos, "write"); if (ret == -EDEV_RETRY) goto start_write; } else if (ret == -EDEV_HOST_ERROR && por_retry_count < POR_MAX_RETRIES) { por_retry_count++; - sleep(5); + struct timespec delay_ts; + delay_ts.tv_sec = 5; + delay_ts.tv_nsec = 0; + nanosleep(&delay_ts, NULL); ret = _clear_por(priv); if (ret == DEVICE_GOOD) { - ret = _handle_block_write_failure(device, pos, "write"); - if (ret == -EDEV_RETRY) - goto start_write; - ret = DEVICE_GOOD; - } else - ret = ret_write; + int handle_ret = _resolve_position_after_io_cmd_failure(device, pos, "write"); + // If the original command did not reach the driver, or it reached it but after failing there is block mismatch; retry + if (handle_ret == -EDEV_RETRY) { + ltfsmsg(LTFS_WARN, 30298W, por_retry_count); + goto start_write; + } + } + // If we could not clear the POR status, just return the _cdb_write() return value + ret = ret_write; } ltfs_profiler_add_entry(priv->profiler, NULL, TAPEBEND_REQ_EXIT(REQ_TC_WRITE)); From 1d1246d31f4d630c855cccdaf630e0a9edc0030f Mon Sep 17 00:00:00 2001 From: madjesc Date: Wed, 19 Aug 2026 11:01:51 -0600 Subject: [PATCH 6/8] fix: return is not valid --- src/tape_drivers/linux/sg/sg_tape.c | 7 +++---- 1 file changed, 3 insertions(+), 4 deletions(-) diff --git a/src/tape_drivers/linux/sg/sg_tape.c b/src/tape_drivers/linux/sg/sg_tape.c index de405bd9..172954c9 100644 --- a/src/tape_drivers/linux/sg/sg_tape.c +++ b/src/tape_drivers/linux/sg/sg_tape.c @@ -2165,14 +2165,13 @@ int sg_write(void *device, const char *buf, size_t count, struct tc_position *po ret = _clear_por(priv); if (ret == DEVICE_GOOD) { int handle_ret = _resolve_position_after_io_cmd_failure(device, pos, "write"); - /* If the original command did not reach the driver, or it reached it but after failing there is block mismatch; retry */ + /* If the original command did not reach the driver, or it reached it but after failing there is a position mismatch; retry */ if (handle_ret == -EDEV_RETRY) { ltfsmsg(LTFS_WARN, 30298W, por_retry_count); goto start_write; } - } - // If we could not clear the POR status, just return the _cdb_write() return value - ret = ret_write; + } else // If we could not clear the POR status, just return the _cdb_write() return value + ret = ret_write; } ltfs_profiler_add_entry(priv->profiler, NULL, TAPEBEND_REQ_EXIT(REQ_TC_WRITE)); From 3c5088f6b4d1d2e6b0194fd41af05c503a36eed9 Mon Sep 17 00:00:00 2001 From: Mauricio de Jesus Cardenas Hernandez Date: Wed, 19 Aug 2026 11:56:09 -0600 Subject: [PATCH 7/8] chore: make readability better Implement dylan suggestion Co-authored-by: DYLAN CARLSON --- src/tape_drivers/linux/sg/sg_tape.c | 7 +++++-- 1 file changed, 5 insertions(+), 2 deletions(-) diff --git a/src/tape_drivers/linux/sg/sg_tape.c b/src/tape_drivers/linux/sg/sg_tape.c index 172954c9..7bbbc356 100644 --- a/src/tape_drivers/linux/sg/sg_tape.c +++ b/src/tape_drivers/linux/sg/sg_tape.c @@ -2170,8 +2170,11 @@ int sg_write(void *device, const char *buf, size_t count, struct tc_position *po ltfsmsg(LTFS_WARN, 30298W, por_retry_count); goto start_write; } - } else // If we could not clear the POR status, just return the _cdb_write() return value - ret = ret_write; + } + // If we could not clear the POR status, just return the _cdb_write() return value + else { + ret = ret_write; + } } ltfs_profiler_add_entry(priv->profiler, NULL, TAPEBEND_REQ_EXIT(REQ_TC_WRITE)); From d07ad3744709ac3d30280eb01f9f36e0eb1bcaf4 Mon Sep 17 00:00:00 2001 From: madjesc Date: Wed, 19 Aug 2026 12:30:30 -0600 Subject: [PATCH 8/8] fix: readibility fix: write soft error handling (POR and transport failures) --- src/tape_drivers/linux/sg/sg_tape.c | 7 +++---- src/tape_drivers/netbsd/scsipi-ibmtape/scsipi_ibmtape.c | 5 +++-- 2 files changed, 6 insertions(+), 6 deletions(-) diff --git a/src/tape_drivers/linux/sg/sg_tape.c b/src/tape_drivers/linux/sg/sg_tape.c index 7bbbc356..3d43e897 100644 --- a/src/tape_drivers/linux/sg/sg_tape.c +++ b/src/tape_drivers/linux/sg/sg_tape.c @@ -2170,11 +2170,10 @@ int sg_write(void *device, const char *buf, size_t count, struct tc_position *po ltfsmsg(LTFS_WARN, 30298W, por_retry_count); goto start_write; } + } else { + // If we could not clear the POR status, just return the _cdb_write() return value + ret = ret_write; } - // If we could not clear the POR status, just return the _cdb_write() return value - else { - ret = ret_write; - } } ltfs_profiler_add_entry(priv->profiler, NULL, TAPEBEND_REQ_EXIT(REQ_TC_WRITE)); diff --git a/src/tape_drivers/netbsd/scsipi-ibmtape/scsipi_ibmtape.c b/src/tape_drivers/netbsd/scsipi-ibmtape/scsipi_ibmtape.c index 67131844..8cbe3000 100644 --- a/src/tape_drivers/netbsd/scsipi-ibmtape/scsipi_ibmtape.c +++ b/src/tape_drivers/netbsd/scsipi-ibmtape/scsipi_ibmtape.c @@ -1728,9 +1728,10 @@ int scsipi_ibmtape_write(void *device, const char *buf, size_t count, struct tc_ ltfsmsg(LTFS_WARN, 30298W, por_retry_count); goto start_write; } + } else { + // If we could not clear the POR status, just return the _cdb_write() return value + ret = ret_write; } - // If we could not clear the POR status, just return the _cdb_write() return value - ret = ret_write; } ltfs_profiler_add_entry(priv->profiler, NULL, TAPEBEND_REQ_EXIT(REQ_TC_WRITE));