Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
3 changes: 3 additions & 0 deletions messages/tape_linux_sg/root.txt
Original file line number Diff line number Diff line change
Expand Up @@ -138,6 +138,9 @@ root:table {
30295I:string { "Have unstable TUR response, start over (Cur = %d, Prev = %d)." }
30296I:string { "Capturing a stable TUR at line %d." }
30297W:string { "Cannot retrieve drive dump: failed to communicate with drive. Tried (%d) times." }
30298W:string { "Retrying %s operation due Power On-Reset event reached. Tried (%d) times." }
30299E:string { "Could not clear POR status, device might be faulty, ret: (%d)." }
30300E:string { "Could resolve block position after failure, ret: (%d)." }

30392D:string { "Backend %s %s." }
30393D:string { "Backend %s: %d %s." }
Expand Down
92 changes: 76 additions & 16 deletions src/tape_drivers/linux/sg/sg_tape.c
Original file line number Diff line number Diff line change
Expand Up @@ -54,6 +54,7 @@
#include <dirent.h>
#include <sys/ioctl.h>

#include "libltfs/ltfs_error.h"
#include "ltfs_copyright.h"
#include "libltfs/ltfslogging.h"
#include "libltfs/fs.h"
Expand Down Expand Up @@ -104,6 +105,7 @@ struct sg_global_data global_data;
#define MAX_RETRY (100)

#define MAX_TAKE_DUMP_ATTEMPTS (10)
#define POR_MAX_RETRIES (3)

/* Forward references (For keep function order to struct tape_ops) */
int sg_readpos(void *device, struct tc_position *pos);
Expand Down Expand Up @@ -604,7 +606,7 @@ int _raw_tur(const int fd)

#define _clear_por(p) _clear_por_raw((p)->dev.fd);

void _clear_por_raw(const int fd)
int _clear_por_raw(const int fd)
{
int i = 0, ret = -1;

Expand All @@ -624,6 +626,7 @@ void _clear_por_raw(const int fd)
}
i++;
}
return ret;
}

#define _get_stable_tur_response(p) _get_stable_tur_response_raw((p)->dev.fd)
Expand Down Expand Up @@ -1861,16 +1864,11 @@ static int _cdb_read(void *device, char *buf, size_t size, bool sili)
return length;
}

static inline int _handle_block_allocation_failure(void *device, struct tc_position *pos,
int *retry, char *op)
static inline int _resolve_position_after_io_cmd_failure(void *device, struct tc_position *pos, char *op)
{
int ret = 0;
struct tc_position tmp_pos = {0, 0};

/* Sleep 3 secs to wait garbage correction in kernel side and retry */
ltfsmsg(LTFS_WARN, 30277W, ++(*retry));
sleep(3);

ret = sg_readpos(device, &tmp_pos);
if (ret == DEVICE_GOOD && pos->partition == tmp_pos.partition) {
if (pos->block == tmp_pos.block) {
Expand Down Expand Up @@ -1923,11 +1921,12 @@ static inline int _handle_block_allocation_failure(void *device, struct tc_posit
int sg_read(void *device, char *buf, size_t size,
struct tc_position *pos, const bool unusual_size)
{
int32_t ret = -EDEV_UNKNOWN;
int32_t ret = -EDEV_UNKNOWN, ret_read = -1;
struct sg_data *priv = (struct sg_data*)device;
size_t datacount = size;
struct tc_position pos_retry = {0, 0};
int retry_count = 0;
int retry_count = 0, por_retry_count = 0;
struct timespec delay_ts = {0};

ltfs_profiler_add_entry(priv->profiler, NULL, TAPEBEND_REQ_ENTER(REQ_TC_READ));
ltfsmsg(LTFS_DEBUG3, 30395D, "read", size, priv->drive_serial);
Expand All @@ -1952,7 +1951,7 @@ int sg_read(void *device, char *buf, size_t size,
}

start_read:
ret = _cdb_read(device, buf, datacount, unusual_size);
ret_read = ret = _cdb_read(device, buf, datacount, unusual_size);
if (ret == -EDEV_LENGTH_MISMATCH) {
if (pos_retry.partition || pos_retry.block) {
/* Return error when retry is already executed */
Expand Down Expand Up @@ -1984,11 +1983,38 @@ int sg_read(void *device, char *buf, size_t size,
priv->use_sili = false;
ret = _cdb_read(device, buf, datacount, unusual_size);
} else if (ret == -EDEV_BUFFER_ALLOCATE_ERROR && retry_count < MAX_RETRY) {
ret = _handle_block_allocation_failure(device, pos, &retry_count, "read");
ltfsmsg(LTFS_WARN, 30277W, ++retry_count);
struct timespec delay_ts;
delay_ts.tv_sec = 3;
delay_ts.tv_nsec = 0;
nanosleep(&delay_ts, NULL); // Wait for kernel GC
ret = _resolve_position_after_io_cmd_failure(device, pos, "read");
if (ret == -EDEV_RETRY)
goto start_read;
} else if (ret == -EDEV_HOST_ERROR && por_retry_count < POR_MAX_RETRIES) {
ltfsmsg(LTFS_WARN, 30298W, "read", por_retry_count);
por_retry_count++;
delay_ts.tv_sec = 5;
delay_ts.tv_nsec = 0;
nanosleep(&delay_ts, NULL);
ret = _clear_por(priv);
if (ret == DEVICE_GOOD) {
ret = _resolve_position_after_io_cmd_failure(device, pos, "read");
/* If the original command did not reach the driver, or it reached it but after failing there is a position mismatch; retry */
if (ret == -EDEV_RETRY) {
goto start_read;
} else if (ret == 0) {
ret_read = datacount; // This is OK? Might need to set the block instead of that
} else {
ltfsmsg(LTFS_ERR, 30300E, ret);
}
} else {
ltfsmsg(LTFS_ERR, 30299E, ret);
}
ret = ret_read;
}


if(ret == -EDEV_FILEMARK_DETECTED)
{
pos->filemarks++;
Expand Down Expand Up @@ -2093,12 +2119,13 @@ static int _cdb_write(void *device, uint8_t *buf, size_t size, bool *ew, bool *p

int sg_write(void *device, const char *buf, size_t count, struct tc_position *pos)
{
int ret, ret_fo;
int ret, ret_fo, ret_write = -1;
bool ew = false, pew = false;
struct sg_data *priv = (struct sg_data*)device;
struct tc_position cur_pos;
size_t datacount = count;
int retry_count = 0;
int retry_count = 0, por_retry_count = 0;
struct timespec delay_ts = {0};

ltfs_profiler_add_entry(priv->profiler, NULL, TAPEBEND_REQ_ENTER(REQ_TC_WRITE));

Expand Down Expand Up @@ -2128,7 +2155,7 @@ int sg_write(void *device, const char *buf, size_t count, struct tc_position *po
}

start_write:
ret = _cdb_write(device, (uint8_t *)buf, datacount, &ew, &pew);
ret_write = ret = _cdb_write(device, (uint8_t *)buf, datacount, &ew, &pew);
if (ret == DEVICE_GOOD) {
pos->block++;
pos->early_warning = ew;
Expand All @@ -2146,9 +2173,34 @@ int sg_write(void *device, const char *buf, size_t count, struct tc_position *po
ret = -EDEV_POR_OR_BUS_RESET;
}
} else if (ret == -EDEV_BUFFER_ALLOCATE_ERROR && retry_count < MAX_RETRY) {
ret = _handle_block_allocation_failure(device, pos, &retry_count, "write");
ltfsmsg(LTFS_WARN, 30277W, ++retry_count);
delay_ts.tv_sec = 3;
delay_ts.tv_nsec = 0;
nanosleep(&delay_ts, NULL); // Wait for kernel GC
ret = _resolve_position_after_io_cmd_failure(device, pos, "write");
if (ret == -EDEV_RETRY)
goto start_write;
} else if (ret == -EDEV_HOST_ERROR && por_retry_count < POR_MAX_RETRIES) {
ltfsmsg(LTFS_WARN, 30298W, "write", por_retry_count);
por_retry_count++;
delay_ts.tv_sec = 5;
delay_ts.tv_nsec = 0;
nanosleep(&delay_ts, NULL);
ret = _clear_por(priv);
if (ret == DEVICE_GOOD) {
ret = _resolve_position_after_io_cmd_failure(device, pos, "write");
/* If the original command did not reach the driver, or it reached it but after failing there is a position mismatch; retry */
if (ret == -EDEV_RETRY) {
goto start_write;
} else if (ret == 0) {
ret_write = datacount; // This is OK? Might need to set the blocksize instead of that
} else {
ltfsmsg(LTFS_ERR, 30300E, ret);
}
} else {
ltfsmsg(LTFS_ERR, 30299E, ret);
}
ret = ret_write;
}

ltfs_profiler_add_entry(priv->profiler, NULL, TAPEBEND_REQ_EXIT(REQ_TC_WRITE));
Expand Down Expand Up @@ -3324,7 +3376,15 @@ int sg_modeselect(void *device, unsigned char *buf, const size_t size)

/* Build CDB */
cdb[0] = MODE_SELECT10;
cdb[1] = 0x10; /* Set PF bit */
/*
* Set PF and SP bit
* NOTE: Not all modepages support SP bit.

Copy link
Copy Markdown
Contributor

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

Now, it is affecting all modes pages, I think it would be useful to only enable SPI bit for READ_WRITE_CTRL page, which is the one that include Append-Only Mode (AOM) and Allow Overwitte modes as we talked.

Additionally, you already have a list of the modepages used by LTFS, where I think SP is supported, so this comment is a bit misleading, could you check it and add that mode pages list in the PR description?

* The SCSI reference says:
* • SP (Save Pages): Only allowed to be set to one when explicitly mentioned in the description of the
* specific mode page
* Right now SDE and LE does not set any non savable pages, but this could change.
* */
cdb[1] = 0x11;

Copy link
Copy Markdown
Contributor

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

@madjesc I believe it would be helpful to write a short comment about why we are enabling SP bit now

ltfs_u16tobe(cdb + 7, size);

timeout = get_timeout(priv->timeouts, cdb[0]);
Expand Down
Loading