diff --git a/Jenkinsfile b/Jenkinsfile index 3b0a8fe0c42..560a0eada9e 100644 --- a/Jenkinsfile +++ b/Jenkinsfile @@ -999,7 +999,7 @@ pipeline { inst_repos: daosRepos(), test_script: 'ci/unit/test_nlt.sh' + ' --system-ram-reserved 4' + - ' --max-log-size 1950MiB' + + ' --max-log-size 2500MiB' + ' --dfuse-dir /localhome/jenkins/' + ' --log-usage-save nltir.xml' + ' --log-usage-export nltr.json' + diff --git a/src/client/dfs/common.c b/src/client/dfs/common.c index 910a43f1ba8..6e5f7e774b0 100644 --- a/src/client/dfs/common.c +++ b/src/client/dfs/common.c @@ -326,7 +326,7 @@ fetch_entry(dfs_layout_ver_t ver, daos_handle_t oh, daos_handle_t th, const char static int remove_hardlink(dfs_t *dfs, daos_handle_t th, daos_handle_t parent_oh, const char *name, size_t len, - struct dfs_entry entry) + struct dfs_entry entry, bool *deleted) { daos_key_t dkey; daos_key_t git_dkey; @@ -373,9 +373,13 @@ remove_hardlink(dfs_t *dfs, daos_handle_t th, daos_handle_t parent_oh, const cha * directory entry still has its hardlink bit set, and the inode metadata * remains in GIT. */ - if (new_link_cnt > 0) + if (new_link_cnt > 0) { + if (deleted) + *deleted = false; D_GOTO(out, rc = git_update_link_cnt(dfs->git_oh, th, &entry.oid, new_link_cnt, NULL)); + } else if (deleted) + *deleted = true; d_iov_set(&git_dkey, &entry.oid, sizeof(daos_obj_id_t)); rc = daos_obj_punch_dkeys(dfs->git_oh, th, 0, 1, &git_dkey, NULL); @@ -415,14 +419,18 @@ remove_hardlink(dfs_t *dfs, daos_handle_t th, daos_handle_t parent_oh, const cha int remove_entry(dfs_t *dfs, daos_handle_t th, daos_handle_t parent_oh, const char *name, size_t len, - struct dfs_entry entry) + struct dfs_entry entry, bool *deleted) { daos_key_t dkey; daos_handle_t oh; int rc; + /* Default: the object is removed unless a surviving hardlink is found below. */ + if (deleted) + *deleted = true; + if (DFS_IS_HARDLINK(entry.mode)) - return remove_hardlink(dfs, th, parent_oh, name, len, entry); + return remove_hardlink(dfs, th, parent_oh, name, len, entry, deleted); if (S_ISLNK(entry.mode)) goto punch_entry; diff --git a/src/client/dfs/dfs_internal.h b/src/client/dfs/dfs_internal.h index 85a50be6e6e..4fde31217fd 100644 --- a/src/client/dfs/dfs_internal.h +++ b/src/client/dfs/dfs_internal.h @@ -437,7 +437,7 @@ fetch_entry(dfs_layout_ver_t ver, daos_handle_t oh, daos_handle_t th, const char void *xvals[], daos_size_t *xsizes); int remove_entry(dfs_t *dfs, daos_handle_t th, daos_handle_t parent_oh, const char *name, size_t len, - struct dfs_entry entry); + struct dfs_entry entry, bool *deleted); int git_fetch_entry(daos_handle_t git_oh, daos_handle_t th, daos_obj_id_t *oid, struct dfs_entry *entry, int xnr, char *xnames[], void *xvals[], daos_size_t *xsizes); diff --git a/src/client/dfs/dir.c b/src/client/dfs/dir.c index 99846c7f19f..154d1f71852 100644 --- a/src/client/dfs/dir.c +++ b/src/client/dfs/dir.c @@ -37,6 +37,8 @@ dfs_mkdir(dfs_t *dfs, dfs_obj_t *parent, const char *name, mode_t mode, daos_ocl if (rc) return rc; + mode = DFS_EXTERNAL_MODE(mode); + strncpy(new_dir.name, name, len + 1); rc = create_dir(dfs, parent, cid, &new_dir); @@ -124,7 +126,7 @@ remove_dir_contents(dfs_t *dfs, daos_handle_t th, struct dfs_entry entry) D_GOTO(out, rc); } - rc = remove_entry(dfs, th, oh, ptr, kds[i].kd_key_len, child_entry); + rc = remove_entry(dfs, th, oh, ptr, kds[i].kd_key_len, child_entry, NULL); if (rc) D_GOTO(out, rc); @@ -138,7 +140,8 @@ remove_dir_contents(dfs_t *dfs, daos_handle_t th, struct dfs_entry entry) } int -dfs_remove(dfs_t *dfs, dfs_obj_t *parent, const char *name, bool force, daos_obj_id_t *oid) +dfs_remove_internal(dfs_t *dfs, dfs_obj_t *parent, const char *name, bool force, daos_obj_id_t *oid, + bool *deleted) { struct dfs_entry entry = {0}; daos_handle_t th = DAOS_TX_NONE; @@ -208,7 +211,7 @@ dfs_remove(dfs_t *dfs, dfs_obj_t *parent, const char *name, bool force, daos_obj } } - rc = remove_entry(dfs, th, parent->oh, name, len, entry); + rc = remove_entry(dfs, th, parent->oh, name, len, entry, deleted); if (rc) D_GOTO(out, rc); @@ -232,6 +235,12 @@ dfs_remove(dfs_t *dfs, dfs_obj_t *parent, const char *name, bool force, daos_obj return rc; } +int +dfs_remove(dfs_t *dfs, dfs_obj_t *parent, const char *name, bool force, daos_obj_id_t *oid) +{ + return dfs_remove_internal(dfs, parent, name, force, oid, NULL); +} + int dfs_obj_set_oclass(dfs_t *dfs, dfs_obj_t *obj, int flags, daos_oclass_id_t cid) { diff --git a/src/client/dfs/obj.c b/src/client/dfs/obj.c index 428c7fc1bc2..e5e2d86ba7b 100644 --- a/src/client/dfs/obj.c +++ b/src/client/dfs/obj.c @@ -392,6 +392,8 @@ open_stat(dfs_t *dfs, dfs_obj_t *parent, const char *name, mode_t mode, int flag if (rc) return rc; + mode = DFS_EXTERNAL_MODE(mode); + /** default for newly created entries; fetch_entry/git_fetch_entry overwrite for existing */ entry.link_cnt = 1; @@ -1757,7 +1759,7 @@ dfs_osetattr(dfs_t *dfs, dfs_obj_t *obj, struct stat *stbuf, int flags) i++; flags &= ~DFS_SET_ATTR_MODE; - rstat.st_mode = stbuf->st_mode; + rstat.st_mode = DFS_EXTERNAL_MODE(stbuf->st_mode); } if (flags & DFS_SET_ATTR_ATIME) { flags &= ~DFS_SET_ATTR_ATIME; @@ -2220,11 +2222,6 @@ dfs_link(dfs_t *dfs, dfs_obj_t *obj, dfs_obj_t *parent, const char *new_name, df stbuf->st_mode = DFS_EXTERNAL_MODE(link_entry->mode); stbuf->st_uid = link_entry->uid; stbuf->st_gid = link_entry->gid; - stbuf->st_mtim.tv_sec = link_entry->mtime; - stbuf->st_mtim.tv_nsec = link_entry->mtime_nano; - stbuf->st_ctim.tv_sec = link_entry->ctime; - stbuf->st_ctim.tv_nsec = link_entry->ctime_nano; - stbuf->st_atim = stbuf->st_mtim; stbuf->st_blksize = link_entry->chunk_size ? link_entry->chunk_size : dfs->attr.da_chunk_size; @@ -2250,6 +2247,12 @@ dfs_link(dfs_t *dfs, dfs_obj_t *obj, dfs_obj_t *parent, const char *new_name, df if (!(new_obj && *new_obj)) daos_array_close(arr_oh, NULL); + /** Reconcile entry times with the array's modification epoch, as normal stat does. + */ + rc = update_stbuf_times(*link_entry, array_stbuf.st_max_epoch, stbuf, NULL); + if (rc) + D_GOTO(out, rc); + stbuf->st_atim = stbuf->st_mtim; stbuf->st_size = array_stbuf.st_size; stbuf->st_blocks = (array_stbuf.st_size + (1 << 9) - 1) >> 9; } diff --git a/src/client/dfs/rename.c b/src/client/dfs/rename.c index 5d237458a5a..464fff9470d 100644 --- a/src/client/dfs/rename.c +++ b/src/client/dfs/rename.c @@ -117,7 +117,7 @@ xattr_copy(daos_handle_t src_oh, const char *src_name, daos_handle_t dst_oh, con int dfs_move_internal(dfs_t *dfs, unsigned int flags, dfs_obj_t *parent, const char *name, dfs_obj_t *new_parent, const char *new_name, daos_obj_id_t *moid, - daos_obj_id_t *oid) + daos_obj_id_t *oid, bool *deleted) { struct dfs_entry entry = {0}, new_entry = {0}; daos_handle_t th = DAOS_TX_NONE; @@ -127,6 +127,9 @@ dfs_move_internal(dfs_t *dfs, unsigned int flags, dfs_obj_t *parent, const char size_t new_len; int rc; + if (deleted) + *deleted = false; + if (dfs == NULL || !dfs->mounted) return EINVAL; if (dfs->amode != O_RDWR) @@ -244,7 +247,7 @@ dfs_move_internal(dfs_t *dfs, unsigned int flags, dfs_obj_t *parent, const char D_GOTO(out, rc = ENOTEMPTY); } - rc = remove_entry(dfs, th, new_parent->oh, new_name, new_len, new_entry); + rc = remove_entry(dfs, th, new_parent->oh, new_name, new_len, new_entry, deleted); if (rc) { D_ERROR("Failed to remove entry %s (%d)\n", new_name, rc); D_GOTO(out, rc); @@ -256,7 +259,7 @@ dfs_move_internal(dfs_t *dfs, unsigned int flags, dfs_obj_t *parent, const char /** rename symlink */ if (S_ISLNK(entry.mode)) { - rc = remove_entry(dfs, th, parent->oh, name, len, entry); + rc = remove_entry(dfs, th, parent->oh, name, len, entry, NULL); if (rc) { D_ERROR("Failed to remove entry %s (%d)\n", name, rc); D_GOTO(out, rc); @@ -346,7 +349,7 @@ int dfs_move(dfs_t *dfs, dfs_obj_t *parent, const char *name, dfs_obj_t *new_parent, const char *new_name, daos_obj_id_t *oid) { - return dfs_move_internal(dfs, 0, parent, name, new_parent, new_name, NULL, oid); + return dfs_move_internal(dfs, 0, parent, name, new_parent, new_name, NULL, oid, NULL); } int diff --git a/src/client/dfuse/SConscript b/src/client/dfuse/SConscript index c699dd2acbb..347e80f4183 100644 --- a/src/client/dfuse/SConscript +++ b/src/client/dfuse/SConscript @@ -15,6 +15,7 @@ OPS_SRC = ['create', 'fgetattr', 'forget', 'getxattr', + 'link', 'listxattr', 'ioctl', 'lookup', diff --git a/src/client/dfuse/dfuse.h b/src/client/dfuse/dfuse.h index 45e9ab47d5f..6bf4bd4d8b8 100644 --- a/src/client/dfuse/dfuse.h +++ b/src/client/dfuse/dfuse.h @@ -400,6 +400,8 @@ struct dfuse_inode_ops { const char *newname, unsigned int flags); void (*symlink)(fuse_req_t req, const char *link, struct dfuse_inode_entry *parent, const char *name); + void (*hardlink)(fuse_req_t req, struct dfuse_inode_entry *inode, + struct dfuse_inode_entry *parent, const char *name); void (*unlink)(fuse_req_t req, struct dfuse_inode_entry *parent, const char *name); void (*setxattr)(fuse_req_t req, struct dfuse_inode_entry *inode, @@ -486,6 +488,7 @@ struct dfuse_pool { ACTION(UNLINK) \ ACTION(READDIR) \ ACTION(SYMLINK) \ + ACTION(LINK) \ ACTION(READLINK) \ ACTION(OPENDIR) \ ACTION(SETXATTR) \ @@ -941,6 +944,15 @@ dfuse_loop(struct dfuse_info *dfuse_info); #define DFUSE_REPLY_IOCTL(desc, req, arg) DFUSE_REPLY_IOCTL_SIZE(desc, req, &(arg), sizeof(arg)) +/* A (parent, name) dentry. A file may have multiple hardlinks, all sharing one inode entry; the + * primary link is stored inline on the inode and any additional links are tracked on ie_dentries. + */ +struct dfuse_dentry { + fuse_ino_t dd_parent; + char dd_name[NAME_MAX + 1]; + d_list_t dd_list; +}; + /** * Inode handle. * @@ -977,6 +989,12 @@ struct dfuse_inode_entry { */ fuse_ino_t ie_parent; + /** Additional hardlink dentries beyond the primary ie_parent/ie_name. */ + d_list_t ie_dentries; + + /** Protects ie_parent, ie_name and ie_dentries. */ + pthread_spinlock_t ie_dentry_lock; + struct dfuse_cont *ie_dfs; /** Hash table of inodes @@ -1189,12 +1207,6 @@ ival_drop_inode(struct dfuse_inode_entry *inode); int ival_update_inode(struct dfuse_inode_entry *inode, double timeout); -/* Queue an on-demand dentry invalidation (parent/name) to be issued from the invalidation thread. - * ie_drop, if non-NULL, is a reference that is released after the invalidation has been issued. - */ -int -dfuse_mark_inval_entry(fuse_ino_t parent, const char *name, struct dfuse_inode_entry *ie_drop); - int ival_init(struct dfuse_info *dfuse_info); @@ -1250,6 +1262,53 @@ dfuse_ie_init(struct dfuse_info *dfuse_info, struct dfuse_inode_entry *ie); void dfuse_ie_close(struct dfuse_info *dfuse_info, struct dfuse_inode_entry *ie); +/* Track an additional hardlink name for a shared inode. */ +int +dfuse_ie_dentry_add(struct dfuse_inode_entry *ie, fuse_ino_t parent, const char *name); + +/* Drop one tracked name; promotes a secondary to primary if the primary was removed. */ +void +dfuse_ie_dentry_remove(struct dfuse_inode_entry *ie, fuse_ino_t parent, const char *name); + +/* Rename a tracked name. If the old name is unknown, release all tracked names into released and + * set the new name as the sole primary. + */ +void +dfuse_ie_dentry_replace(struct dfuse_inode_entry *ie, fuse_ino_t old_parent, const char *old_name, + fuse_ino_t new_parent, const char *new_name, struct dfuse_dentry *released); + +/* Make (parent, name) the inode's sole tracked name, moving every other name into released. Used + * for a single-link file where all other cached names are stale. + */ +void +dfuse_ie_dentry_set_single(struct dfuse_inode_entry *ie, fuse_ino_t parent, const char *name, + struct dfuse_dentry *released); + +/* Snapshot every tracked name into released, moving secondaries out of the inode. */ +void +dfuse_ie_dentry_snapshot(struct dfuse_inode_entry *ie, struct dfuse_dentry *released); + +/* Issue fuse_lowlevel_notify_inval_entry() for every name in released and free secondaries. Must + * only be called from the invalidation thread. Returns -EBADF if the session is dead. + */ +int +dfuse_ie_dentry_inval(struct dfuse_info *dfuse_info, struct dfuse_dentry *released); + +/* Queue every name in released for invalidation on the invalidation thread, consuming released. + * ie_drop, if set, is released after the final name is invalidated. + */ +int +dfuse_queue_inval_dentries(struct dfuse_dentry *released, struct dfuse_inode_entry *ie_drop); + +/* Invalidate cached data/attrs synchronously, then queue a delete on the invalidation thread for + * every name in released, skipping (exclude_parent, exclude_name) which the kernel already handled. + * Consumes released. + */ +void +dfuse_ie_inode_delete(struct dfuse_info *dfuse_info, struct dfuse_inode_entry *ie, + struct dfuse_dentry *released, fuse_ino_t exclude_parent, + const char *exclude_name); + /* ops/...c */ void @@ -1312,6 +1371,9 @@ void dfuse_cb_symlink(fuse_req_t, const char *, struct dfuse_inode_entry *, const char *); +void +dfuse_cb_link(fuse_req_t, struct dfuse_inode_entry *, struct dfuse_inode_entry *, const char *); + void dfuse_cb_setxattr(fuse_req_t, struct dfuse_inode_entry *, const char *, const char *, size_t, int); @@ -1358,6 +1420,13 @@ void dfuse_oid_unlinked(struct dfuse_info *dfuse_info, fuse_req_t req, daos_obj_id_t *oid, struct dfuse_inode_entry *parent, const char *name); +/* Handle removal of one hardlink where the file still exists (other links remain). The inode is + * left intact; only the fuse reply is issued. + */ +void +dfuse_hardlink_removed(struct dfuse_info *dfuse_info, fuse_req_t req, daos_obj_id_t *oid, + struct dfuse_inode_entry *parent, const char *name); + /* dfuse_cont.c */ void dfuse_cont_lookup(fuse_req_t req, struct dfuse_inode_entry *parent, diff --git a/src/client/dfuse/dfuse_core.c b/src/client/dfuse/dfuse_core.c index bf898de7cbf..867e8d80604 100644 --- a/src/client/dfuse/dfuse_core.c +++ b/src/client/dfuse/dfuse_core.c @@ -1312,6 +1312,222 @@ dfuse_open_handle_init(struct dfuse_info *dfuse_info, struct dfuse_obj_hdl *oh, atomic_fetch_add_relaxed(&dfuse_info->di_fh_count, 1); } +/* Hardlink dentry tracking. An inode is keyed by its DAOS object id, so all hardlinks to the same + * file share a single inode entry. The primary (parent, name) lives in ie_parent/ie_name; any + * additional links are tracked on ie_dentries. All access is serialized by ie_dentry_lock. + */ +int +dfuse_ie_dentry_add(struct dfuse_inode_entry *ie, fuse_ino_t parent, const char *name) +{ + struct dfuse_dentry *dd; + struct dfuse_dentry *new_dd; + + /* Preallocate outside the spinlock. */ + D_ALLOC_PTR(new_dd); + if (new_dd == NULL) + return ENOMEM; + + D_SPIN_LOCK(&ie->ie_dentry_lock); + + if (ie->ie_name[0] == '\0') { + ie->ie_parent = parent; + strncpy(ie->ie_name, name, NAME_MAX); + ie->ie_name[NAME_MAX] = '\0'; + goto out; + } + + if (ie->ie_parent == parent && strncmp(ie->ie_name, name, NAME_MAX) == 0) + goto out; + + d_list_for_each_entry(dd, &ie->ie_dentries, dd_list) { + if (dd->dd_parent == parent && strncmp(dd->dd_name, name, NAME_MAX) == 0) + goto out; + } + + new_dd->dd_parent = parent; + strncpy(new_dd->dd_name, name, NAME_MAX); + new_dd->dd_name[NAME_MAX] = '\0'; + d_list_add_tail(&new_dd->dd_list, &ie->ie_dentries); + new_dd = NULL; + +out: + D_SPIN_UNLOCK(&ie->ie_dentry_lock); + if (new_dd != NULL) + D_FREE(new_dd); + return 0; +} + +void +dfuse_ie_dentry_remove(struct dfuse_inode_entry *ie, fuse_ino_t parent, const char *name) +{ + struct dfuse_dentry *dd; + struct dfuse_dentry *free_dd = NULL; + + D_SPIN_LOCK(&ie->ie_dentry_lock); + + if (ie->ie_parent == parent && strncmp(ie->ie_name, name, NAME_MAX) == 0) { + /* Promote the first secondary into the primary slot, or clear it. */ + if (!d_list_empty(&ie->ie_dentries)) { + dd = d_list_entry(ie->ie_dentries.next, struct dfuse_dentry, dd_list); + ie->ie_parent = dd->dd_parent; + strncpy(ie->ie_name, dd->dd_name, NAME_MAX); + ie->ie_name[NAME_MAX] = '\0'; + d_list_del(&dd->dd_list); + free_dd = dd; + } else { + ie->ie_parent = 0; + ie->ie_name[0] = '\0'; + } + goto out; + } + + d_list_for_each_entry(dd, &ie->ie_dentries, dd_list) { + if (dd->dd_parent == parent && strncmp(dd->dd_name, name, NAME_MAX) == 0) { + d_list_del(&dd->dd_list); + free_dd = dd; + goto out; + } + } + +out: + D_SPIN_UNLOCK(&ie->ie_dentry_lock); + if (free_dd != NULL) + D_FREE(free_dd); +} + +void +dfuse_ie_dentry_replace(struct dfuse_inode_entry *ie, fuse_ino_t old_parent, const char *old_name, + fuse_ino_t new_parent, const char *new_name, struct dfuse_dentry *released) +{ + struct dfuse_dentry *dd; + struct dfuse_dentry *dd_old = NULL; + struct dfuse_dentry *dd_new = NULL; + struct dfuse_dentry *free_dd = NULL; + + released->dd_parent = 0; + released->dd_name[0] = '\0'; + D_INIT_LIST_HEAD(&released->dd_list); + + D_SPIN_LOCK(&ie->ie_dentry_lock); + + if (ie->ie_parent == old_parent && strncmp(ie->ie_name, old_name, NAME_MAX) == 0) { + ie->ie_parent = new_parent; + strncpy(ie->ie_name, new_name, NAME_MAX); + ie->ie_name[NAME_MAX] = '\0'; + /* If the new name was already a secondary, collapse it into this primary. */ + d_list_for_each_entry(dd, &ie->ie_dentries, dd_list) { + if (dd->dd_parent == new_parent && + strncmp(dd->dd_name, new_name, NAME_MAX) == 0) { + d_list_del(&dd->dd_list); + free_dd = dd; + break; + } + } + goto out; + } + + /* Locate the secondary being renamed and any secondary that already holds the new name. */ + d_list_for_each_entry(dd, &ie->ie_dentries, dd_list) { + if (dd->dd_parent == old_parent && strncmp(dd->dd_name, old_name, NAME_MAX) == 0) + dd_old = dd; + else if (dd->dd_parent == new_parent && + strncmp(dd->dd_name, new_name, NAME_MAX) == 0) + dd_new = dd; + } + + if (dd_old != NULL) { + /* Drop the renamed entry if the new name is already tracked (primary or a + * secondary), otherwise rename it in place. + */ + if (dd_new != NULL || (ie->ie_parent == new_parent && + strncmp(ie->ie_name, new_name, NAME_MAX) == 0)) { + d_list_del(&dd_old->dd_list); + free_dd = dd_old; + } else { + dd_old->dd_parent = new_parent; + strncpy(dd_old->dd_name, new_name, NAME_MAX); + dd_old->dd_name[NAME_MAX] = '\0'; + } + goto out; + } + + /* Old name unknown: release every tracked name and set the new one as the sole primary. */ + released->dd_parent = ie->ie_parent; + strncpy(released->dd_name, ie->ie_name, NAME_MAX); + released->dd_name[NAME_MAX] = '\0'; + d_list_splice_init(&ie->ie_dentries, &released->dd_list); + + ie->ie_parent = new_parent; + strncpy(ie->ie_name, new_name, NAME_MAX); + ie->ie_name[NAME_MAX] = '\0'; + +out: + D_SPIN_UNLOCK(&ie->ie_dentry_lock); + if (free_dd != NULL) + D_FREE(free_dd); +} + +void +dfuse_ie_dentry_set_single(struct dfuse_inode_entry *ie, fuse_ino_t parent, const char *name, + struct dfuse_dentry *released) +{ + struct dfuse_dentry *dd, *ddn; + struct dfuse_dentry *free_dd = NULL; + bool kept = false; + + released->dd_parent = 0; + released->dd_name[0] = '\0'; + D_INIT_LIST_HEAD(&released->dd_list); + + D_SPIN_LOCK(&ie->ie_dentry_lock); + + /* The primary either survives as the single name or is released as stale. */ + if (ie->ie_name[0] != '\0') { + if (ie->ie_parent == parent && strncmp(ie->ie_name, name, NAME_MAX) == 0) { + kept = true; + } else { + released->dd_parent = ie->ie_parent; + strncpy(released->dd_name, ie->ie_name, NAME_MAX); + released->dd_name[NAME_MAX] = '\0'; + } + } + + /* Keep the matching secondary (if the primary was not the match); release the rest. */ + d_list_for_each_entry_safe(dd, ddn, &ie->ie_dentries, dd_list) { + if (!kept && dd->dd_parent == parent && strncmp(dd->dd_name, name, NAME_MAX) == 0) { + d_list_del(&dd->dd_list); + free_dd = dd; + kept = true; + continue; + } + d_list_del(&dd->dd_list); + d_list_add_tail(&dd->dd_list, &released->dd_list); + } + + ie->ie_parent = parent; + strncpy(ie->ie_name, name, NAME_MAX); + ie->ie_name[NAME_MAX] = '\0'; + + D_SPIN_UNLOCK(&ie->ie_dentry_lock); + if (free_dd != NULL) + D_FREE(free_dd); +} + +void +dfuse_ie_dentry_snapshot(struct dfuse_inode_entry *ie, struct dfuse_dentry *released) +{ + released->dd_parent = 0; + released->dd_name[0] = '\0'; + D_INIT_LIST_HEAD(&released->dd_list); + + D_SPIN_LOCK(&ie->ie_dentry_lock); + released->dd_parent = ie->ie_parent; + strncpy(released->dd_name, ie->ie_name, NAME_MAX); + released->dd_name[NAME_MAX] = '\0'; + d_list_splice_init(&ie->ie_dentries, &released->dd_list); + D_SPIN_UNLOCK(&ie->ie_dentry_lock); +} + void dfuse_ie_init(struct dfuse_info *dfuse_info, struct dfuse_inode_entry *ie) { @@ -1322,12 +1538,15 @@ dfuse_ie_init(struct dfuse_info *dfuse_info, struct dfuse_inode_entry *ie) atomic_init(&ie->ie_linear_read, true); atomic_fetch_add_relaxed(&dfuse_info->di_inode_count, 1); D_INIT_LIST_HEAD(&ie->ie_evict_entry); + D_INIT_LIST_HEAD(&ie->ie_dentries); D_RWLOCK_INIT(&ie->ie_wlock, 0); + D_SPIN_INIT(&ie->ie_dentry_lock, PTHREAD_PROCESS_PRIVATE); } void dfuse_ie_close(struct dfuse_info *dfuse_info, struct dfuse_inode_entry *ie) { + struct dfuse_dentry *dd; int rc; uint32_t ref; @@ -1362,6 +1581,13 @@ dfuse_ie_close(struct dfuse_info *dfuse_info, struct dfuse_inode_entry *ie) d_hash_rec_decref(dfp->dfp_cont_table, &dfc->dfs_entry); } + while (!d_list_empty(&ie->ie_dentries)) { + dd = d_list_entry(ie->ie_dentries.next, struct dfuse_dentry, dd_list); + d_list_del(&dd->dd_list); + D_FREE(dd); + } + D_SPIN_DESTROY(&ie->ie_dentry_lock); + dfuse_ie_free(dfuse_info, ie); } diff --git a/src/client/dfuse/dfuse_fuseops.c b/src/client/dfuse/dfuse_fuseops.c index d3183f480cc..ea0347c76a6 100644 --- a/src/client/dfuse/dfuse_fuseops.c +++ b/src/client/dfuse/dfuse_fuseops.c @@ -1,6 +1,6 @@ /** * (C) Copyright 2016-2024 Intel Corporation. - * (C) Copyright 2025 Hewlett Packard Enterprise Development LP + * (C) Copyright 2025-2026 Hewlett Packard Enterprise Development LP * * SPDX-License-Identifier: BSD-2-Clause-Patent */ @@ -398,6 +398,32 @@ df_ll_symlink(fuse_req_t req, const char *link, fuse_ino_t parent, const char *n DFUSE_REPLY_ERR_RAW(dfuse_info, req, rc); } +static void +df_ll_link(fuse_req_t req, fuse_ino_t ino, fuse_ino_t newparent, const char *newname) +{ + struct dfuse_info *dfuse_info = fuse_req_userdata(req); + struct dfuse_inode_entry *inode; + struct dfuse_inode_entry *parent_inode; + int rc; + + inode = dfuse_inode_lookup_nf(dfuse_info, ino); + parent_inode = dfuse_inode_lookup_nf(dfuse_info, newparent); + + if (inode->ie_dfs != parent_inode->ie_dfs) + D_GOTO(err, rc = EXDEV); + + if (!parent_inode->ie_dfs->dfs_ops->hardlink) + D_GOTO(err, rc = ENOTSUP); + + DFUSE_IE_STAT_ADD(parent_inode, DS_LINK); + + parent_inode->ie_dfs->dfs_ops->hardlink(req, inode, parent_inode, newname); + + return; +err: + DFUSE_REPLY_ERR_RAW(dfuse_info, req, rc); +} + /* Do not allow security xattrs to be set or read, see DAOS-14639 */ #define XATTR_SEC "security." /* Do not allow either system.posix_acl_default or system.posix_acl_access */ @@ -603,6 +629,7 @@ const struct dfuse_inode_ops dfuse_dfs_ops = { .create = dfuse_cb_create, .rename = dfuse_cb_rename, .symlink = dfuse_cb_symlink, + .hardlink = dfuse_cb_link, .setxattr = dfuse_cb_setxattr, .getxattr = dfuse_cb_getxattr, .listxattr = dfuse_cb_listxattr, @@ -638,6 +665,7 @@ const struct dfuse_inode_ops dfuse_pool_ops = { ACTION(mknod, df_ll_mknod, true) \ ACTION(rename, df_ll_rename, true) \ ACTION(symlink, df_ll_symlink, true) \ + ACTION(link, df_ll_link, true) \ ACTION(setxattr, df_ll_setxattr, true) \ ACTION(getxattr, df_ll_getxattr, false) \ ACTION(listxattr, df_ll_listxattr, false) \ diff --git a/src/client/dfuse/inval.c b/src/client/dfuse/inval.c index 505c6ecc862..3e8e2dbc714 100644 --- a/src/client/dfuse/inval.c +++ b/src/client/dfuse/inval.c @@ -98,29 +98,23 @@ struct dfuse_ival { }; /* A single on-demand dentry invalidation request. Request handlers running on the fixed worker - * pool enqueue these rather than calling fuse_lowlevel_notify_inval_entry() directly, as that - * blocks acquiring the parent's kernel i_rwsem, which may be held by a client waiting on the same - * worker pool - deadlocking it. The dedicated invalidation thread drains the queue instead. + * pool enqueue these rather than calling fuse_lowlevel_notify_inval_entry() or + * fuse_lowlevel_notify_delete() directly, as both block acquiring the parent's kernel i_rwsem, + * which may be held by a client waiting on the same worker pool - deadlocking it. The dedicated + * invalidation thread drains the queue instead. */ struct dfuse_inval_item { d_list_t link; fuse_ino_t parent; char name[NAME_MAX + 1]; + /* Child inode for a delete request; unused (0) for a plain invalidation. */ + fuse_ino_t ino; + /* When set, issue notify_delete(parent, ino, name) rather than notify_inval_entry(). */ + bool delete_entry; /* Optional inode reference to drop once the invalidation has completed, or NULL */ struct dfuse_inode_entry *ie_drop; }; -/* The core data from struct dfuse_inode_entry. No additional inode references are held on inodes - * because of there place on invalidate lists, rather inodes are removed from any list on close. - * Therefore once a decision is made to evict an inode then a copy of the data is needed as once - * the ival_lock is dropped the inode could be freed. This is not a problem if this happens as the - * kernel will simply return ENOENT. - */ -struct inode_core { - char name[NAME_MAX + 1]; - fuse_ino_t parent; -}; - /* Number of dentries to invalidate per iteration. This value affects how long the lock is held, * after the invalidations happen then another iteration will start immediately. Invalidation of * directories however trigger many forget calls so we want to make use of this where possible so @@ -145,7 +139,8 @@ static bool ival_loop(int *sleep_time) { struct dfuse_time_entry *dte, *dtep; - struct inode_core ic[EVICT_COUNT] = {}; + struct dfuse_dentry ic[EVICT_COUNT] = {}; + struct dfuse_dentry *dd, *ddn; int idx = 0; double sleep = (60 * 1) - 1; @@ -180,9 +175,8 @@ ival_loop(int *sleep_time) continue; } - ic[idx].parent = inode->ie_parent; - strncpy(ic[idx].name, inode->ie_name, NAME_MAX + 1); - ic[idx].name[NAME_MAX] = '\0'; + /* Snapshot every name so all hardlinks of the inode are invalidated. */ + dfuse_ie_dentry_snapshot(inode, &ic[idx]); d_list_del_init(&inode->ie_evict_entry); @@ -198,19 +192,27 @@ ival_loop(int *sleep_time) DFUSE_TRA_DEBUG(&ival_data, "Unlocking, allowing to sleep for %d seconds", *sleep_time); D_MUTEX_UNLOCK(&ival_lock); - if (idx == 0 || ival_data.session_dead) + if (ival_data.session_dead) { + /* Session is gone; free the snapshotted names without notifying the kernel. */ + for (int i = 0; i < idx; i++) { + d_list_for_each_entry_safe(dd, ddn, &ic[i].dd_list, dd_list) { + d_list_del(&dd->dd_list); + D_FREE(dd); + } + } + return false; + } + + if (idx == 0) return false; for (int i = 0; i < idx; i++) { int rc; - DFUSE_TRA_DEBUG(&ival_data, "Evicting entry %#lx " DF_DE, ic[i].parent, - DP_DE(ic[i].name)); + DFUSE_TRA_DEBUG(&ival_data, "Evicting entry %#lx " DF_DE, ic[i].dd_parent, + DP_DE(ic[i].dd_name)); - rc = fuse_lowlevel_notify_inval_entry(ival_data.session, ic[i].parent, ic[i].name, - strnlen(ic[i].name, NAME_MAX)); - if (rc && rc != -ENOENT && rc != -EBADF) - DHS_ERROR(&ival_data, -rc, "notify_inval_entry() failed"); + rc = dfuse_ie_dentry_inval(ival_data.dfuse_info, &ic[i]); if (rc == -EBADF) ival_data.session_dead = true; } @@ -218,32 +220,6 @@ ival_loop(int *sleep_time) return (idx == EVICT_COUNT); } -/* Queue a dentry invalidation for the invalidation thread. Takes ownership of ie_drop, which is - * released after the invalidation has been issued. On failure ie_drop is not touched. - */ -int -dfuse_mark_inval_entry(fuse_ino_t parent, const char *name, struct dfuse_inode_entry *ie_drop) -{ - struct dfuse_inval_item *item; - - D_ALLOC_PTR(item); - if (item == NULL) - return ENOMEM; - - item->parent = parent; - item->ie_drop = ie_drop; - strncpy(item->name, name, NAME_MAX); - item->name[NAME_MAX] = '\0'; - - D_MUTEX_LOCK(&ival_lock); - d_list_add_tail(&item->link, &ival_queue); - D_MUTEX_UNLOCK(&ival_lock); - - sem_post(&ival_sem); - - return 0; -} - /* Drain the on-demand invalidation queue. Runs on the invalidation thread so the blocking * fuse_lowlevel_notify_inval_entry() call is issued off the worker pool that services /dev/fuse. * Once shutdown has begun (ival_stop) items are freed without issuing notifies, as the worker pool @@ -266,11 +242,17 @@ ival_drain_queue(void) if (!ival_stop && !ival_data.session_dead) { int rc; - rc = fuse_lowlevel_notify_inval_entry(ival_data.session, item->parent, - item->name, - strnlen(item->name, NAME_MAX)); + if (item->delete_entry) + rc = fuse_lowlevel_notify_delete(ival_data.session, item->parent, + item->ino, item->name, + strnlen(item->name, NAME_MAX)); + else + rc = fuse_lowlevel_notify_inval_entry( + ival_data.session, item->parent, item->name, + strnlen(item->name, NAME_MAX)); if (rc && rc != -ENOENT && rc != -EBADF) - DHS_ERROR(&ival_data, -rc, "notify_inval_entry() failed"); + DHS_ERROR(&ival_data, -rc, "notify_%s() failed", + item->delete_entry ? "delete" : "inval_entry"); if (rc == -EBADF) ival_data.session_dead = true; } @@ -281,6 +263,176 @@ ival_drain_queue(void) } } +int +dfuse_ie_dentry_inval(struct dfuse_info *dfuse_info, struct dfuse_dentry *released) +{ + struct dfuse_dentry *dd, *ddn; + int rc; + int ret = 0; + + if (released->dd_name[0] != '\0') { + rc = fuse_lowlevel_notify_inval_entry(dfuse_info->di_session, released->dd_parent, + released->dd_name, + strnlen(released->dd_name, NAME_MAX)); + if (rc == -EBADF) + ret = -EBADF; + else if (rc != 0 && rc != -ENOENT) + DS_ERROR(-rc, "notify_inval_entry() failed"); + } + + d_list_for_each_entry_safe(dd, ddn, &released->dd_list, dd_list) { + rc = fuse_lowlevel_notify_inval_entry(dfuse_info->di_session, dd->dd_parent, + dd->dd_name, strnlen(dd->dd_name, NAME_MAX)); + if (rc == -EBADF) + ret = -EBADF; + else if (rc != 0 && rc != -ENOENT) + DS_ERROR(-rc, "notify_inval_entry() failed"); + d_list_del(&dd->dd_list); + D_FREE(dd); + } + + return ret; +} + +int +dfuse_queue_inval_dentries(struct dfuse_dentry *released, struct dfuse_inode_entry *ie_drop) +{ + struct dfuse_dentry *dd, *ddn; + struct dfuse_inval_item *item; + struct dfuse_inval_item *last = NULL; + d_list_t items; + int rc = 0; + + D_INIT_LIST_HEAD(&items); + + /* Build one queue item per tracked name, primary first then secondaries. */ + if (released->dd_name[0] != '\0') { + D_ALLOC_PTR(item); + if (item == NULL) + D_GOTO(fail, rc = ENOMEM); + item->parent = released->dd_parent; + item->ie_drop = NULL; + item->ino = 0; + item->delete_entry = false; + strncpy(item->name, released->dd_name, NAME_MAX); + item->name[NAME_MAX] = '\0'; + d_list_add_tail(&item->link, &items); + last = item; + } + + d_list_for_each_entry_safe(dd, ddn, &released->dd_list, dd_list) { + D_ALLOC_PTR(item); + if (item == NULL) + D_GOTO(fail, rc = ENOMEM); + item->parent = dd->dd_parent; + item->ie_drop = NULL; + item->ino = 0; + item->delete_entry = false; + strncpy(item->name, dd->dd_name, NAME_MAX); + item->name[NAME_MAX] = '\0'; + d_list_add_tail(&item->link, &items); + last = item; + d_list_del(&dd->dd_list); + D_FREE(dd); + } + + if (last != NULL) { + /* Drop the inode reference only after the final name is invalidated. */ + last->ie_drop = ie_drop; + D_MUTEX_LOCK(&ival_lock); + while ((item = d_list_pop_entry(&items, struct dfuse_inval_item, link)) != NULL) + d_list_add_tail(&item->link, &ival_queue); + D_MUTEX_UNLOCK(&ival_lock); + sem_post(&ival_sem); + } else if (ie_drop != NULL) { + /* Nothing to invalidate; release the reference immediately. */ + dfuse_inode_decref(ival_data.dfuse_info, ie_drop); + } + + return 0; + +fail: + while ((item = d_list_pop_entry(&items, struct dfuse_inval_item, link)) != NULL) + D_FREE(item); + d_list_for_each_entry_safe(dd, ddn, &released->dd_list, dd_list) { + d_list_del(&dd->dd_list); + D_FREE(dd); + } + if (ie_drop != NULL) + dfuse_inode_decref(ival_data.dfuse_info, ie_drop); + + return rc; +} + +void +dfuse_ie_inode_delete(struct dfuse_info *dfuse_info, struct dfuse_inode_entry *ie, + struct dfuse_dentry *released, fuse_ino_t exclude_parent, + const char *exclude_name) +{ + struct dfuse_dentry *dd, *ddn; + struct dfuse_inval_item *item; + fuse_ino_t ino = ie->ie_stat.st_ino; + d_list_t items; + int rc; + + D_INIT_LIST_HEAD(&items); + + /* Drop cached data and attributes (a no-op if caching is off). The kernel just did a + * lookup for this unlink/rename and has often destroyed the inode already, so this races + * and usually returns -ENOENT, which is expected and ignored. This operates on the inode + * itself rather than the parent, so it takes no parent i_rwsem and stays synchronous. + */ + rc = fuse_lowlevel_notify_inval_inode(dfuse_info->di_session, ino, 0, 0); + if (rc && rc != -ENOENT) + DHS_ERROR(ie, -rc, "inval_inode() error"); + + /* Queue a delete for every cached name so the kernel issues a forget for each, except + * (exclude_parent, exclude_name) which the kernel already handled and forgets on its own + * via this call. notify_delete() blocks acquiring the parent's i_rwsem, so it is deferred + * to the invalidation thread (see struct dfuse_inval_item). Delaying is safe because the + * kernel matches the child nodeid before deleting, so a name re-created in the meantime is + * left untouched. + */ + if (released->dd_name[0] != '\0' && + (released->dd_parent != exclude_parent || + strncmp(released->dd_name, exclude_name, NAME_MAX) != 0)) { + D_ALLOC_PTR(item); + if (item != NULL) { + item->parent = released->dd_parent; + item->ino = ino; + item->delete_entry = true; + strncpy(item->name, released->dd_name, NAME_MAX); + item->name[NAME_MAX] = '\0'; + d_list_add_tail(&item->link, &items); + } + } + + d_list_for_each_entry_safe(dd, ddn, &released->dd_list, dd_list) { + if (dd->dd_parent != exclude_parent || + strncmp(dd->dd_name, exclude_name, NAME_MAX) != 0) { + D_ALLOC_PTR(item); + if (item != NULL) { + item->parent = dd->dd_parent; + item->ino = ino; + item->delete_entry = true; + strncpy(item->name, dd->dd_name, NAME_MAX); + item->name[NAME_MAX] = '\0'; + d_list_add_tail(&item->link, &items); + } + } + d_list_del(&dd->dd_list); + D_FREE(dd); + } + + if (!d_list_empty(&items)) { + D_MUTEX_LOCK(&ival_lock); + while ((item = d_list_pop_entry(&items, struct dfuse_inval_item, link)) != NULL) + d_list_add_tail(&item->link, &ival_queue); + D_MUTEX_UNLOCK(&ival_lock); + sem_post(&ival_sem); + } +} + /* Main loop for eviction thread. Spins until ready for exit waking after one second and iterates * over all newly expired dentries. */ diff --git a/src/client/dfuse/ops/link.c b/src/client/dfuse/ops/link.c new file mode 100644 index 00000000000..8b7b9618e09 --- /dev/null +++ b/src/client/dfuse/ops/link.c @@ -0,0 +1,50 @@ +/** + * (C) Copyright 2026 Hewlett Packard Enterprise Development LP + * + * SPDX-License-Identifier: BSD-2-Clause-Patent + */ + +#include "dfuse_common.h" +#include "dfuse.h" + +void +dfuse_cb_link(fuse_req_t req, struct dfuse_inode_entry *inode, struct dfuse_inode_entry *parent, + const char *name) +{ + struct dfuse_info *dfuse_info = fuse_req_userdata(req); + struct dfuse_inode_entry *ie; + int rc; + + D_ALLOC_PTR(ie); + if (!ie) + D_GOTO(err, rc = ENOMEM); + + DFUSE_TRA_UP(ie, parent, "inode"); + + dfuse_ie_init(dfuse_info, ie); + + rc = dfs_link(parent->ie_dfs->dfs_ns, inode->ie_obj, parent->ie_obj, name, &ie->ie_obj, + &ie->ie_stat); + if (rc != 0) + D_GOTO(err, rc); + + DFUSE_TRA_DEBUG(ie, "obj is %p", ie->ie_obj); + + strncpy(ie->ie_name, name, NAME_MAX); + ie->ie_name[NAME_MAX] = '\0'; + ie->ie_parent = parent->ie_stat.st_ino; + ie->ie_dfs = parent->ie_dfs; + + dfs_obj2id(ie->ie_obj, &ie->ie_oid); + + dfuse_compute_inode(ie->ie_dfs, &ie->ie_oid, &ie->ie_stat.st_ino); + + dfuse_cache_evict_dir(dfuse_info, parent); + + dfuse_reply_entry(dfuse_info, ie, NULL, true, req); + + return; +err: + DFUSE_REPLY_ERR_RAW(parent, req, rc); + dfuse_ie_free(dfuse_info, ie); +} diff --git a/src/client/dfuse/ops/lookup.c b/src/client/dfuse/ops/lookup.c index 606008849c6..6412328edc3 100644 --- a/src/client/dfuse/ops/lookup.c +++ b/src/client/dfuse/ops/lookup.c @@ -18,10 +18,11 @@ dfuse_reply_entry(struct dfuse_info *dfuse_info, struct dfuse_inode_entry *ie, { struct fuse_entry_param entry = {0}; d_list_t *rlink; - ino_t wipe_parent = 0; - char wipe_name[NAME_MAX + 1]; + struct dfuse_dentry released = {0}; int rc; + D_INIT_LIST_HEAD(&released.dd_list); + D_ASSERT(ie->ie_parent); D_ASSERT(ie->ie_dfs); @@ -110,21 +111,21 @@ dfuse_reply_entry(struct dfuse_info *dfuse_info, struct dfuse_inode_entry *ie, if (ie->ie_stat.st_ino == ie->ie_dfs->dfs_ino) { DFUSE_TRA_DEBUG(inode, "Not updating parent"); - } else if ((inode->ie_parent != ie->ie_parent) || - (strncmp(inode->ie_name, ie->ie_name, NAME_MAX) != 0)) { - DFUSE_TRA_DEBUG(inode, "File has moved from " DF_DE " to " DF_DE, - DP_DE(inode->ie_name), DP_DE(ie->ie_name)); - - dfs_update_parent(inode->ie_obj, ie->ie_obj, ie->ie_name); - - /* Save the old name so that we can invalidate it in later */ - wipe_parent = inode->ie_parent; - strncpy(wipe_name, inode->ie_name, NAME_MAX); - wipe_name[NAME_MAX] = '\0'; - - inode->ie_parent = ie->ie_parent; - strncpy(inode->ie_name, ie->ie_name, NAME_MAX); - inode->ie_name[NAME_MAX] = '\0'; + } else { + if (S_ISREG(ie->ie_stat.st_mode) && ie->ie_stat.st_nlink > 1) { + /* Hardlink: track this additional name for the shared inode. */ + rc = dfuse_ie_dentry_add(inode, ie->ie_parent, ie->ie_name); + if (rc != 0) + DHS_ERROR(inode, rc, "dfuse_ie_dentry_add() failed"); + dfs_update_parent(inode->ie_obj, ie->ie_obj, ie->ie_name); + } else { + /* Single link: this is the only valid name, release any stale ones. + */ + dfuse_ie_dentry_set_single(inode, ie->ie_parent, ie->ie_name, + &released); + if (released.dd_name[0] != '\0') + dfs_update_parent(inode->ie_obj, ie->ie_obj, ie->ie_name); + } } atomic_fetch_sub_relaxed(&ie->ie_ref, 1); dfuse_ie_close(dfuse_info, ie); @@ -160,12 +161,7 @@ dfuse_reply_entry(struct dfuse_info *dfuse_info, struct dfuse_inode_entry *ie, DFUSE_REPLY_ENTRY(ie, req, entry); } - if (wipe_parent == 0) - return; - - rc = dfuse_mark_inval_entry(wipe_parent, wipe_name, NULL); - if (rc) - DS_ERROR(rc, "dfuse_mark_inval_entry() failed"); + dfuse_queue_inval_dentries(&released, NULL); return; out_err: diff --git a/src/client/dfuse/ops/open.c b/src/client/dfuse/ops/open.c index 580bef13a8a..bd0e6a154db 100644 --- a/src/client/dfuse/ops/open.c +++ b/src/client/dfuse/ops/open.c @@ -153,6 +153,7 @@ dfuse_cb_release(fuse_req_t req, fuse_ino_t ino, struct fuse_file_info *fi) struct dfuse_info *dfuse_info = fuse_req_userdata(req); struct dfuse_obj_hdl *oh = (struct dfuse_obj_hdl *)fi->fh; struct dfuse_inode_entry *ie = NULL; + struct dfuse_dentry released = {0}; int rc; uint32_t il_calls; @@ -242,11 +243,11 @@ dfuse_cb_release(fuse_req_t req, fuse_ino_t ino, struct fuse_file_info *fi) dfuse_inode_decref(dfuse_info, oh->doh_parent_dir); } if (ie) { - rc = dfuse_mark_inval_entry(ie->ie_parent, ie->ie_name, ie); - if (rc) { - DHS_ERROR(ie, rc, "dfuse_mark_inval_entry() failed"); - dfuse_inode_decref(dfuse_info, ie); - } + D_INIT_LIST_HEAD(&released.dd_list); + dfuse_ie_dentry_snapshot(ie, &released); + rc = dfuse_queue_inval_dentries(&released, ie); + if (rc) + DHS_ERROR(ie, rc, "dfuse_queue_inval_dentries() failed"); } dfuse_oh_free(dfuse_info, oh); } diff --git a/src/client/dfuse/ops/opendir.c b/src/client/dfuse/ops/opendir.c index dc80b04664d..deb3de9cd02 100644 --- a/src/client/dfuse/ops/opendir.c +++ b/src/client/dfuse/ops/opendir.c @@ -54,6 +54,8 @@ dfuse_cb_releasedir(fuse_req_t req, struct dfuse_inode_entry *ino, struct fuse_f struct dfuse_info *dfuse_info = fuse_req_userdata(req); struct dfuse_obj_hdl *oh = (struct dfuse_obj_hdl *)fi->fh; struct dfuse_inode_entry *ie = NULL; + struct dfuse_dentry released = {0}; + int rc; /* Perform the opposite of what the ioctl call does, always change the open handle count * but the inode only tracks number of open handles with non-zero ioctl counts @@ -82,13 +84,11 @@ dfuse_cb_releasedir(fuse_req_t req, struct dfuse_inode_entry *ino, struct fuse_f DFUSE_REPLY_ZERO_OH(oh, req); if (ie) { - int rc; - - rc = dfuse_mark_inval_entry(ie->ie_parent, ie->ie_name, ie); - if (rc) { - DHS_ERROR(ie, rc, "dfuse_mark_inval_entry() failed"); - dfuse_inode_decref(dfuse_info, ie); - } + D_INIT_LIST_HEAD(&released.dd_list); + dfuse_ie_dentry_snapshot(ie, &released); + rc = dfuse_queue_inval_dentries(&released, ie); + if (rc) + DHS_ERROR(ie, rc, "dfuse_queue_inval_dentries() failed"); } dfuse_oh_free(dfuse_info, oh); }; diff --git a/src/client/dfuse/ops/readdir.c b/src/client/dfuse/ops/readdir.c index 10ea454fd42..4bd93bd0c53 100644 --- a/src/client/dfuse/ops/readdir.c +++ b/src/client/dfuse/ops/readdir.c @@ -1,5 +1,6 @@ /** * (C) Copyright 2019-2024 Intel Corporation. + * (C) Copyright 2026 Hewlett Packard Enterprise Development LP * * SPDX-License-Identifier: BSD-2-Clause-Patent */ @@ -180,6 +181,7 @@ create_entry(struct dfuse_info *dfuse_info, struct dfuse_inode_entry *parent, st dfs_obj_t *obj, char *name, char *attr, daos_size_t attr_len, d_list_t **rlinkp) { struct dfuse_inode_entry *ie; + struct dfuse_dentry released = {0}; d_list_t *rlink; int rc = 0; @@ -240,15 +242,29 @@ create_entry(struct dfuse_info *dfuse_info, struct dfuse_inode_entry *parent, st /** update the chunk size and oclass of inode entry */ dfs_obj_copy_attr(inode->ie_obj, ie->ie_obj); + D_INIT_LIST_HEAD(&released.dd_list); + if (ie->ie_stat.st_ino == ie->ie_dfs->dfs_ino) { + /* Container root: its UNS name is fixed, nothing to update. */ DFUSE_TRA_DEBUG(inode, "Not updating parent"); - } else { + } else if (S_ISREG(ie->ie_stat.st_mode) && ie->ie_stat.st_nlink > 1) { + /* Hardlink: track this additional name for the shared inode. */ + rc = dfuse_ie_dentry_add(inode, ie->ie_parent, ie->ie_name); + if (rc != 0) + DFUSE_TRA_DEBUG(inode, "dfuse_ie_dentry_add() failed %d", rc); rc = dfs_update_parent(inode->ie_obj, ie->ie_obj, ie->ie_name); if (rc != 0) DFUSE_TRA_DEBUG(inode, "dfs_update_parent() failed %d", rc); + } else { + /* Single link: this is the only valid name, release any stale ones. */ + dfuse_ie_dentry_set_single(inode, ie->ie_parent, ie->ie_name, &released); + if (released.dd_name[0] != '\0') { + rc = dfs_update_parent(inode->ie_obj, ie->ie_obj, ie->ie_name); + if (rc != 0) + DFUSE_TRA_DEBUG(inode, "dfs_update_parent() failed %d", rc); + } } - inode->ie_parent = ie->ie_parent; - strncpy(inode->ie_name, ie->ie_name, NAME_MAX + 1); + dfuse_queue_inval_dentries(&released, NULL); atomic_fetch_sub_relaxed(&ie->ie_ref, 1); dfuse_ie_close(dfuse_info, ie); diff --git a/src/client/dfuse/ops/rename.c b/src/client/dfuse/ops/rename.c index 96fef8e2b8f..22e9013497f 100644 --- a/src/client/dfuse/ops/rename.c +++ b/src/client/dfuse/ops/rename.c @@ -19,8 +19,11 @@ dfuse_oid_moved(struct dfuse_info *dfuse_info, daos_obj_id_t *oid, struct dfuse_ const char *name, struct dfuse_inode_entry *newparent, const char *newname) { struct dfuse_inode_entry *ie; - int rc; + struct dfuse_dentry released = {0}; ino_t ino; + int rc; + + D_INIT_LIST_HEAD(&released.dd_list); dfuse_compute_inode(parent->ie_dfs, oid, &ino); @@ -30,23 +33,17 @@ dfuse_oid_moved(struct dfuse_info *dfuse_info, daos_obj_id_t *oid, struct dfuse_ if (!ie) return; - /* If the move is not from where we thought the file was then invalidate the old entry */ - if ((ie->ie_parent != parent->ie_stat.st_ino) || - (strncmp(ie->ie_name, name, NAME_MAX) != 0)) { - DFUSE_TRA_DEBUG(ie, "Invalidating old name"); - - rc = dfuse_mark_inval_entry(ie->ie_parent, ie->ie_name, NULL); - if (rc) - DFUSE_TRA_ERROR(ie, "dfuse_mark_inval_entry() failed: %d", rc); - } - - /* Update the inode entry data */ - ie->ie_parent = newparent->ie_stat.st_ino; - strncpy(ie->ie_name, newname, NAME_MAX); + /* Replace the moved name; if the old location was unknown, release all stale names. */ + dfuse_ie_dentry_replace(ie, parent->ie_stat.st_ino, name, newparent->ie_stat.st_ino, + newname, &released); /* Set the new parent and name */ dfs_update_parentfd(ie->ie_obj, newparent->ie_obj, newname); + rc = dfuse_queue_inval_dentries(&released, NULL); + if (rc) + DFUSE_TRA_ERROR(ie, "dfuse_queue_inval_dentries() failed: %d", rc); + /* Drop the ref again */ dfuse_inode_decref(dfuse_info, ie); } @@ -59,6 +56,7 @@ dfuse_cb_rename(fuse_req_t req, struct dfuse_inode_entry *parent, struct dfuse_info *dfuse_info = fuse_req_userdata(req); daos_obj_id_t moid = {}; daos_obj_id_t oid = {}; + bool deleted = false; int rc; if (flags != 0) { @@ -85,7 +83,7 @@ dfuse_cb_rename(fuse_req_t req, struct dfuse_inode_entry *parent, } rc = dfs_move_internal(parent->ie_dfs->dfs_ns, flags, parent->ie_obj, (char *)name, - newparent->ie_obj, (char *)newname, &moid, &oid); + newparent->ie_obj, (char *)newname, &moid, &oid, &deleted); if (rc) D_GOTO(out, rc); @@ -94,11 +92,18 @@ dfuse_cb_rename(fuse_req_t req, struct dfuse_inode_entry *parent, /* update moid */ dfuse_oid_moved(dfuse_info, &moid, parent, name, newparent, newname); - /* Check if a file was unlinked and see if anything needs updating */ - if (oid.lo || oid.hi) - dfuse_oid_unlinked(dfuse_info, req, &oid, newparent, newname); - else + /* Check if a file was unlinked and see if anything needs updating. A clobbered hardlink + * whose file still exists must keep its inode, so only treat it as fully unlinked when the + * object was actually deleted. + */ + if (oid.lo || oid.hi) { + if (deleted) + dfuse_oid_unlinked(dfuse_info, req, &oid, newparent, newname); + else + dfuse_hardlink_removed(dfuse_info, req, &oid, newparent, newname); + } else { DFUSE_REPLY_ZERO(newparent, req); + } return; diff --git a/src/client/dfuse/ops/setxattr.c b/src/client/dfuse/ops/setxattr.c index 7682e9af1d4..559a7ecea81 100644 --- a/src/client/dfuse/ops/setxattr.c +++ b/src/client/dfuse/ops/setxattr.c @@ -18,8 +18,9 @@ dfuse_cb_setxattr(fuse_req_t req, struct dfuse_inode_entry *inode, const char *name, const char *value, size_t size, int flags) { - int rc; - bool duns_attr = false; + struct dfuse_dentry released = {0}; + int rc; + bool duns_attr = false; DFUSE_TRA_DEBUG(inode, "Attribute '%s'", name); @@ -55,9 +56,11 @@ dfuse_cb_setxattr(fuse_req_t req, struct dfuse_inode_entry *inode, * locks that may be held by a client waiting on this worker pool, deadlocking it. */ if (duns_attr && inode->ie_dfs->dfc_dentry_dir_timeout > 0) { - rc = dfuse_mark_inval_entry(inode->ie_parent, inode->ie_name, NULL); + D_INIT_LIST_HEAD(&released.dd_list); + dfuse_ie_dentry_snapshot(inode, &released); + rc = dfuse_queue_inval_dentries(&released, NULL); if (rc) - DHS_ERROR(inode, rc, "dfuse_mark_inval_entry() failed"); + DHS_ERROR(inode, rc, "dfuse_queue_inval_dentries() failed"); } DFUSE_REPLY_ZERO(inode, req); return; diff --git a/src/client/dfuse/ops/unlink.c b/src/client/dfuse/ops/unlink.c index 778d44e0bed..e9529728985 100644 --- a/src/client/dfuse/ops/unlink.c +++ b/src/client/dfuse/ops/unlink.c @@ -1,5 +1,6 @@ /** * (C) Copyright 2016-2023 Intel Corporation. + * (C) Copyright 2026 Hewlett Packard Enterprise Development LP * * SPDX-License-Identifier: BSD-2-Clause-Patent */ @@ -7,6 +8,30 @@ #include "dfuse_common.h" #include "dfuse.h" +/* Handle removal of one hardlink where the file still exists because other links remain. The file + * object was not deleted, so unlike dfuse_oid_unlinked() the inode is left intact and not marked + * unlinked - just reply to the kernel, which already dropped the removed name. + */ +void +dfuse_hardlink_removed(struct dfuse_info *dfuse_info, fuse_req_t req, daos_obj_id_t *oid, + struct dfuse_inode_entry *parent, const char *name) +{ + struct dfuse_inode_entry *ie; + fuse_ino_t ino; + + dfuse_compute_inode(parent->ie_dfs, oid, &ino); + + ie = dfuse_inode_lookup(dfuse_info, ino); + if (ie) { + DFUSE_TRA_DEBUG(ie, "Hardlink " DF_DE " removed, file still exists", DP_DE(name)); + dfuse_ie_dentry_remove(ie, parent->ie_stat.st_ino, name); + dfuse_mcache_evict(ie); + dfuse_inode_decref(dfuse_info, ie); + } + + DFUSE_REPLY_ZERO(parent, req); +} + /* Handle a file that has been unlinked via dfuse. This means that either a unlink or rename call * caused the file to be deleted. * Takes the oid of the deleted file, and the parent/name where the delete happened. @@ -18,9 +43,11 @@ dfuse_oid_unlinked(struct dfuse_info *dfuse_info, fuse_req_t req, daos_obj_id_t struct dfuse_inode_entry *parent, const char *name) { struct dfuse_inode_entry *ie; - int rc; + struct dfuse_dentry released = {0}; fuse_ino_t ino; - ino_t parent_ino; + fuse_ino_t parent_ino = parent->ie_stat.st_ino; + + D_INIT_LIST_HEAD(&released.dd_list); dfuse_compute_inode(parent->ie_dfs, oid, &ino); @@ -34,35 +61,18 @@ dfuse_oid_unlinked(struct dfuse_info *dfuse_info, fuse_req_t req, daos_obj_id_t ie->ie_unlinked = true; - parent_ino = parent->ie_stat.st_ino; + /* Snapshot every cached name so all of them can be removed from the kernel. */ + dfuse_ie_dentry_snapshot(ie, &released); /* At this point the request is complete so the kernel is free to drop any refs on parent * so it should not be accessed. */ DFUSE_REPLY_ZERO(parent, req); - /* If caching is enabled then invalidate the data and attribute caches. As this came a - * unlink/rename call the kernel will have just done a lookup and knows what was likely - * unlinked so will destroy it anyway, but there is a race here so try and destroy it - * even though most of the time we expect this to fail. - */ - rc = fuse_lowlevel_notify_inval_inode(dfuse_info->di_session, ino, 0, 0); - if (rc && rc != -ENOENT) - DHS_ERROR(ie, -rc, "inval_inode() error"); - - /* If the kernel was aware of this inode at an old location then remove that which should - * trigger a forget call. Checking the test logs shows that we do see the forget anyway - * for cases where the kernel knows which file it deleted. + /* Invalidate cached data/attrs and delete every known name, except the one the kernel + * already handled via this unlink/rename. */ - if ((ie->ie_parent != parent_ino) || (strncmp(ie->ie_name, name, NAME_MAX) != 0)) { - DFUSE_TRA_DEBUG(ie, "Telling kernel to forget %#lx " DF_DE, ie->ie_parent, - DP_DE(ie->ie_name)); - - rc = fuse_lowlevel_notify_delete(dfuse_info->di_session, ie->ie_parent, ino, - ie->ie_name, strnlen(ie->ie_name, NAME_MAX)); - if (rc && rc != -ENOENT) - DHS_ERROR(ie, -rc, "notify_delete() error"); - } + dfuse_ie_inode_delete(dfuse_info, ie, &released, parent_ino, name); /* Drop the ref again */ dfuse_inode_decref(dfuse_info, ie); @@ -73,11 +83,13 @@ dfuse_cb_unlink(fuse_req_t req, struct dfuse_inode_entry *parent, const char *na { struct dfuse_info *dfuse_info = fuse_req_userdata(req); int rc; - daos_obj_id_t oid = {}; + daos_obj_id_t oid = {}; + bool deleted = true; dfuse_cache_evict_dir(dfuse_info, parent); - rc = dfs_remove(parent->ie_dfs->dfs_ns, parent->ie_obj, name, false, &oid); + rc = dfs_remove_internal(parent->ie_dfs->dfs_ns, parent->ie_obj, name, false, &oid, + &deleted); if (rc != 0) { DFUSE_REPLY_ERR_RAW(parent, req, rc); return; @@ -85,5 +97,10 @@ dfuse_cb_unlink(fuse_req_t req, struct dfuse_inode_entry *parent, const char *na D_ASSERT(oid.lo || oid.hi); + if (!deleted) { + dfuse_hardlink_removed(dfuse_info, req, &oid, parent, name); + return; + } + dfuse_oid_unlinked(dfuse_info, req, &oid, parent, name); } diff --git a/src/include/daos/dfs_lib_int.h b/src/include/daos/dfs_lib_int.h index 34837564d39..650795cebdd 100644 --- a/src/include/daos/dfs_lib_int.h +++ b/src/include/daos/dfs_lib_int.h @@ -56,14 +56,22 @@ dfs_lookupx(dfs_t *dfs, dfs_obj_t *parent, const char *name, int flags, dfs_obj_ mode_t *mode, struct stat *stbuf, int xnr, char *xnames[], void *xvals[], daos_size_t *xsizes); -/* moid is moved oid, oid is clobbered file. - * This isn't yet fully compatible with dfuse because we also want to pass in a flag for if the - * destination exists. +/* moid is the moved oid, oid is the clobbered file (if any). deleted, when non-NULL, reports + * whether the clobbered object was actually deleted (true for a regular file or the last hardlink, + * false when the clobbered name was one of several hardlinks and the file still exists). */ int dfs_move_internal(dfs_t *dfs, unsigned int flags, dfs_obj_t *parent, const char *name, dfs_obj_t *new_parent, const char *new_name, daos_obj_id_t *moid, - daos_obj_id_t *oid); + daos_obj_id_t *oid, bool *deleted); + +/* As dfs_remove() but reports whether the object was actually deleted. This is true for a regular + * file or when the last hardlink is removed, and false when only one of several hardlinks is + * removed and the file still exists. + */ +int +dfs_remove_internal(dfs_t *dfs, dfs_obj_t *parent, const char *name, bool force, daos_obj_id_t *oid, + bool *deleted); /* Set the in-memory parent, but takes the parent, rather than another file object */ void diff --git a/src/tests/ftest/dfuse/simul.py b/src/tests/ftest/dfuse/simul.py index 02380f36f50..aaf8e50dc8c 100644 --- a/src/tests/ftest/dfuse/simul.py +++ b/src/tests/ftest/dfuse/simul.py @@ -1,6 +1,6 @@ """ (C) Copyright 2018-2024 Intel Corporation. - (C) Copyright 2025 Hewlett Packard Enterprise Development LP + (C) Copyright 2025-2026 Hewlett Packard Enterprise Development LP SPDX-License-Identifier: BSD-2-Clause-Patent """ @@ -135,13 +135,10 @@ def test_posix_simul(self): :avocado: tags=PosixSimul,test_posix_simul """ # test 9, readdir, shared mode, dfuse returns NULL for readdir of an empty dir - # test 18, link, shared mode, daos does not support hard link # test 20, fcntl locking, shared mode, daos does not support flock # test 30, readdir, individual mode, dfuse returns NULL for readdir of an empty dir - # test 39, link, individual mode, daos does not support hard link - # test 40, link, individual mode, daos does not support hard link # test 41, fcntl locking, individual mode, daos does not support flock - self.run_simul(exclude="9,18,20,30,39,40,41") + self.run_simul(exclude="9,20,30,41") self.log.info('Test passed') def test_posix_expected_failures(self): @@ -152,6 +149,6 @@ def test_posix_expected_failures(self): :avocado: tags=posix,simul,dfuse :avocado: tags=PosixSimul,test_posix_expected_failures """ - faillist = {"9", "18", "20", "30", "39", "40", "41"} + faillist = {"9", "20", "30", "41"} self.run_simul(faillist=faillist) self.log.info('Test passed') diff --git a/src/tests/nlt/dfuse.py b/src/tests/nlt/dfuse.py index 8271cb2d68c..4f6bcaec8a1 100644 --- a/src/tests/nlt/dfuse.py +++ b/src/tests/nlt/dfuse.py @@ -354,7 +354,7 @@ def evict_and_wait(self, paths, qpath=None): rc = self.check_usage(inode, qpath=qpath) print(rc) found = rc['resident'] - if not found: + if found: sleeps += 1 assert sleeps < 10, 'Path still present 10 seconds after eviction' time.sleep(1) diff --git a/src/tests/nlt/posix_tests.py b/src/tests/nlt/posix_tests.py index b1c0745cf79..a8438afc0a0 100644 --- a/src/tests/nlt/posix_tests.py +++ b/src/tests/nlt/posix_tests.py @@ -711,6 +711,943 @@ def test_open_unlinked(self): os.unlink(fname) print(os.fstat(ofd.fileno())) + @needs_dfuse_with_opt(caching_variants=[False]) + def test_hardlink_basic(self): + """Create a hardlink and check both names share one inode. + + Exercises the dfuse hardlink op (dfuse_cb_link) and that a second name resolves to the + same shared inode with the correct link count and shared contents. + """ + base = self.dfuse.check_usage()['inodes'] + + fname = join(self.dfuse.dir, 'file') + lname = join(self.dfuse.dir, 'link') + with open(fname, 'w') as fd: + fd.write('hello') + + os.link(fname, lname) + + fstat = os.stat(fname) + lstat = os.stat(lname) + print(fstat) + print(lstat) + assert fstat.st_ino == lstat.st_ino, 'hardlink inode mismatch' + assert fstat.st_nlink == 2, f'unexpected nlink {fstat.st_nlink}' + assert lstat.st_nlink == 2, f'unexpected nlink {lstat.st_nlink}' + + # Both names share a single dfuse inode, so only one new inode is tracked. + self.dfuse.check_usage(inodes=base + 1) + + # Content is shared: read via the link, then rewrite via the original. + with open(lname, 'r') as fd: + assert fd.read() == 'hello' + with open(fname, 'w') as fd: + fd.write('world') + with open(lname, 'r') as fd: + assert fd.read() == 'world' + + # Linking onto a name that already exists must fail with EEXIST, exercising the + # dfuse_cb_link error reply. The existing names and shared inode are unaffected. + try: + os.link(fname, lname) + assert False, 'link onto an existing name should fail with EEXIST' + except FileExistsError: + pass + assert os.stat(fname).st_nlink == 2, 'failed link changed the link count' + assert os.stat(fname).st_ino == os.stat(lname).st_ino + + @needs_dfuse_with_opt(caching_variants=[False]) + def test_hardlink_unlink(self): + """Remove hardlinks one at a time and check the file survives until the last is gone. + + Exercises the one-link removal path (dfuse_hardlink_removed) versus the final removal + (dfuse_oid_unlinked / dfuse_ie_inode_delete). + """ + base = self.dfuse.check_usage()['inodes'] + + names = [join(self.dfuse.dir, f'name_{idx}') for idx in range(3)] + with open(names[0], 'w') as fd: + fd.write('data') + os.link(names[0], names[1]) + os.link(names[0], names[2]) + + assert os.stat(names[0]).st_nlink == 3 + + # All three names share a single dfuse inode. + self.dfuse.check_usage(inodes=base + 1) + + # Remove the first two links; the file remains and is readable via the survivors. + os.unlink(names[0]) + assert os.stat(names[1]).st_nlink == 2 + with open(names[2], 'r') as fd: + assert fd.read() == 'data' + + os.unlink(names[1]) + assert os.stat(names[2]).st_nlink == 1 + with open(names[2], 'r') as fd: + assert fd.read() == 'data' + + # The shared inode is a single object: evicting the surviving name drains it, so the + # inode count returns to the baseline. + self.dfuse.evict_and_wait([names[2]]) + self.dfuse.check_usage(inodes=base) + + # The file still exists on disk; the next access repopulates the inode, then the final + # unlink removes it for good. + assert os.stat(names[2]).st_nlink == 1 + os.unlink(names[2]) + try: + os.stat(names[2]) + assert False, 'file should be gone after last link removed' + except FileNotFoundError: + pass + + @needs_dfuse_with_opt(caching_variants=[False]) + def test_hardlink_rename(self): + """Check rename behavior for hardlinked files across several scenarios. + + Covers single-link vs hardlink rename disambiguation, clobbering one of several + hardlinks, moving a name across directories, and moving a secondary (non-primary) + hardlink name across directories. Caching is disabled so every stat reads authoritative + metadata. + """ + # Single link: rename is a move, the old name must disappear. + src = join(self.dfuse.dir, 'single') + dst = join(self.dfuse.dir, 'single_moved') + with open(src, 'w') as fd: + fd.write('one') + os.rename(src, dst) + assert os.stat(dst).st_nlink == 1 + try: + os.stat(src) + assert False, 'old name should be gone after move' + except FileNotFoundError: + pass + + # Hardlinked file: renaming one name keeps the other and the link count. + a = join(self.dfuse.dir, 'hl_a') + b = join(self.dfuse.dir, 'hl_b') + c = join(self.dfuse.dir, 'hl_c') + with open(a, 'w') as fd: + fd.write('two') + os.link(a, b) + assert os.stat(a).st_nlink == 2 + os.rename(b, c) + assert os.stat(a).st_nlink == 2 + assert os.stat(c).st_ino == os.stat(a).st_ino + with open(c, 'r') as fd: + assert fd.read() == 'two' + + # Rename-clobber: the overwritten destination is itself a hardlink, so the file object + # must not be marked unlinked while another link still refers to it + # (dfs_move_internal 'deleted' out-param -> dfuse_hardlink_removed vs dfuse_oid_unlinked). + clobber_src = join(self.dfuse.dir, 'clobber_src') + clobber_src_link = join(self.dfuse.dir, 'clobber_src_link') + with open(clobber_src, 'w') as fd: + fd.write('src') + os.link(clobber_src, clobber_src_link) + + victim = join(self.dfuse.dir, 'victim') + survivor = join(self.dfuse.dir, 'survivor') + with open(victim, 'w') as fd: + fd.write('victim') + os.link(victim, survivor) + assert os.stat(victim).st_nlink == 2 + + # Hold an open handle to the surviving name. If dfuse wrongly marks the still-live inode + # as unlinked, getattr returns stale cached attributes (link count 2) instead of the live 1. + fd = os.open(survivor, os.O_RDONLY) + try: + os.rename(clobber_src, victim) + + info = os.fstat(fd) + assert info.st_nlink == 1, f'clobbered hardlink wrongly reports nlink {info.st_nlink}' + assert os.stat(survivor).st_nlink == 1 + with open(survivor, 'r') as rfd: + assert rfd.read() == 'victim' + + # The moved file now answers to 'victim' and shares the source inode with its link. + assert os.stat(victim).st_ino == os.stat(clobber_src_link).st_ino + assert os.stat(victim).st_nlink == 2 + with open(victim, 'r') as rfd: + assert rfd.read() == 'src' + + # The old source name is gone. + try: + os.stat(clobber_src) + assert False, 'source name should be gone after rename' + except FileNotFoundError: + pass + finally: + os.close(fd) + + # Cross-directory rename of the primary name: moving one name of a hardlinked file into + # another directory must preserve the shared inode, link count and data, and leave the + # other name reachable (parent-change path and primary-name update). + move_sub = join(self.dfuse.dir, 'move_sub') + os.mkdir(move_sub) + move_a = join(self.dfuse.dir, 'move_a') + move_b = join(self.dfuse.dir, 'move_b') + move_dst = join(move_sub, 'move_dst') + with open(move_a, 'w') as fd: + fd.write('link') + os.link(move_a, move_b) + assert os.stat(move_a).st_nlink == 2 + + os.rename(move_a, move_dst) + + assert os.stat(move_dst).st_ino == os.stat(move_b).st_ino, \ + 'shared inode lost after cross-dir rename' + assert os.stat(move_dst).st_nlink == 2 + assert os.stat(move_b).st_nlink == 2 + with open(move_dst, 'r') as fd: + assert fd.read() == 'link' + with open(move_b, 'r') as fd: + assert fd.read() == 'link' + try: + os.stat(move_a) + assert False, 'old name should be gone after cross-dir rename' + except FileNotFoundError: + pass + assert 'move_dst' in os.listdir(move_sub) + + # Cross-directory rename of a secondary name: the shared inode tracks one primary name + # (ie_name) plus extra names as secondaries. Moving a secondary into another directory + # exercises the secondary-match branch of dfuse_ie_dentry_replace, updating the tracked + # dentry parent and name in place rather than the primary branch or the stale-name + # fallback. + sec_sub = join(self.dfuse.dir, 'sec_sub') + os.mkdir(sec_sub) + primary = join(self.dfuse.dir, 'primary') + secondary = join(self.dfuse.dir, 'secondary') + with open(primary, 'w') as fd: + fd.write('data') + os.link(primary, secondary) + + # Look up both names (caching is off, so each stat forces a fresh lookup) so the shared + # inode holds 'primary' in its primary slot and tracks 'secondary' as a secondary dentry + # before the rename. + assert os.stat(primary).st_ino == os.stat(secondary).st_ino, 'hardlink inode mismatch' + assert os.stat(primary).st_nlink == 2 + + sec_moved = join(sec_sub, 'sec_moved') + os.rename(secondary, sec_moved) + + assert os.stat(sec_moved).st_ino == os.stat(primary).st_ino, \ + 'shared inode lost after secondary cross-dir rename' + assert os.stat(sec_moved).st_nlink == 2 + assert os.stat(primary).st_nlink == 2 + with open(sec_moved, 'r') as fd: + assert fd.read() == 'data' + with open(primary, 'r') as fd: + assert fd.read() == 'data' + try: + os.stat(secondary) + assert False, 'old secondary name should be gone after rename' + except FileNotFoundError: + pass + assert 'sec_moved' in os.listdir(sec_sub) + + def test_hardlink_dir_cache(self): + """Check a hardlink invalidates the parent's cached directory listing. + + With caching on, dfuse keeps a shared in-memory readdir handle (ie_rd_hdl) for a directory + while any stream is open, and a newly opened stream is served from it. dfuse_cb_link() + must evict that handle (as create/unlink/rename do) or the new stream reuses the stale + cached listing and omits the new name. + """ + # Mount config mirrors the manual reproducer: caching on, write-back and data cache off, + # with a dentry timeout long enough that the readdir cache survives across the link. + cont_attrs = {'dfuse-data-cache': 'off', + 'dfuse-attr-time': '60', + 'dfuse-dentry-time': '60', + 'dfuse-ndentry-time': '60'} + self.container.set_attrs(cont_attrs) + dfuse = DFuse(self.server, self.conf, caching=True, wbcache=False, + container=self.container) + dfuse.start(v_hint='hardlink_dir_cache') + try: + directory = join(dfuse.dir, 'linkdir') + os.mkdir(directory) + seeds = ('seed0', 'seed1', 'seed2') + for name in seeds: + with open(join(directory, name), 'w'): + pass + source = join(directory, seeds[0]) + link_name = 'new-link' + + # Retain the directory stream: read every entry but break before exhausting the + # iterator, since os.scandir() auto-closes on exhaustion which would drop the shared + # dfuse readdir handle. Keeping it open holds that handle resident across the link. + held = os.scandir(directory) + try: + before = [] + for entry in held: + before.append(entry.name) + if len(before) == len(seeds): + break + assert link_name not in before + + os.link(source, join(directory, link_name)) + + # An independently opened stream shares the retained handle. Without the eviction + # in dfuse_cb_link it is served the stale cached listing and omits the new name. + after = set(os.listdir(directory)) + finally: + held.close() + + assert link_name in after, 'hardlink missing from an independently opened stream' + assert after == set(before) | {link_name} + finally: + if dfuse.stop(): + self.fatal_errors = True + + def test_hardlink_mtime(self): + """Check a hardlink does not move the cached mtime of the shared inode backward. + + dfs_link() builds its stat from stored inode metadata, which lags the array's modification + epoch. With attribute caching on and write-back off, publishing that stale mtime through + the dfuse attr cache would regress the modification time for both names. dfuse must reuse + the reconciled timestamp so an immediate stat reports the real mtime. + """ + # Mount config mirrors the manual reproducer: attribute caching on, write-back and data + # cache off, so the stat after the link is served from the attr cache the link publishes. + cont_attrs = {'dfuse-data-cache': 'off', + 'dfuse-attr-time': '60', + 'dfuse-dentry-time': '60', + 'dfuse-ndentry-time': '60'} + self.container.set_attrs(cont_attrs) + dfuse = DFuse(self.server, self.conf, caching=True, wbcache=False, + container=self.container) + dfuse.start(v_hint='hardlink_mtime') + try: + source = join(dfuse.dir, 'mtime-source') + target = join(dfuse.dir, 'mtime-link') + + with open(source, 'w') as fd: + fd.write('data') + + # Pin the stored entry mtime to the past, then a fresh write advances the modification + # epoch beyond it so the stored value and the reconciled mtime diverge. + old = 946684800 # 2000-01-01T00:00:00Z + os.utime(source, (old, old)) + with open(source, 'w') as fd: + fd.write('newer data') + + before = os.stat(source) + assert before.st_mtime_ns > old * 1000000000, 'write did not advance mtime' + + os.link(source, target) + + assert os.stat(source).st_mtime_ns == before.st_mtime_ns, \ + 'link moved the source mtime backward' + assert os.stat(target).st_mtime_ns == before.st_mtime_ns, \ + 'new hardlink name reports a stale mtime' + finally: + if dfuse.stop(): + self.fatal_errors = True + + def test_hardlink_two_mounts(self): + """Create a hardlink on one mount and check it is visible from a second mount. + + Also checks that removing one link out-of-band leaves the file reachable via the other + name. Both mounts run with caching disabled so metadata is always authoritative. + """ + dfuse0 = DFuse(self.server, self.conf, caching=False, container=self.container) + dfuse0.start(v_hint='hardlink_two_0') + + dfuse1 = DFuse(self.server, self.conf, caching=False, container=self.container) + dfuse1.start(v_hint='hardlink_two_1') + + base0 = dfuse0.check_usage()['inodes'] + + fname0 = join(dfuse0.dir, 'file') + lname0 = join(dfuse0.dir, 'link') + with open(fname0, 'w') as fd: + fd.write('shared') + os.link(fname0, lname0) + + # Both names share a single dfuse inode on the creating mount. + dfuse0.check_usage(inodes=base0 + 1) + + # The second mount sees both names, the shared inode and the link count. + fname1 = join(dfuse1.dir, 'file') + lname1 = join(dfuse1.dir, 'link') + assert os.stat(fname1).st_ino == os.stat(lname1).st_ino + assert os.stat(fname1).st_nlink == 2 + with open(lname1, 'r') as fd: + assert fd.read() == 'shared' + + # Remove one link on the second mount; the file remains via the other name on the first. + os.unlink(lname1) + assert os.stat(fname0).st_nlink == 1 + with open(fname0, 'r') as fd: + assert fd.read() == 'shared' + + if dfuse1.stop(): + self.fatal_errors = True + if dfuse0.stop(): + self.fatal_errors = True + + def test_hardlink_stale_notify_delete(self): + """Remove hardlink names out-of-band, then the last name locally, all in one directory. + + Two mounts share a container. Three hardlink names are created in the same directory on + the first mount and looked up so dfuse tracks them as a primary plus two secondaries. The + two secondary names are then removed out-of-band via the second mount, so the first mount + still holds stale dentries for names that no longer exist. Removing the surviving name + locally drives the final-removal path (dfuse_oid_unlinked -> dfuse_ie_inode_delete), which + must issue notify_delete for the stale secondary names in the same parent the kernel is + unlinking - a path the other hardlink tests do not reach. + + A long dentry cache is used so the stale names can only disappear from the first mount + because dfuse invalidated them, not because the cache timed out. + """ + cache_time = 30 + cont_attrs = {'dfuse-data-cache': False, + 'dfuse-attr-time': cache_time, + 'dfuse-dentry-time': cache_time, + 'dfuse-ndentry-time': cache_time} + self.container.set_attrs(cont_attrs) + + dfuse0 = DFuse(self.server, self.conf, caching=True, wbcache=False, + container=self.container) + dfuse0.start(v_hint='hardlink_stale_notify_0') + + dfuse1 = DFuse(self.server, self.conf, caching=False, container=self.container) + dfuse1.start(v_hint='hardlink_stale_notify_1') + + try: + names = ['file', 'link1', 'link2'] + paths0 = [join(dfuse0.dir, name) for name in names] + paths1 = [join(dfuse1.dir, name) for name in names] + + with open(paths0[0], 'w') as fd: + fd.write('data') + os.link(paths0[0], paths0[1]) + os.link(paths0[0], paths0[2]) + + # Look up every name on the first mount so the kernel caches the primary plus two + # secondaries and dfuse tracks them, all in the same parent directory. + assert len({os.stat(p).st_ino for p in paths0}) == 1, \ + 'hardlink names should share one inode' + assert os.stat(paths0[0]).st_nlink == 3 + + # Pin the shared inode with an open fd so the stale dentries stay resident on the + # first mount after the out-of-band removals below. + pin = os.open(paths0[0], os.O_RDONLY) + try: + # Remove the two secondary names out-of-band via the second mount. The first + # mount is not notified, so it keeps stale cached dentries for link1 and link2 - + # confirm they still resolve there from cache before the final unlink, otherwise + # the invalidation check below would prove nothing. + os.unlink(paths1[1]) + os.unlink(paths1[2]) + assert os.path.exists(paths0[1]), 'link1 should still be cached on the first mount' + assert os.path.exists(paths0[2]), 'link2 should still be cached on the first mount' + + # Remove the last surviving name locally. This is the final on-disk link, so it + # drives dfuse_oid_unlinked -> dfuse_ie_inode_delete, which issues notify_delete + # for the stale link1/link2 dentries in the same parent. + os.unlink(paths0[0]) + try: + os.stat(paths0[0]) + assert False, 'file should be gone after last link removed' + except FileNotFoundError: + pass + + # notify_delete is issued after the unlink reply, so poll briefly. The dentry + # cache is still well within its timeout, so the stale names can only become + # ENOENT because dfuse invalidated their cached dentries. + deadline = time.perf_counter() + 10 + while time.perf_counter() < deadline: + if not os.path.exists(paths0[1]) and not os.path.exists(paths0[2]): + break + time.sleep(0.5) + for stale in (paths0[1], paths0[2]): + try: + os.stat(stale) + assert False, f'stale dentry {stale} not invalidated by notify_delete' + except FileNotFoundError: + pass + finally: + os.close(pin) + + # The directory is empty on both mounts and dfuse survived the stale notify_delete. + assert os.listdir(dfuse0.dir) == [] + assert os.listdir(dfuse1.dir) == [] + finally: + if dfuse1.stop(): + self.fatal_errors = True + if dfuse0.stop(): + self.fatal_errors = True + + def test_hardlink_cache_expire(self): + """Check the invalidation thread drops every name of a hardlink inode on timeout. + + With caching on, all hardlink names of a closed file are cached and each pins the shared + inode. Once the dentry timeout (plus grace) elapses the proactive invalidation thread must + invalidate the primary name and every secondary name, so the kernel forgets the inode and + the dfuse inode table drains back to the baseline. If only the primary were invalidated a + secondary would keep the inode resident and the count would never return to the baseline. + """ + cache_time = 5 + cont_attrs = {'dfuse-data-cache': False, + 'dfuse-attr-time': cache_time, + 'dfuse-dentry-time': cache_time, + 'dfuse-ndentry-time': cache_time} + self.container.set_attrs(cont_attrs) + + dfuse = DFuse(self.server, self.conf, caching=True, wbcache=False, + container=self.container) + dfuse.start(v_hint='hardlink_expire') + + resident = None + base = dfuse.check_usage()['inodes'] + try: + names = [join(dfuse.dir, f'hl_{idx}') for idx in range(3)] + with open(names[0], 'w') as fd: + fd.write('data') + os.link(names[0], names[1]) + os.link(names[0], names[2]) + + # Populate the kernel cache for every name and confirm they share one inode. + inos = {os.stat(name).st_ino for name in names} + assert len(inos) == 1, 'hardlink names should share one inode' + dfuse.check_usage(inodes=base + 1) + + # Do not access the names again. Wait past the dentry timeout plus grace and poll + # until the invalidation thread has driven the forget that drains the shared inode. + resident = base + 1 + deadline = time.perf_counter() + 40 + while time.perf_counter() < deadline: + resident = dfuse.check_usage()['inodes'] + if resident == base: + break + time.sleep(1) + finally: + if dfuse.stop(): + self.fatal_errors = True + + assert resident == base, \ + f'hardlink inode not drained after timeout: {resident} != {base}' + + def test_hardlink_readdir(self): + """Discover hardlinks via readdir on a second mount. + + Links created on one mount are listed via readdirplus (os.scandir) on a second mount that + has never looked them up, exercising the readdir path that registers the extra hardlink + names for the shared inode (dfuse_cb_readdir -> dfuse_ie_dentry_add). The lookup-based + hardlink tests do not cover this path. + """ + dfuse0 = DFuse(self.server, self.conf, caching=False, container=self.container) + dfuse0.start(v_hint='hardlink_readdir_0') + + dfuse1 = DFuse(self.server, self.conf, caching=False, container=self.container) + dfuse1.start(v_hint='hardlink_readdir_1') + + names = ['file', 'link1', 'link2'] + with open(join(dfuse0.dir, names[0]), 'w') as fd: + fd.write('data') + os.link(join(dfuse0.dir, names[0]), join(dfuse0.dir, names[1])) + os.link(join(dfuse0.dir, names[0]), join(dfuse0.dir, names[2])) + + base1 = dfuse1.check_usage()['inodes'] + + # Pin the shared inode on the second mount so the usage check below is stable, then + # discover every name purely through readdirplus. The secondary names are registered by + # the readdir path, not by lookup. + fd = os.open(join(dfuse1.dir, names[0]), os.O_RDONLY) + try: + inodes = {} + nlinks = {} + with os.scandir(dfuse1.dir) as entries: + for entry in entries: + info = entry.stat() + inodes[entry.name] = entry.inode() + nlinks[entry.name] = info.st_nlink + + print(inodes) + print(nlinks) + for name in names: + assert name in inodes, f'{name} not returned by readdir' + assert nlinks[name] == 3, f'unexpected nlink {nlinks[name]} for {name}' + assert len({inodes[name] for name in names}) == 1, \ + 'hardlink names should share one inode' + + # All names were registered against a single shared inode discovered via readdir. + dfuse1.check_usage(inodes=base1 + 1) + finally: + os.close(fd) + + if dfuse1.stop(): + self.fatal_errors = True + if dfuse0.stop(): + self.fatal_errors = True + + @needs_dfuse_with_opt(caching_variants=[False]) + def test_hardlink_setattr(self): + """Attribute changes on one hardlink name are visible via every name. + + Exercises setattr/stat on a shared inode with multiple links: chmod and truncate via one + name must be reflected through the others, while the shared inode and link count stay + consistent. Caching is disabled so every stat reads authoritative metadata. + """ + a = join(self.dfuse.dir, 'hl_a') + b = join(self.dfuse.dir, 'hl_b') + with open(a, 'w') as fd: + fd.write('content') + os.link(a, b) + + assert os.stat(a).st_ino == os.stat(b).st_ino, 'hardlink inode mismatch' + assert os.stat(a).st_nlink == 2, f'unexpected nlink {os.stat(a).st_nlink}' + + # chmod via one name is visible via the other. + os.chmod(a, 0o600) + assert stat.S_IMODE(os.stat(b).st_mode) == 0o600 + os.chmod(b, 0o644) + assert stat.S_IMODE(os.stat(a).st_mode) == 0o644 + + # truncate via one name is visible via the other; the link count is unchanged. + os.truncate(a, 3) + assert os.stat(b).st_size == 3 + assert os.stat(a).st_nlink == 2 + with open(b, 'r') as fd: + assert fd.read() == 'con' + + # fchmod on an open handle is also visible via the other name. + with open(a, 'r') as fd: + os.fchmod(fd.fileno(), 0o640) + assert stat.S_IMODE(os.stat(b).st_mode) == 0o640 + + @needs_dfuse_with_opt(caching_variants=[False]) + def test_hardlink_across_dirs(self): + """Create a hardlink whose two names live in different directories. + + Every other hardlink test keeps both names in one directory, so the tracked dentries + always share a parent. Here the second name is created in a sibling directory, giving + the shared inode two dentries with different parents. Both names must resolve to one + inode, and evicting the survivor after one name is removed must drain that inode, proving + both per-parent dentries were cleaned up. + """ + dir_a = join(self.dfuse.dir, 'dir_a') + dir_b = join(self.dfuse.dir, 'dir_b') + os.mkdir(dir_a) + os.mkdir(dir_b) + + base = self.dfuse.check_usage()['inodes'] + + fname = join(dir_a, 'file') + lname = join(dir_b, 'link') + with open(fname, 'w') as fd: + fd.write('shared') + + # The link target and the new name hang off different parent directories. + os.link(fname, lname) + + fstat = os.stat(fname) + lstat = os.stat(lname) + print(fstat) + print(lstat) + assert fstat.st_ino == lstat.st_ino, 'hardlink inode mismatch across dirs' + assert fstat.st_nlink == 2, f'unexpected nlink {fstat.st_nlink}' + assert lstat.st_nlink == 2, f'unexpected nlink {lstat.st_nlink}' + + # Both names share a single dfuse inode even though they hang off different parents. + self.dfuse.check_usage(inodes=base + 1) + + # Content is shared across the two directories. + with open(lname, 'r') as fd: + assert fd.read() == 'shared' + with open(fname, 'w') as fd: + fd.write('updated') + with open(lname, 'r') as fd: + assert fd.read() == 'updated' + + # Remove the name in dir_b; the file survives via the name in dir_a. + os.unlink(lname) + assert os.stat(fname).st_nlink == 1 + with open(fname, 'r') as fd: + assert fd.read() == 'updated' + + # Evicting the survivor drains the shared inode back to the baseline, so both + # per-parent dentries were released. + self.dfuse.evict_and_wait([fname]) + self.dfuse.check_usage(inodes=base) + + # The file still exists; the next access repopulates it, then the final unlink removes it. + assert os.stat(fname).st_nlink == 1 + os.unlink(fname) + try: + os.stat(fname) + assert False, 'file should be gone after last link removed' + except FileNotFoundError: + pass + + def test_hardlink_caching(self): + """Exercise the hardlink life cycle with kernel caching enabled. + + Every other functional hardlink test runs with caching disabled, so the cached-metadata + paths never see multiple links. With caching on, each looked-up name pins the shared + inode and is cached in the kernel. This drives setattr, unlink and rename through that + cached state and then confirms the shared inode still drains once every name is gone, + exercising the snapshot-and-invalidate paths (dfuse_ie_dentry_snapshot -> + dfuse_queue_inval_dentries) that only run when caching keeps the extra names resident. + + """ + cache_time = 5 + cont_attrs = {'dfuse-data-cache': False, + 'dfuse-attr-time': cache_time, + 'dfuse-dentry-time': cache_time, + 'dfuse-ndentry-time': cache_time} + self.container.set_attrs(cont_attrs) + + dfuse = DFuse(self.server, self.conf, caching=True, wbcache=False, + container=self.container) + dfuse.start(v_hint='hardlink_caching') + + base = dfuse.check_usage()['inodes'] + try: + names = [join(dfuse.dir, f'hl_{idx}') for idx in range(3)] + with open(names[0], 'w') as fd: + fd.write('content') + os.link(names[0], names[1]) + os.link(names[0], names[2]) + + # Populate the kernel cache for every name; all share one dfuse inode. + assert len({os.stat(name).st_ino for name in names}) == 1, \ + 'hardlink names should share one inode' + assert os.stat(names[0]).st_nlink == 3 + dfuse.check_usage(inodes=base + 1) + + # setattr through the cache: a change via one cached name is visible via the others. + os.chmod(names[0], 0o600) + assert stat.S_IMODE(os.stat(names[1]).st_mode) == 0o600 + assert stat.S_IMODE(os.stat(names[2]).st_mode) == 0o600 + os.truncate(names[1], 3) + assert os.stat(names[0]).st_size == 3 + assert os.stat(names[2]).st_size == 3 + with open(names[2], 'r') as fd: + assert fd.read() == 'con' + + # unlink one cached name: the removed name is gone and the file survives via the + # others. The peer-name unlink evicts the metadata cache of the shared inode, so + # the follow-up GETATTR reports the refreshed link count. + os.unlink(names[0]) + try: + os.stat(names[0]) + assert False, 'unlinked name should be gone' + except FileNotFoundError: + pass + assert os.stat(names[1]).st_nlink == 2, 'stale nlink after peer-name unlink' + with open(names[1], 'r') as fd: + assert fd.read() == 'con' + with open(names[2], 'r') as fd: + assert fd.read() == 'con' + + # rename a cached name: the old name is gone and the new one shares the same inode. + renamed = join(dfuse.dir, 'hl_renamed') + os.rename(names[1], renamed) + try: + os.stat(names[1]) + assert False, 'old name should be gone after rename' + except FileNotFoundError: + pass + assert os.stat(renamed).st_ino == os.stat(names[2]).st_ino + with open(renamed, 'r') as fd: + assert fd.read() == 'con' + + # Still a single shared inode after all the cached-state mutations. + dfuse.check_usage(inodes=base + 1) + + # Remove one of the two surviving names. The peer-name unlink evicts the metadata + # cache of the shared inode, so the survivor link count refreshes to 1 on the next + # GETATTR without needing an eviction. + os.unlink(renamed) + assert os.stat(names[2]).st_nlink == 1, 'stale nlink after peer-name unlink' + + # Evict the last name so the shared inode drains to the baseline, confirming every + # cached name was invalidated. + dfuse.evict_and_wait([names[2]]) + dfuse.check_usage(inodes=base) + + # The final unlink removes the file for good. + os.unlink(names[2]) + try: + os.stat(names[2]) + assert False, 'file should be gone after last link removed' + except FileNotFoundError: + pass + finally: + if dfuse.stop(): + self.fatal_errors = True + + def test_hardlink_stale_name(self): + """Operate on the surviving name of a hardlink after the other name is removed. + + A dfuse inode holds a single dfs_obj whose (parent, name) is refreshed on every lookup. + Several DFS calls (dfs_osetattr, dfs_link, dfs_ostatx) still resolve that object by its + cached name, so if the object is left pointing at a name that has just been removed those + calls fail with ENOENT on an inode that is still perfectly alive. This reproduces the + regression where 'ln a b; rm b; ' returned ENOENT. + + Every operation is run twice: once removing the secondary name and operating on the + primary survivor, and once removing the primary name and operating on the secondary + survivor. The whole matrix runs under two mounts: caching off (every stat is a fresh + lookup, so the lookup of the doomed name just before it is removed points the shared + object at the stale name) and caching on (looked-up names stay resident, so the shared + object is not re-allocated and reliably retains the stale name). + """ + cache_time = 30 + cont_attrs = {'dfuse-data-cache': False, + 'dfuse-attr-time': cache_time, + 'dfuse-dentry-time': cache_time, + 'dfuse-ndentry-time': cache_time} + self.container.set_attrs(cont_attrs) + + for caching in (False, True): + dfuse = DFuse(self.server, self.conf, caching=caching, wbcache=False, + container=self.container) + dfuse.start(v_hint=f'hardlink_stale_{"on" if caching else "off"}') + try: + self._hardlink_stale_name(dfuse.dir) + finally: + if dfuse.stop(): + self.fatal_errors = True + + @staticmethod + def _hardlink_stale_name(root): + """Run the stale-name matrix (chmod/truncate/ln/mv x both removal patterns) under root.""" + def op_chmod(target): + os.chmod(target, 0o600) + assert stat.S_IMODE(os.stat(target).st_mode) == 0o600, 'chmod not applied to survivor' + return target + + def op_truncate(target): + os.truncate(target, 1) + assert os.stat(target).st_size == 1, 'truncate not applied to survivor' + return target + + def op_link(target): + extra = f'{target}_c' + os.link(target, extra) + assert os.stat(target).st_ino == os.stat(extra).st_ino, 'new link does not share inode' + os.unlink(extra) + return target + + def run_rename_over(label): + # Renaming a fresh file over one hardlink name must drop that name and leave the + # survivor with an accurate link count. + + # Pattern 1: echo hi > file1; ln file1 file2; echo bye > file3; mv file3 file2; + # stat file1 + file1 = join(root, f'{label}_1_file1') + file2 = join(root, f'{label}_1_file2') + file3 = join(root, f'{label}_1_file3') + with open(file1, 'w') as fd: + fd.write('hi') + os.link(file1, file2) + assert os.stat(file1).st_nlink == 2, 'link count wrong after creating hardlink' + with open(file3, 'w') as fd: + fd.write('bye') + # Look up the doomed name last so the shared object caches 'file2' as its name. + os.stat(file1) + os.stat(file2) + # Renaming file3 onto file2 removes the old file2, so file1 drops back to one name. + os.rename(file3, file2) + assert os.stat(file1).st_nlink == 1, 'stale nlink on survivor after rename removed name' + os.unlink(file1) + os.unlink(file2) + + # Pattern 2: echo hi > file1; ln file1 file2; echo bye > file3; mv file3 file1; + # stat file2 + file1 = join(root, f'{label}_2_file1') + file2 = join(root, f'{label}_2_file2') + file3 = join(root, f'{label}_2_file3') + with open(file1, 'w') as fd: + fd.write('hi') + os.link(file1, file2) + assert os.stat(file2).st_nlink == 2, 'link count wrong after creating hardlink' + with open(file3, 'w') as fd: + fd.write('bye') + # Look up the doomed name last so the shared object caches 'file1' as its name. + os.stat(file2) + os.stat(file1) + # Renaming file3 onto file1 removes the old file1, so file2 drops back to one name. + os.rename(file3, file1) + assert os.stat(file2).st_nlink == 1, 'stale nlink on survivor after rename removed name' + os.unlink(file1) + os.unlink(file2) + + def run(op, label): + # Pattern 1: echo hi > a; ln a b; rm b; ; rm a + a = join(root, f'{label}_1_a') + b = join(root, f'{label}_1_b') + with open(a, 'w') as fd: + fd.write('hi') + os.link(a, b) + # Look up the doomed name last so the shared object caches 'b' as its name. + os.stat(a) + assert os.stat(b).st_ino == os.stat(a).st_ino, 'hardlink inode mismatch' + os.unlink(b) + os.unlink(op(a)) + + # Pattern 2: echo hi > a; ln a b; rm a; ; rm b + a = join(root, f'{label}_2_a') + b = join(root, f'{label}_2_b') + with open(a, 'w') as fd: + fd.write('hi') + os.link(a, b) + # Look up the doomed name last so the shared object caches 'a' as its name. + os.stat(b) + assert os.stat(a).st_ino == os.stat(b).st_ino, 'hardlink inode mismatch' + os.unlink(a) + os.unlink(op(b)) + + run(op_chmod, 'chmod') + run(op_truncate, 'truncate') + run(op_link, 'link') + run_rename_over('rename') + + def test_hardlink_cross_container(self): + """Reject a hardlink whose two names live in different containers with EXDEV. + + A pool-level dfuse mount (no container) exposes every container in the pool as a + top-level directory, each backed by its own dfs mount (a distinct ie_dfs). df_ll_link() + must reject a hardlink whose source and destination resolve to different containers with + EXDEV before ever calling into DFS: object ids are container-scoped, so a cross-container + link is meaningless and would otherwise fail later with a misleading ENOENT. + """ + cont0 = create_cont(self.conf, self.pool, ctype="POSIX", label='hardlink_xcont_0') + cont1 = create_cont(self.conf, self.pool, ctype="POSIX", label='hardlink_xcont_1') + + dfuse = DFuse(self.server, self.conf, caching=False, pool=self.pool) + dfuse.start(v_hint='hardlink_cross_cont') + + try: + # A pool-level mount resolves container sub-directories by UUID only (dfuse_cont.c + # rejects non-UUID names), so address the containers by uuid rather than label. + fname = join(dfuse.dir, cont0.uuid, 'file') + lname = join(dfuse.dir, cont1.uuid, 'link') + with open(fname, 'w') as fd: + fd.write('data') + + # Linking across two containers must fail with EXDEV and must not create the target. + try: + os.link(fname, lname) + assert False, 'cross-container link should fail with EXDEV' + except OSError as error: + assert error.errno == errno.EXDEV, \ + f'cross-container link failed with {os.strerror(error.errno)}, expected EXDEV' + + assert not os.path.exists(lname), 'cross-container link target should not exist' + assert os.stat(fname).st_nlink == 1, 'source link count changed after failed link' + finally: + if dfuse.stop(): + self.fatal_errors = True + cont0.destroy() + cont1.destroy() + @needs_dfuse def test_chown_self(self): """Test that a file can be chowned to the current user, but not to other users""" diff --git a/src/tests/suite/dfs_unit_test.c b/src/tests/suite/dfs_unit_test.c index e4d3a772af8..2a394866328 100644 --- a/src/tests/suite/dfs_unit_test.c +++ b/src/tests/suite/dfs_unit_test.c @@ -5956,6 +5956,219 @@ dfs_test_orphan_name_hardlink(void **state) assert_int_equal(rc, 0); } +/* + * Test the "deleted" out-parameter of dfs_remove_internal(), which reports whether the underlying + * file object was actually destroyed. It must be false while other hardlinks remain, and true for + * a regular file, the last surviving hardlink, a directory, or a symlink. + */ +static void +dfs_test_remove_deleted_flag(void **state) +{ + test_arg_t *arg = *state; + dfs_obj_t *dir; + dfs_obj_t *reg_obj; + dfs_obj_t *l1_obj, *l2_obj, *l3_obj; + dfs_obj_t *sub_dir; + dfs_obj_t *sym_obj; + struct stat stbuf; + daos_obj_id_t oid; + bool deleted; + int rc; + + if (arg->myrank != 0) + return; + + rc = dfs_open(dfs_mt, NULL, "df_dir", S_IFDIR | S_IWUSR | S_IRUSR, + O_RDWR | O_CREAT | O_EXCL, 0, 0, NULL, &dir); + assert_int_equal(rc, 0); + + /* A regular (non-hardlinked) file: removal destroys the object. */ + print_message("Step 1: remove a regular file -> deleted == true\n"); + rc = dfs_open(dfs_mt, dir, "reg", S_IFREG | S_IWUSR | S_IRUSR, O_RDWR | O_CREAT | O_EXCL, 0, + 0, NULL, ®_obj); + assert_int_equal(rc, 0); + rc = dfs_release(reg_obj); + assert_int_equal(rc, 0); + deleted = false; + rc = dfs_remove_internal(dfs_mt, dir, "reg", false, &oid, &deleted); + assert_int_equal(rc, 0); + assert_true(deleted); + + /* Build a hardlink chain of three names (link_cnt == 3). */ + print_message("Step 2: create l1 and hardlinks l2, l3 (link_cnt == 3)\n"); + rc = dfs_open(dfs_mt, dir, "l1", S_IFREG | S_IWUSR | S_IRUSR, O_RDWR | O_CREAT | O_EXCL, 0, + 0, NULL, &l1_obj); + assert_int_equal(rc, 0); + rc = dfs_link(dfs_mt, l1_obj, dir, "l2", &l2_obj, NULL); + assert_int_equal(rc, 0); + rc = dfs_link(dfs_mt, l1_obj, dir, "l3", &l3_obj, &stbuf); + assert_int_equal(rc, 0); + assert_int_equal((int)stbuf.st_nlink, 3); + + /* Removing a link while others remain must report deleted == false. */ + print_message("Step 3: remove l1 -> deleted == false, nlink == 2\n"); + deleted = true; + rc = dfs_remove_internal(dfs_mt, dir, "l1", false, &oid, &deleted); + assert_int_equal(rc, 0); + assert_false(deleted); + rc = dfs_stat(dfs_mt, dir, "l2", &stbuf); + assert_int_equal(rc, 0); + assert_int_equal((int)stbuf.st_nlink, 2); + + print_message("Step 4: remove l2 -> deleted == false, nlink == 1\n"); + deleted = true; + rc = dfs_remove_internal(dfs_mt, dir, "l2", false, &oid, &deleted); + assert_int_equal(rc, 0); + assert_false(deleted); + rc = dfs_stat(dfs_mt, dir, "l3", &stbuf); + assert_int_equal(rc, 0); + assert_int_equal((int)stbuf.st_nlink, 1); + + /* Removing the last surviving link destroys the object. */ + print_message("Step 5: remove l3 (last link) -> deleted == true\n"); + deleted = false; + rc = dfs_remove_internal(dfs_mt, dir, "l3", false, &oid, &deleted); + assert_int_equal(rc, 0); + assert_true(deleted); + rc = dfs_stat(dfs_mt, dir, "l3", &stbuf); + assert_int_equal(rc, ENOENT); + + rc = dfs_release(l1_obj); + assert_int_equal(rc, 0); + rc = dfs_release(l2_obj); + assert_int_equal(rc, 0); + rc = dfs_release(l3_obj); + assert_int_equal(rc, 0); + + /* A directory: removal reports deleted == true. */ + print_message("Step 6: remove a directory -> deleted == true\n"); + rc = dfs_open(dfs_mt, dir, "sub", S_IFDIR | S_IWUSR | S_IRUSR, O_RDWR | O_CREAT | O_EXCL, 0, + 0, NULL, &sub_dir); + assert_int_equal(rc, 0); + rc = dfs_release(sub_dir); + assert_int_equal(rc, 0); + deleted = false; + rc = dfs_remove_internal(dfs_mt, dir, "sub", false, &oid, &deleted); + assert_int_equal(rc, 0); + assert_true(deleted); + + /* A symlink: removal reports deleted == true. */ + print_message("Step 7: remove a symlink -> deleted == true\n"); + rc = dfs_open(dfs_mt, dir, "sym", S_IFLNK | S_IWUSR | S_IRUSR, O_RDWR | O_CREAT | O_EXCL, 0, + 0, "target", &sym_obj); + assert_int_equal(rc, 0); + rc = dfs_release(sym_obj); + assert_int_equal(rc, 0); + deleted = false; + rc = dfs_remove_internal(dfs_mt, dir, "sym", false, &oid, &deleted); + assert_int_equal(rc, 0); + assert_true(deleted); + + rc = dfs_release(dir); + assert_int_equal(rc, 0); + rc = dfs_remove(dfs_mt, NULL, "df_dir", true, NULL); + assert_int_equal(rc, 0); +} + +/* + * Test the "deleted" out-parameter of dfs_move_internal(), which reports whether the clobbered + * rename destination object was actually destroyed. It stays false when the rename clobbers + * nothing, is false when the clobbered destination still has surviving hardlinks, and is true when + * a regular destination or the last surviving hardlink is clobbered. + */ +static void +dfs_test_move_deleted_flag(void **state) +{ + test_arg_t *arg = *state; + dfs_obj_t *dir; + dfs_obj_t *obj; + dfs_obj_t *hl_obj; + struct stat stbuf; + daos_obj_id_t oid; + bool deleted; + int rc; + + if (arg->myrank != 0) + return; + + rc = dfs_open(dfs_mt, NULL, "mv_dir", S_IFDIR | S_IWUSR | S_IRUSR, + O_RDWR | O_CREAT | O_EXCL, 0, 0, NULL, &dir); + assert_int_equal(rc, 0); + + /* Rename with no existing destination: nothing is clobbered, deleted stays false. */ + print_message("Step 1: rename with no clobber -> deleted == false\n"); + rc = dfs_open(dfs_mt, dir, "src", S_IFREG | S_IWUSR | S_IRUSR, O_RDWR | O_CREAT | O_EXCL, 0, + 0, NULL, &obj); + assert_int_equal(rc, 0); + rc = dfs_release(obj); + assert_int_equal(rc, 0); + deleted = true; + oid.lo = oid.hi = 0; + rc = dfs_move_internal(dfs_mt, 0, dir, "src", dir, "dst", NULL, &oid, &deleted); + assert_int_equal(rc, 0); + assert_false(deleted); + assert_true(oid.lo == 0 && oid.hi == 0); + + /* Rename clobbering a regular (single-link) file: the destination object is destroyed. */ + print_message("Step 2: rename clobbering a regular file -> deleted == true\n"); + rc = dfs_open(dfs_mt, dir, "src2", S_IFREG | S_IWUSR | S_IRUSR, O_RDWR | O_CREAT | O_EXCL, + 0, 0, NULL, &obj); + assert_int_equal(rc, 0); + rc = dfs_release(obj); + assert_int_equal(rc, 0); + rc = dfs_open(dfs_mt, dir, "victim", S_IFREG | S_IWUSR | S_IRUSR, O_RDWR | O_CREAT | O_EXCL, + 0, 0, NULL, &obj); + assert_int_equal(rc, 0); + rc = dfs_release(obj); + assert_int_equal(rc, 0); + deleted = false; + rc = dfs_move_internal(dfs_mt, 0, dir, "src2", dir, "victim", NULL, &oid, &deleted); + assert_int_equal(rc, 0); + assert_true(deleted); + + /* Build a hardlinked destination (hl1, hl2) with link_cnt == 2. */ + print_message("Step 3: rename clobbering a surviving hardlink -> deleted == false\n"); + rc = dfs_open(dfs_mt, dir, "hl1", S_IFREG | S_IWUSR | S_IRUSR, O_RDWR | O_CREAT | O_EXCL, 0, + 0, NULL, &hl_obj); + assert_int_equal(rc, 0); + rc = dfs_link(dfs_mt, hl_obj, dir, "hl2", NULL, &stbuf); + assert_int_equal(rc, 0); + assert_int_equal((int)stbuf.st_nlink, 2); + rc = dfs_release(hl_obj); + assert_int_equal(rc, 0); + + /* Clobber hl1 by renaming another file onto it: hl2 still refers to the object. */ + rc = dfs_open(dfs_mt, dir, "mover", S_IFREG | S_IWUSR | S_IRUSR, O_RDWR | O_CREAT | O_EXCL, + 0, 0, NULL, &obj); + assert_int_equal(rc, 0); + rc = dfs_release(obj); + assert_int_equal(rc, 0); + deleted = true; + rc = dfs_move_internal(dfs_mt, 0, dir, "mover", dir, "hl1", NULL, &oid, &deleted); + assert_int_equal(rc, 0); + assert_false(deleted); + rc = dfs_stat(dfs_mt, dir, "hl2", &stbuf); + assert_int_equal(rc, 0); + assert_int_equal((int)stbuf.st_nlink, 1); + + /* Clobber the last surviving link (hl2): the object is destroyed. */ + print_message("Step 4: rename clobbering the last hardlink -> deleted == true\n"); + rc = dfs_open(dfs_mt, dir, "mover2", S_IFREG | S_IWUSR | S_IRUSR, O_RDWR | O_CREAT | O_EXCL, + 0, 0, NULL, &obj); + assert_int_equal(rc, 0); + rc = dfs_release(obj); + assert_int_equal(rc, 0); + deleted = false; + rc = dfs_move_internal(dfs_mt, 0, dir, "mover2", dir, "hl2", NULL, &oid, &deleted); + assert_int_equal(rc, 0); + assert_true(deleted); + + rc = dfs_release(dir); + assert_int_equal(rc, 0); + rc = dfs_remove(dfs_mt, NULL, "mv_dir", true, NULL); + assert_int_equal(rc, 0); +} + static const struct CMUnitTest dfs_unit_tests[] = { {"DFS_UNIT_TEST1: DFS mount / umount", dfs_test_mount, async_disable, test_case_teardown}, {"DFS_UNIT_TEST2: DFS container modes", dfs_test_modes, async_disable, test_case_teardown}, @@ -6010,6 +6223,10 @@ static const struct CMUnitTest dfs_unit_tests[] = { test_case_teardown}, {"DFS_UNIT_TEST35: dfs hardlink orphaned name", dfs_test_orphan_name_hardlink, async_disable, test_case_teardown}, + {"DFS_UNIT_TEST36: dfs remove deleted flag", dfs_test_remove_deleted_flag, async_disable, + test_case_teardown}, + {"DFS_UNIT_TEST37: dfs move deleted flag", dfs_test_move_deleted_flag, async_disable, + test_case_teardown}, }; static int diff --git a/utils/cq/words.dict b/utils/cq/words.dict index 643f134100e..bc4d876aafd 100644 --- a/utils/cq/words.dict +++ b/utils/cq/words.dict @@ -154,6 +154,7 @@ debuginfo deduplicated defusedxml del +dentries dentry deps dereference @@ -224,6 +225,9 @@ groupdel groupname grp hackery +hardlink +hardlinked +hardlinks hexdump hfi hostfile @@ -395,6 +399,7 @@ rc rdb rdbt readdir +readdirplus readlink realloc rebase @@ -424,10 +429,12 @@ sbatch sbin scalable scancel +scandir scm scons scontrol seekable +setattr sharedctypes shlex simul