aboutsummaryrefslogtreecommitdiff
path: root/sys/ufs
diff options
context:
space:
mode:
authorKirk McKusick <mckusick@FreeBSD.org>2023-03-30 04:09:39 +0000
committerKirk McKusick <mckusick@FreeBSD.org>2023-03-30 04:13:27 +0000
commitfe5e6e2cc5d6f2e4121eccdb3a8ceba646aef2c9 (patch)
treef74a7144169b6ec5ee7758817e409d55296b031a /sys/ufs
parent1fb7d2cf999e52e3682174d0c2f20cb3baf414f3 (diff)
Diffstat (limited to 'sys/ufs')
-rw-r--r--sys/ufs/ffs/ffs_alloc.c71
-rw-r--r--sys/ufs/ffs/ffs_softdep.c26
-rw-r--r--sys/ufs/ufs/dinode.h10
-rw-r--r--sys/ufs/ufs/ufs_vnops.c5
4 files changed, 73 insertions, 39 deletions
diff --git a/sys/ufs/ffs/ffs_alloc.c b/sys/ufs/ffs/ffs_alloc.c
index 4b0c7b108cb6..6d37afcfadf6 100644
--- a/sys/ufs/ffs/ffs_alloc.c
+++ b/sys/ufs/ffs/ffs_alloc.c
@@ -1179,6 +1179,8 @@ retry:
}
ip->i_flags = 0;
DIP_SET(ip, i_flags, 0);
+ if ((mode & IFMT) == IFDIR)
+ DIP_SET(ip, i_dirdepth, DIP(pip, i_dirdepth) + 1);
/*
* Set up a new generation number for this inode.
*/
@@ -1238,10 +1240,10 @@ static ino_t
ffs_dirpref(struct inode *pip)
{
struct fs *fs;
- int cg, prefcg, dirsize, cgsize;
+ int cg, prefcg, curcg, dirsize, cgsize;
+ int depth, range, start, end, numdirs, power, numerator, denominator;
u_int avgifree, avgbfree, avgndir, curdirsize;
u_int minifree, minbfree, maxndir;
- u_int mincg, minndir;
u_int maxcontigdirs;
mtx_assert(UFS_MTX(ITOUMP(pip)), MA_OWNED);
@@ -1252,35 +1254,53 @@ ffs_dirpref(struct inode *pip)
avgndir = fs->fs_cstotal.cs_ndir / fs->fs_ncg;
/*
- * Force allocation in another cg if creating a first level dir.
+ * Select a preferred cylinder group to place a new directory.
+ * If we are near the root of the filesystem we aim to spread
+ * them out as much as possible. As we descend deeper from the
+ * root we cluster them closer together around their parent as
+ * we expect them to be more closely interactive. Higher-level
+ * directories like usr/src/sys and usr/src/bin should be
+ * separated while the directories in these areas are more
+ * likely to be accessed together so should be closer.
+ *
+ * We pick a range of cylinder groups around the cylinder group
+ * of the directory in which we are being created. The size of
+ * the range for our search is based on our depth from the root
+ * of our filesystem. We then probe that range based on how many
+ * directories are already present. The first new directory is at
+ * 1/2 (middle) of the range; the second is in the first 1/4 of the
+ * range, then at 3/4, 1/8, 3/8, 5/8, 7/8, 1/16, 3/16, 5/16, etc.
*/
- ASSERT_VOP_LOCKED(ITOV(pip), "ffs_dirpref");
- if (ITOV(pip)->v_vflag & VV_ROOT) {
- prefcg = arc4random() % fs->fs_ncg;
- mincg = prefcg;
- minndir = fs->fs_ipg;
- for (cg = prefcg; cg < fs->fs_ncg; cg++)
- if (fs->fs_cs(fs, cg).cs_ndir < minndir &&
- fs->fs_cs(fs, cg).cs_nifree >= avgifree &&
- fs->fs_cs(fs, cg).cs_nbfree >= avgbfree) {
- mincg = cg;
- minndir = fs->fs_cs(fs, cg).cs_ndir;
- }
- for (cg = 0; cg < prefcg; cg++)
- if (fs->fs_cs(fs, cg).cs_ndir < minndir &&
- fs->fs_cs(fs, cg).cs_nifree >= avgifree &&
- fs->fs_cs(fs, cg).cs_nbfree >= avgbfree) {
- mincg = cg;
- minndir = fs->fs_cs(fs, cg).cs_ndir;
- }
- return ((ino_t)(fs->fs_ipg * mincg));
- }
+ depth = DIP(pip, i_dirdepth);
+ range = fs->fs_ncg / (1 << depth);
+ curcg = ino_to_cg(fs, pip->i_number);
+ start = curcg - (range / 2);
+ if (start < 0)
+ start += fs->fs_ncg;
+ end = curcg + (range / 2);
+ if (end >= fs->fs_ncg)
+ end -= fs->fs_ncg;
+ numdirs = pip->i_effnlink - 1;
+ power = fls(numdirs);
+ numerator = (numdirs & ~(1 << (power - 1))) * 2 + 1;
+ denominator = 1 << power;
+ prefcg = (curcg - (range / 2) + (range * numerator / denominator));
+ if (prefcg < 0)
+ prefcg += fs->fs_ncg;
+ if (prefcg >= fs->fs_ncg)
+ prefcg -= fs->fs_ncg;
+ /*
+ * If this filesystem is not tracking directory depths,
+ * revert to the old algorithm.
+ */
+ if (depth == 0 && pip->i_number != UFS_ROOTINO)
+ prefcg = curcg;
/*
* Count various limits which used for
* optimal allocation of a directory inode.
*/
- maxndir = min(avgndir + fs->fs_ipg / 16, fs->fs_ipg);
+ maxndir = min(avgndir + (1 << depth), fs->fs_ipg);
minifree = avgifree - avgifree / 4;
if (minifree < 1)
minifree = 1;
@@ -1324,7 +1344,6 @@ ffs_dirpref(struct inode *pip)
* in new cylinder groups so finds every possible block after
* one pass over the filesystem.
*/
- prefcg = ino_to_cg(fs, pip->i_number);
for (cg = prefcg; cg < fs->fs_ncg; cg++)
if (fs->fs_cs(fs, cg).cs_ndir < maxndir &&
fs->fs_cs(fs, cg).cs_nifree >= minifree &&
diff --git a/sys/ufs/ffs/ffs_softdep.c b/sys/ufs/ffs/ffs_softdep.c
index ef1753862595..1e245d82f8b8 100644
--- a/sys/ufs/ffs/ffs_softdep.c
+++ b/sys/ufs/ffs/ffs_softdep.c
@@ -12486,17 +12486,6 @@ softdep_update_inodeblock(
("softdep_update_inodeblock called on non-softdep filesystem"));
fs = ump->um_fs;
/*
- * Preserve the freelink that is on disk. clear_unlinked_inodedep()
- * does not have access to the in-core ip so must write directly into
- * the inode block buffer when setting freelink.
- */
- if (fs->fs_magic == FS_UFS1_MAGIC)
- DIP_SET(ip, i_freelink, ((struct ufs1_dinode *)bp->b_data +
- ino_to_fsbo(fs, ip->i_number))->di_freelink);
- else
- DIP_SET(ip, i_freelink, ((struct ufs2_dinode *)bp->b_data +
- ino_to_fsbo(fs, ip->i_number))->di_freelink);
- /*
* If the effective link count is not equal to the actual link
* count, then we must track the difference in an inodedep while
* the inode is (potentially) tossed out of the cache. Otherwise,
@@ -12511,6 +12500,21 @@ again:
panic("softdep_update_inodeblock: bad link count");
return;
}
+ /*
+ * Preserve the freelink that is on disk. clear_unlinked_inodedep()
+ * does not have access to the in-core ip so must write directly into
+ * the inode block buffer when setting freelink.
+ */
+ if ((inodedep->id_state & UNLINKED) != 0) {
+ if (fs->fs_magic == FS_UFS1_MAGIC)
+ DIP_SET(ip, i_freelink,
+ ((struct ufs1_dinode *)bp->b_data +
+ ino_to_fsbo(fs, ip->i_number))->di_freelink);
+ else
+ DIP_SET(ip, i_freelink,
+ ((struct ufs2_dinode *)bp->b_data +
+ ino_to_fsbo(fs, ip->i_number))->di_freelink);
+ }
KASSERT(ip->i_nlink >= inodedep->id_nlinkdelta,
("softdep_update_inodeblock inconsistent ip %p i_nlink %d "
"inodedep %p id_nlinkdelta %jd",
diff --git a/sys/ufs/ufs/dinode.h b/sys/ufs/ufs/dinode.h
index 840a4cc7d40f..e4a424abe2e6 100644
--- a/sys/ufs/ufs/dinode.h
+++ b/sys/ufs/ufs/dinode.h
@@ -156,7 +156,10 @@ struct ufs2_dinode {
[(UFS_NDADDR + UFS_NIADDR) * sizeof(ufs2_daddr_t)];
};
u_int64_t di_modrev; /* 232: i_modrev for NFSv4 */
- uint32_t di_freelink; /* 240: SUJ: Next unlinked inode. */
+ union {
+ uint32_t di_freelink; /* 240: SUJ: Next unlinked inode. */
+ uint32_t di_dirdepth; /* 240: IFDIR: depth from root dir */
+ };
uint32_t di_ckhash; /* 244: if CK_INODE, its check-hash */
uint32_t di_spare[2]; /* 248: Reserved; currently unused */
};
@@ -179,7 +182,10 @@ struct ufs2_dinode {
struct ufs1_dinode {
u_int16_t di_mode; /* 0: IFMT, permissions; see below. */
int16_t di_nlink; /* 2: File link count. */
- uint32_t di_freelink; /* 4: SUJ: Next unlinked inode. */
+ union {
+ uint32_t di_freelink; /* 4: SUJ: Next unlinked inode. */
+ uint32_t di_dirdepth; /* 4: IFDIR: depth from root dir */
+ };
u_int64_t di_size; /* 8: File byte count. */
int32_t di_atime; /* 16: Last access time. */
int32_t di_atimensec; /* 20: Last access time. */
diff --git a/sys/ufs/ufs/ufs_vnops.c b/sys/ufs/ufs/ufs_vnops.c
index c13aec4e2175..7815293b92a7 100644
--- a/sys/ufs/ufs/ufs_vnops.c
+++ b/sys/ufs/ufs/ufs_vnops.c
@@ -1711,6 +1711,10 @@ relock:
*/
if (doingdirectory && newparent) {
/*
+ * Set the directory depth based on its new parent.
+ */
+ DIP_SET(fip, i_dirdepth, DIP(tdp, i_dirdepth) + 1);
+ /*
* If tip exists we simply use its link, otherwise we must
* add a new one.
*/
@@ -2121,6 +2125,7 @@ ufs_mkdir(
ip->i_effnlink = 2;
ip->i_nlink = 2;
DIP_SET(ip, i_nlink, 2);
+ DIP_SET(ip, i_dirdepth, DIP(dp,i_dirdepth) + 1);
if (cnp->cn_flags & ISWHITEOUT) {
ip->i_flags |= UF_OPAQUE;