mirror of
https://git.proxmox.com/git/mirror_zfs.git
synced 2025-01-09 01:30:33 +03:00
82a37189aa
The current ZFS implementation stores xattrs on disk using a hidden directory. In this directory a file name represents the xattr name and the file contexts are the xattr binary data. This approach is very flexible and allows for arbitrarily large xattrs. However, it also suffers from a significant performance penalty. Accessing a single xattr can requires up to three disk seeks. 1) Lookup the dnode object. 2) Lookup the dnodes's xattr directory object. 3) Lookup the xattr object in the directory. To avoid this performance penalty Linux filesystems such as ext3 and xfs try to store the xattr as part of the inode on disk. When the xattr is to large to store in the inode then a single external block is allocated for them. In practice most xattrs are small and this approach works well. The addition of System Attributes (SA) to zfs provides us a clean way to make this optimization. When the dataset property 'xattr=sa' is set then xattrs will be preferentially stored as System Attributes. This allows tiny xattrs (~100 bytes) to be stored with the dnode and up to 64k of xattrs to be stored in the spill block. If additional xattr space is required, which is unlikely under Linux, they will be stored using the traditional directory approach. This optimization results in roughly a 3x performance improvement when accessing xattrs which brings zfs roughly to parity with ext4 and xfs (see table below). When multiple xattrs are stored per-file the performance improvements are even greater because all of the xattrs stored in the spill block will be cached. However, by default SA based xattrs are disabled in the Linux port to maximize compatibility with other implementations. If you do enable SA based xattrs then they will not be visible on platforms which do not support this feature. ---------------------------------------------------------------------- Time in seconds to get/set one xattr of N bytes on 100,000 files ------+--------------------------------+------------------------------ | setxattr | getxattr bytes | ext4 xfs zfs-dir zfs-sa | ext4 xfs zfs-dir zfs-sa ------+--------------------------------+------------------------------ 1 | 2.33 31.88 21.50 4.57 | 2.35 2.64 6.29 2.43 32 | 2.79 30.68 21.98 4.60 | 2.44 2.59 6.78 2.48 256 | 3.25 31.99 21.36 5.92 | 2.32 2.71 6.22 3.14 1024 | 3.30 32.61 22.83 8.45 | 2.40 2.79 6.24 3.27 4096 | 3.57 317.46 22.52 10.73 | 2.78 28.62 6.90 3.94 16384 | n/a 2342.39 34.30 19.20 | n/a 45.44 145.90 7.55 65536 | n/a 2941.39 128.15 131.32* | n/a 141.92 256.85 262.12* Legend: * ext4 - Stock RHEL6.1 ext4 mounted with '-o user_xattr'. * xfs - Stock RHEL6.1 xfs mounted with default options. * zfs-dir - Directory based xattrs only. * zfs-sa - Prefer SAs but spill in to directories as needed, a trailing * indicates overflow in to directories occured. NOTE: Ext4 supports 4096 bytes of xattr name/value pairs per file. NOTE: XFS and ZFS have no limit on xattr name/value pairs per file. NOTE: Linux limits individual name/value pairs to 65536 bytes. NOTE: All setattr/getattr's were done after dropping the cache. NOTE: All tests were run against a single hard drive. Signed-off-by: Brian Behlendorf <behlendorf1@llnl.gov> Issue #443
206 lines
7.6 KiB
C
206 lines
7.6 KiB
C
/*
|
|
* CDDL HEADER START
|
|
*
|
|
* The contents of this file are subject to the terms of the
|
|
* Common Development and Distribution License (the "License").
|
|
* You may not use this file except in compliance with the License.
|
|
*
|
|
* You can obtain a copy of the license at usr/src/OPENSOLARIS.LICENSE
|
|
* or http://www.opensolaris.org/os/licensing.
|
|
* See the License for the specific language governing permissions
|
|
* and limitations under the License.
|
|
*
|
|
* When distributing Covered Code, include this CDDL HEADER in each
|
|
* file and include the License file at usr/src/OPENSOLARIS.LICENSE.
|
|
* If applicable, add the following below this CDDL HEADER, with the
|
|
* fields enclosed by brackets "[]" replaced with your own identifying
|
|
* information: Portions Copyright [yyyy] [name of copyright owner]
|
|
*
|
|
* CDDL HEADER END
|
|
*/
|
|
/*
|
|
* Copyright (c) 2005, 2010, Oracle and/or its affiliates. All rights reserved.
|
|
*/
|
|
|
|
#ifndef _SYS_FS_ZFS_VFSOPS_H
|
|
#define _SYS_FS_ZFS_VFSOPS_H
|
|
|
|
#include <sys/isa_defs.h>
|
|
#include <sys/types32.h>
|
|
#include <sys/list.h>
|
|
#include <sys/vfs.h>
|
|
#include <sys/zil.h>
|
|
#include <sys/sa.h>
|
|
#include <sys/rrwlock.h>
|
|
#include <sys/zfs_ioctl.h>
|
|
|
|
#ifdef __cplusplus
|
|
extern "C" {
|
|
#endif
|
|
|
|
struct zfs_sb;
|
|
struct znode;
|
|
|
|
typedef struct zfs_sb {
|
|
struct super_block *z_sb; /* generic super_block */
|
|
struct backing_dev_info z_bdi; /* generic backing dev info */
|
|
struct zfs_sb *z_parent; /* parent fs */
|
|
objset_t *z_os; /* objset reference */
|
|
uint64_t z_flags; /* super_block flags */
|
|
uint64_t z_root; /* id of root znode */
|
|
uint64_t z_unlinkedobj; /* id of unlinked zapobj */
|
|
uint64_t z_max_blksz; /* maximum block size for files */
|
|
uint64_t z_fuid_obj; /* fuid table object number */
|
|
uint64_t z_fuid_size; /* fuid table size */
|
|
avl_tree_t z_fuid_idx; /* fuid tree keyed by index */
|
|
avl_tree_t z_fuid_domain; /* fuid tree keyed by domain */
|
|
krwlock_t z_fuid_lock; /* fuid lock */
|
|
boolean_t z_fuid_loaded; /* fuid tables are loaded */
|
|
boolean_t z_fuid_dirty; /* need to sync fuid table ? */
|
|
struct zfs_fuid_info *z_fuid_replay; /* fuid info for replay */
|
|
zilog_t *z_log; /* intent log pointer */
|
|
uint_t z_acl_inherit; /* acl inheritance behavior */
|
|
zfs_case_t z_case; /* case-sense */
|
|
boolean_t z_utf8; /* utf8-only */
|
|
int z_norm; /* normalization flags */
|
|
boolean_t z_atime; /* enable atimes mount option */
|
|
boolean_t z_unmounted; /* unmounted */
|
|
rrwlock_t z_teardown_lock;
|
|
krwlock_t z_teardown_inactive_lock;
|
|
list_t z_all_znodes; /* all vnodes in the fs */
|
|
kmutex_t z_znodes_lock; /* lock for z_all_znodes */
|
|
struct inode *z_ctldir; /* .zfs directory inode */
|
|
boolean_t z_show_ctldir; /* expose .zfs in the root dir */
|
|
boolean_t z_issnap; /* true if this is a snapshot */
|
|
boolean_t z_vscan; /* virus scan on/off */
|
|
boolean_t z_use_fuids; /* version allows fuids */
|
|
boolean_t z_replay; /* set during ZIL replay */
|
|
boolean_t z_use_sa; /* version allow system attributes */
|
|
boolean_t z_xattr_sa; /* allow xattrs to be stores as SA */
|
|
uint64_t z_version; /* ZPL version */
|
|
uint64_t z_shares_dir; /* hidden shares dir */
|
|
kmutex_t z_lock;
|
|
uint64_t z_userquota_obj;
|
|
uint64_t z_groupquota_obj;
|
|
uint64_t z_replay_eof; /* New end of file - replay only */
|
|
sa_attr_type_t *z_attr_table; /* SA attr mapping->id */
|
|
#define ZFS_OBJ_MTX_SZ 64
|
|
kmutex_t z_hold_mtx[ZFS_OBJ_MTX_SZ]; /* znode hold locks */
|
|
} zfs_sb_t;
|
|
|
|
#define ZFS_SUPER_MAGIC 0x2fc12fc1
|
|
|
|
#define ZSB_XATTR 0x0001 /* Enable user xattrs */
|
|
|
|
|
|
/*
|
|
* Minimal snapshot helpers, the bulk of the Linux snapshot implementation
|
|
* lives in the zpl_snap.c file which is part of the zpl source.
|
|
*/
|
|
#define ZFS_CTLDIR_NAME ".zfs"
|
|
|
|
#define zfs_has_ctldir(zdp) \
|
|
((zdp)->z_id == ZTOZSB(zdp)->z_root && \
|
|
(ZTOZSB(zdp)->z_ctldir != NULL))
|
|
#define zfs_show_ctldir(zdp) \
|
|
(zfs_has_ctldir(zdp) && \
|
|
(ZTOZSB(zdp)->z_show_ctldir))
|
|
|
|
#define ZFSCTL_INO_ROOT 0x1
|
|
#define ZFSCTL_INO_SNAPDIR 0x2
|
|
#define ZFSCTL_INO_SHARES 0x3
|
|
|
|
/*
|
|
* Allow a maximum number of links. While ZFS does not internally limit
|
|
* this most Linux filesystems do. It's probably a good idea to limit
|
|
* this to a large value until it is validated that this is safe.
|
|
*/
|
|
#define ZFS_LINK_MAX 65536
|
|
|
|
/*
|
|
* Normal filesystems (those not under .zfs/snapshot) have a total
|
|
* file ID size limited to 12 bytes (including the length field) due to
|
|
* NFSv2 protocol's limitation of 32 bytes for a filehandle. For historical
|
|
* reasons, this same limit is being imposed by the Solaris NFSv3 implementation
|
|
* (although the NFSv3 protocol actually permits a maximum of 64 bytes). It
|
|
* is not possible to expand beyond 12 bytes without abandoning support
|
|
* of NFSv2.
|
|
*
|
|
* For normal filesystems, we partition up the available space as follows:
|
|
* 2 bytes fid length (required)
|
|
* 6 bytes object number (48 bits)
|
|
* 4 bytes generation number (32 bits)
|
|
*
|
|
* We reserve only 48 bits for the object number, as this is the limit
|
|
* currently defined and imposed by the DMU.
|
|
*/
|
|
typedef struct zfid_short {
|
|
uint16_t zf_len;
|
|
uint8_t zf_object[6]; /* obj[i] = obj >> (8 * i) */
|
|
uint8_t zf_gen[4]; /* gen[i] = gen >> (8 * i) */
|
|
} zfid_short_t;
|
|
|
|
/*
|
|
* Filesystems under .zfs/snapshot have a total file ID size of 22 bytes
|
|
* (including the length field). This makes files under .zfs/snapshot
|
|
* accessible by NFSv3 and NFSv4, but not NFSv2.
|
|
*
|
|
* For files under .zfs/snapshot, we partition up the available space
|
|
* as follows:
|
|
* 2 bytes fid length (required)
|
|
* 6 bytes object number (48 bits)
|
|
* 4 bytes generation number (32 bits)
|
|
* 6 bytes objset id (48 bits)
|
|
* 4 bytes currently just zero (32 bits)
|
|
*
|
|
* We reserve only 48 bits for the object number and objset id, as these are
|
|
* the limits currently defined and imposed by the DMU.
|
|
*/
|
|
typedef struct zfid_long {
|
|
zfid_short_t z_fid;
|
|
uint8_t zf_setid[6]; /* obj[i] = obj >> (8 * i) */
|
|
uint8_t zf_setgen[4]; /* gen[i] = gen >> (8 * i) */
|
|
} zfid_long_t;
|
|
|
|
#define SHORT_FID_LEN (sizeof (zfid_short_t) - sizeof (uint16_t))
|
|
#define LONG_FID_LEN (sizeof (zfid_long_t) - sizeof (uint16_t))
|
|
|
|
extern uint_t zfs_fsyncer_key;
|
|
|
|
extern int zfs_suspend_fs(zfs_sb_t *zsb);
|
|
extern int zfs_resume_fs(zfs_sb_t *zsb, const char *osname);
|
|
extern int zfs_userspace_one(zfs_sb_t *zsb, zfs_userquota_prop_t type,
|
|
const char *domain, uint64_t rid, uint64_t *valuep);
|
|
extern int zfs_userspace_many(zfs_sb_t *zsb, zfs_userquota_prop_t type,
|
|
uint64_t *cookiep, void *vbuf, uint64_t *bufsizep);
|
|
extern int zfs_set_userquota(zfs_sb_t *zsb, zfs_userquota_prop_t type,
|
|
const char *domain, uint64_t rid, uint64_t quota);
|
|
extern boolean_t zfs_owner_overquota(zfs_sb_t *zsb, struct znode *,
|
|
boolean_t isgroup);
|
|
extern boolean_t zfs_fuid_overquota(zfs_sb_t *zsb, boolean_t isgroup,
|
|
uint64_t fuid);
|
|
extern int zfs_set_version(zfs_sb_t *zsb, uint64_t newvers);
|
|
extern int zfs_get_zplprop(objset_t *os, zfs_prop_t prop,
|
|
uint64_t *value);
|
|
extern int zfs_sb_create(const char *name, zfs_sb_t **zsbp);
|
|
extern int zfs_sb_setup(zfs_sb_t *zsb, boolean_t mounting);
|
|
extern void zfs_sb_free(zfs_sb_t *zsb);
|
|
extern int zfs_sb_teardown(zfs_sb_t *zsb, boolean_t unmounting);
|
|
extern int zfs_check_global_label(const char *dsname, const char *hexsl);
|
|
extern boolean_t zfs_is_readonly(zfs_sb_t *zsb);
|
|
|
|
extern int zfs_register_callbacks(zfs_sb_t *zsb);
|
|
extern void zfs_unregister_callbacks(zfs_sb_t *zsb);
|
|
extern int zfs_domount(struct super_block *sb, void *data, int silent);
|
|
extern int zfs_umount(struct super_block *sb);
|
|
extern int zfs_remount(struct super_block *sb, int *flags, char *data);
|
|
extern int zfs_root(zfs_sb_t *zsb, struct inode **ipp);
|
|
extern int zfs_statvfs(struct dentry *dentry, struct kstatfs *statp);
|
|
extern int zfs_vget(struct super_block *sb, struct inode **ipp, fid_t *fidp);
|
|
|
|
#ifdef __cplusplus
|
|
}
|
|
#endif
|
|
|
|
#endif /* _SYS_FS_ZFS_VFSOPS_H */
|