#define __Z_EROFS_BVSET(name, total) \ struct name { \ /* point to the next page which contains the following bvecs */ \ struct page *nextpage; \ struct z_erofs_bvec bvec[total]; \
}
__Z_EROFS_BVSET(z_erofs_bvset,);
__Z_EROFS_BVSET(z_erofs_bvset_inline, Z_EROFS_INLINE_BVECS);
staticstruct page *z_erofs_bvset_flip(struct z_erofs_bvec_iter *iter)
{ unsignedlong base = (unsignedlong)((struct z_erofs_bvset *)0)->bvec; /* have to access nextpage in advance, otherwise it will be unmapped */ struct page *nextpage = iter->bvset->nextpage; struct page *oldpage;
staticvoid z_erofs_destroy_pcluster_pool(void)
{ int i;
for (i = 0; i < ARRAY_SIZE(pcluster_pool); ++i) { if (!pcluster_pool[i].slab) continue;
kmem_cache_destroy(pcluster_pool[i].slab);
pcluster_pool[i].slab = NULL;
}
}
worker = erofs_init_percpu_worker(cpu); if (IS_ERR(worker)) return PTR_ERR(worker);
spin_lock(&z_erofs_pcpu_worker_lock);
old = rcu_dereference_protected(z_erofs_pcpu_workers[cpu],
lockdep_is_held(&z_erofs_pcpu_worker_lock)); if (!old)
rcu_assign_pointer(z_erofs_pcpu_workers[cpu], worker);
spin_unlock(&z_erofs_pcpu_worker_lock); if (old)
kthread_destroy_worker(worker); return0;
}
if (i_blocksize(fe->inode) != PAGE_SIZE ||
fe->mode < Z_EROFS_PCLUSTER_FOLLOWED) return;
for (i = 0; i < pclusterpages; ++i) { /* Inaccurate check w/o locking to avoid unneeded lookups */ if (READ_ONCE(pcl->compressed_bvecs[i].page)) continue;
folio = filemap_get_folio(mc, poff + i); if (IS_ERR(folio)) {
may_bypass = false; if (!shouldalloc) continue;
/* *Don'tperformin-placeI/Oifallcompressedpagesareavailablein *themanagedcache,asthepclustercanbemovedtothebypassqueue.
*/ if (may_bypass)
fe->mode = Z_EROFS_PCLUSTER_FOLLOWED_NOINPLACE;
}
/* (erofs_shrinker) disconnect cached encoded data with pclusters */ staticint erofs_try_to_free_all_cached_folios(struct erofs_sb_info *sbi, struct z_erofs_pcluster *pcl)
{ unsignedint pclusterpages = z_erofs_pclusterpages(pcl); struct folio *folio; int i;
DBG_BUGON(pcl->from_meta); /* Each cached folio contains one page unless bs > ps is supported */ for (i = 0; i < pclusterpages; ++i) { if (pcl->compressed_bvecs[i].page) {
folio = page_folio(pcl->compressed_bvecs[i].page); /* Avoid reclaiming or migrating this folio */ if (!folio_trylock(folio)) return -EBUSY;
/* callers must be with pcluster lock held */ staticint z_erofs_attach_page(struct z_erofs_frontend *fe, struct z_erofs_bvec *bvec, bool exclusive)
{ struct z_erofs_pcluster *pcl = fe->pcl; int ret;
if (exclusive) { /* Inplace I/O is limited to one page for uncompressed data */ if (pcl->algorithmformat < Z_EROFS_COMPRESSION_MAX ||
fe->icur <= 1) { /* Try to prioritize inplace I/O here */
spin_lock(&pcl->lockref.lock); while (fe->icur > 0) { if (pcl->compressed_bvecs[--fe->icur].page) continue;
pcl->compressed_bvecs[fe->icur] = *bvec;
spin_unlock(&pcl->lockref.lock); return0;
}
spin_unlock(&pcl->lockref.lock);
}
/* otherwise, check if it can be used as a bvpage */ if (fe->mode >= Z_EROFS_PCLUSTER_FOLLOWED &&
!fe->candidate_bvpage)
fe->candidate_bvpage = bvec->page;
}
ret = z_erofs_bvec_enqueue(&fe->biter, bvec, &fe->candidate_bvpage,
&fe->pagepool);
fe->pcl->vcnt += (ret >= 0); return ret;
}
staticbool z_erofs_get_pcluster(struct z_erofs_pcluster *pcl)
{ if (lockref_get_not_zero(&pcl->lockref)) returntrue;
spin_lock(&pcl->lockref.lock); if (__lockref_is_dead(&pcl->lockref)) {
spin_unlock(&pcl->lockref.lock); returnfalse;
}
if (!pcl->lockref.count++)
atomic_long_dec(&erofs_global_shrink_cnt);
spin_unlock(&pcl->lockref.lock); returntrue;
}
if (pcl) {
fe->pcl = pcl;
ret = -EEXIST;
} else {
ret = z_erofs_register_pcluster(fe);
}
if (ret == -EEXIST) {
mutex_lock(&fe->pcl->lock); /* check if this pcluster hasn't been linked into any chain. */ if (!cmpxchg(&fe->pcl->next, NULL, fe->head)) { /* .. so it can be attached to our submission chain */
fe->head = fe->pcl;
fe->mode = Z_EROFS_PCLUSTER_FOLLOWED;
} else { /* otherwise, it belongs to an inflight chain */
fe->mode = Z_EROFS_PCLUSTER_INFLIGHT;
}
} elseif (ret) { return ret;
}
z_erofs_bvec_iter_begin(&fe->biter, &fe->pcl->bvset,
Z_EROFS_INLINE_BVECS, fe->pcl->vcnt); if (!fe->pcl->from_meta) { /* bind cache first when cached decompression is preferred */
z_erofs_bind_cache(fe);
} else {
ret = erofs_init_metabuf(&map->buf, sb,
erofs_inode_in_metabox(fe->inode)); if (ret) return ret;
ptr = erofs_bread(&map->buf, map->m_pa, false); if (IS_ERR(ptr)) {
ret = PTR_ERR(ptr);
erofs_err(sb, "failed to get inline folio %d", ret); return ret;
}
folio_get(page_folio(map->buf.page));
WRITE_ONCE(fe->pcl->compressed_bvecs[0].page, map->buf.page);
fe->pcl->pageofs_in = map->m_pa & ~PAGE_MASK;
fe->mode = Z_EROFS_PCLUSTER_FOLLOWED_NOINPLACE;
} /* file-backed inplace I/O pages are traversed in reverse order */
fe->icur = z_erofs_pclusterpages(fe->pcl); return0;
}
if (fe->candidate_bvpage)
fe->candidate_bvpage = NULL;
/* Drop refcount if it doesn't belong to our processing chain */ if (fe->mode < Z_EROFS_PCLUSTER_FOLLOWED_NOINPLACE)
z_erofs_put_pcluster(EROFS_I_SB(fe->inode), pcl, false);
fe->pcl = NULL;
}
tight = (bs == PAGE_SIZE);
erofs_onlinefolio_init(folio); do { if (offset + end - 1 < map->m_la ||
offset + end - 1 >= map->m_la + map->m_llen) {
z_erofs_pcluster_end(f);
map->m_la = offset + end - 1;
map->m_llen = 0;
err = z_erofs_map_blocks_iter(inode, map, 0); if (err) break;
}
cur = offset > map->m_la ? 0 : map->m_la - offset;
pgs = round_down(cur, PAGE_SIZE); /* bump split parts first to avoid several separate cases */
++split;
struct z_erofs_backend { struct page *onstack_pages[Z_EROFS_ONSTACK_PAGES]; struct super_block *sb; struct z_erofs_pcluster *pcl; /* pages with the longest decompressed length for deduplication */ struct page **decompressed_pages; /* pages to keep the compressed data */ struct page **compressed_pages;
struct list_head decompressed_secondary_bvecs; struct page **pagepool; unsignedint onstack_used, nr_pages; /* indicate if temporary copies should be preserved for later use */ bool keepxcpy;
};
/* must handle all compressed pages before actual file pages */ if (pcl->from_meta) {
folio_put(page_folio(pcl->compressed_bvecs[0].page));
WRITE_ONCE(pcl->compressed_bvecs[0].page, NULL);
} else { /* managed folios are still left in compressed_bvecs[] */ for (i = 0; i < pclusterpages; ++i) {
page = be->compressed_pages[i]; if (!page) continue; if (erofs_folio_is_managed(sbi, page_folio(page))) {
try_free = false; continue;
}
(void)z_erofs_put_shortlivedpage(be->pagepool, page);
WRITE_ONCE(pcl->compressed_bvecs[i].page, NULL);
}
} if (be->compressed_pages < be->onstack_pages ||
be->compressed_pages >= be->onstack_pages + Z_EROFS_ONSTACK_PAGES)
kvfree(be->compressed_pages);
jtop = 0;
z_erofs_fill_other_copies(be, err); for (i = 0; i < be->nr_pages; ++i) {
page = be->decompressed_pages[i]; if (!page) continue;
DBG_BUGON(z_erofs_page_is_invalidated(page)); if (!z_erofs_is_shortlived_page(page)) {
erofs_onlinefolio_end(page_folio(page), err, true); continue;
} if (pcl->algorithmformat != Z_EROFS_COMPRESSION_LZ4) {
erofs_pagepool_add(be->pagepool, page); continue;
} for (j = 0; j < jtop && be->decompressed_pages[j] != page; ++j)
; if (j >= jtop) /* this bounce page is newly detected */
be->decompressed_pages[jtop++] = page;
} while (jtop)
erofs_pagepool_add(be->pagepool,
be->decompressed_pages[--jtop]); if (be->decompressed_pages != be->onstack_pages)
kvfree(be->decompressed_pages);
/* Use (kthread_)work in atomic contexts to minimize scheduling overhead */ staticinlinebool z_erofs_in_atomic(void)
{ if (IS_ENABLED(CONFIG_PREEMPTION) && rcu_preempt_depth()) returntrue; if (!IS_ENABLED(CONFIG_PREEMPT_COUNT)) returntrue; return !preemptible();
}
/* wake up the caller thread for sync decompression */ if (io->sync) { if (!atomic_add_return(bios, &io->pending_bios))
complete(&io->u.done); return;
}
if (atomic_add_return(bios, &io->pending_bios)) return; if (z_erofs_in_atomic()) { #ifdef CONFIG_EROFS_FS_PCPU_KTHREAD struct kthread_worker *worker;
/* Except for inplace folios, the entire folio can be used for I/Os */
bvec->bv_offset = 0;
bvec->bv_len = PAGE_SIZE;
repeat:
spin_lock(&pcl->lockref.lock);
zbv = pcl->compressed_bvecs[nr];
spin_unlock(&pcl->lockref.lock); if (!zbv.page) goto out_allocfolio;
DBG_BUGON(folio_test_uptodate(folio));
DBG_BUGON(z_erofs_page_is_invalidated(&folio->page)); if (!erofs_folio_is_managed(EROFS_SB(q->sb), folio)) continue;
if (!err)
folio_mark_uptodate(folio);
folio_unlock(folio);
} if (err)
q->eio = true;
z_erofs_decompress_kickoff(q, -1); if (bio->bi_bdev)
bio_put(bio);
}
staticvoid z_erofs_submit_queue(struct z_erofs_frontend *f, struct z_erofs_decompressqueue *fgq, bool *force_fg, bool readahead)
{ struct super_block *sb = f->inode->i_sb; struct address_space *mc = MNGD_MAPPING(EROFS_SB(sb)); struct z_erofs_pcluster **qtail[NR_JOBQUEUES]; struct z_erofs_decompressqueue *q[NR_JOBQUEUES]; struct z_erofs_pcluster *pcl, *next; /* bio is NULL initially, so no need to initialize last_{index,bdev} */
erofs_off_t last_pa; unsignedint nr_bios = 0; struct bio *bio = NULL; unsignedlong pflags; int memstall = 0;
/* No need to read from device for pclusters in the bypass queue. */
q[JQ_BYPASS] = jobqueue_init(sb, fgq + JQ_BYPASS, NULL);
q[JQ_SUBMIT] = jobqueue_init(sb, fgq + JQ_SUBMIT, force_fg);
/* by default, all need io submission */
q[JQ_SUBMIT]->head = next = f->head;
do { struct erofs_map_dev mdev;
erofs_off_t cur, end; struct bio_vec bvec; unsignedint i = 0; bool bypass = true;
pcl = next;
next = READ_ONCE(pcl->next); if (pcl->from_meta) {
z_erofs_move_to_bypass_queue(pcl, next, qtail); continue;
}
/* no device id here, thus it will always succeed */
mdev = (struct erofs_map_dev) {
.m_pa = round_down(pcl->pos, sb->s_blocksize),
};
(void)erofs_map_dev(sb, &mdev);
cur = mdev.m_pa;
end = round_up(cur + pcl->pageofs_in + pcl->pclustersize,
sb->s_blocksize); do {
bvec.bv_page = NULL; if (bio && (cur != last_pa ||
bio->bi_bdev != mdev.m_bdev)) {
drain_io: if (erofs_is_fileio_mode(EROFS_SB(sb)))
erofs_fileio_submit_bio(bio); elseif (erofs_is_fscache_mode(sb))
erofs_fscache_submit_bio(bio); else
submit_bio(bio);
if (memstall) {
psi_memstall_leave(&pflags);
memstall = 0;
}
bio = NULL;
}
if (!bvec.bv_page) {
z_erofs_fill_bio_vec(&bvec, f, pcl, i++, mc); if (!bvec.bv_page) continue; if (cur + bvec.bv_len > end)
bvec.bv_len = end - cur;
DBG_BUGON(bvec.bv_len < sb->s_blocksize);
}
if (unlikely(PageWorkingset(bvec.bv_page)) &&
!memstall) {
psi_memstall_enter(&pflags);
memstall = 1;
}
if (!bio) { if (erofs_is_fileio_mode(EROFS_SB(sb)))
bio = erofs_fileio_bio_alloc(&mdev); elseif (erofs_is_fscache_mode(sb))
bio = erofs_fscache_bio_alloc(&mdev); else
bio = bio_alloc(mdev.m_bdev, BIO_MAX_VECS,
REQ_OP_READ, GFP_NOIO);
bio->bi_end_io = z_erofs_endio;
bio->bi_iter.bi_sector =
(mdev.m_dif->fsoff + cur) >> 9;
bio->bi_private = q[JQ_SUBMIT]; if (readahead)
bio->bi_opf |= REQ_RAHEAD;
++nr_bios;
}
if (!bio_add_page(bio, bvec.bv_page, bvec.bv_len,
bvec.bv_offset)) goto drain_io;
last_pa = cur + bvec.bv_len;
bypass = false;
} while ((cur += bvec.bv_len) < end);
if (!bypass)
qtail[JQ_SUBMIT] = &pcl->next; else
z_erofs_move_to_bypass_queue(pcl, next, qtail);
} while (next != Z_EROFS_PCLUSTER_TAIL);
if (bio) { if (erofs_is_fileio_mode(EROFS_SB(sb)))
erofs_fileio_submit_bio(bio); elseif (erofs_is_fscache_mode(sb))
erofs_fscache_submit_bio(bio); else
submit_bio(bio);
} if (memstall)
psi_memstall_leave(&pflags);
if (backmost) { if (rac)
end = headoffset + readahead_length(rac) - 1; else
end = headoffset + PAGE_SIZE - 1;
map->m_la = end;
err = z_erofs_map_blocks_iter(inode, map,
EROFS_GET_BLOCKS_READMORE); if (err || !(map->m_flags & EROFS_MAP_ENCODED)) return;
/* expand ra for the trailing edge if readahead */ if (rac) {
cur = round_up(map->m_la + map->m_llen, PAGE_SIZE);
readahead_expand(rac, headoffset, cur - headoffset); return;
}
end = round_up(end, PAGE_SIZE);
} else {
end = round_up(map->m_la, PAGE_SIZE); if (!(map->m_flags & EROFS_MAP_ENCODED) || !map->m_llen) return;
}
cur = map->m_la + map->m_llen - 1; while ((cur >= end) && (cur < i_size_read(inode))) {
pgoff_t index = cur >> PAGE_SHIFT; struct folio *folio;
folio = erofs_grab_folio_nowait(inode->i_mapping, index); if (!IS_ERR_OR_NULL(folio)) { if (folio_test_uptodate(folio))
folio_unlock(folio); else
z_erofs_scan_folio(f, folio, !!rac);
folio_put(folio);
}
if (cur < PAGE_SIZE) break;
cur = (index << PAGE_SHIFT) - 1;
}
}
/* if some pclusters are ready, need submit them anyway */
err = z_erofs_runqueue(&f, 0) ?: err; if (err && err != -EINTR)
erofs_err(inode->i_sb, "read error %d @ %lu of nid %llu",
err, folio->index, EROFS_I(inode)->nid);
Die Informationen auf dieser Webseite wurden
nach bestem Wissen sorgfältig zusammengestellt. Es wird jedoch weder Vollständigkeit, noch Richtigkeit,
noch Qualität der bereit gestellten Informationen zugesichert.
Bemerkung:
Die farbliche Syntaxdarstellung und die Messung sind noch experimentell.