diff --git a/Documentation/admin-guide/cgroup-v2.rst b/Documentation/admin-guide/cgroup-v2.rst index 86a2a0099178..f0b49da2a64f 100644 --- a/Documentation/admin-guide/cgroup-v2.rst +++ b/Documentation/admin-guide/cgroup-v2.rst @@ -2127,6 +2127,7 @@ IO Interface Files [r|w]bps The maximum sequential IO throughput [r|w]seqiops The maximum 4k sequential IOs per second [r|w]randiops The maximum 4k random IOs per second + flushiops The maximum flushes per second ============= ======================================== From the above, the builtin linear model determines the base @@ -2134,6 +2135,12 @@ IO Interface Files for the IO size. While simple, this model can cover most common device classes acceptably. + "flushiops" determines the cost of a cache flush: a write with + a preceding cache flush is charged one flush on top of its data + cost, and a FUA write one more flush on devices without native + FUA support. It is zero in the builtin profiles, so flushes + stay free until it is configured. + The IO cost model isn't expected to be accurate in absolute sense and is scaled to the device behavior dynamically. diff --git a/block/blk-iocost.c b/block/blk-iocost.c index 2745bffcd5ee..a81488c755ca 100644 --- a/block/blk-iocost.c +++ b/block/blk-iocost.c @@ -353,6 +353,7 @@ enum { I_LCOEF_WBPS, I_LCOEF_WSEQIOPS, I_LCOEF_WRANDIOPS, + I_LCOEF_FLUSHIOPS, NR_I_LCOEFS, }; @@ -363,6 +364,7 @@ enum { LCOEF_WPAGE, LCOEF_WSEQIO, LCOEF_WRANDIO, + LCOEF_FLUSH, NR_LCOEFS, }; @@ -883,6 +885,9 @@ static void ioc_refresh_lcoefs(struct ioc *ioc) &c[LCOEF_RPAGE], &c[LCOEF_RSEQIO], &c[LCOEF_RRANDIO]); calc_lcoefs(u[I_LCOEF_WBPS], u[I_LCOEF_WSEQIOPS], u[I_LCOEF_WRANDIOPS], &c[LCOEF_WPAGE], &c[LCOEF_WSEQIO], &c[LCOEF_WRANDIO]); + + c[LCOEF_FLUSH] = u[I_LCOEF_FLUSHIOPS] ? + DIV64_U64_ROUND_UP(VTIME_PER_SEC, u[I_LCOEF_FLUSHIOPS]) : 0; } /* @@ -2533,7 +2538,16 @@ static void calc_vtime_cost_builtin(struct bio *bio, struct ioc_gq *iocg, u64 seek_pages = 0; u64 cost = 0; - /* Can't calculate cost for empty bio */ + /* + * FUA on a device without native support becomes a post-flush; + * charge it like PREFLUSH from the flush coefficient. + */ + if (bio->bi_opf & REQ_PREFLUSH) + cost += ioc->params.lcoefs[LCOEF_FLUSH]; + if ((bio->bi_opf & REQ_FUA) && !bdev_fua(bio->bi_bdev)) + cost += ioc->params.lcoefs[LCOEF_FLUSH]; + + /* Can't calculate data cost for empty bio */ if (!bio->bi_iter.bi_size) goto out; @@ -2543,6 +2557,7 @@ static void calc_vtime_cost_builtin(struct bio *bio, struct ioc_gq *iocg, coef_randio = ioc->params.lcoefs[LCOEF_RRANDIO]; coef_page = ioc->params.lcoefs[LCOEF_RPAGE]; break; + case REQ_OP_ZONE_APPEND: case REQ_OP_WRITE: coef_seqio = ioc->params.lcoefs[LCOEF_WSEQIO]; coef_randio = ioc->params.lcoefs[LCOEF_WRANDIO]; @@ -2586,6 +2601,7 @@ static void calc_size_vtime_cost_builtin(struct request *rq, struct ioc *ioc, case REQ_OP_READ: *costp = pages * ioc->params.lcoefs[LCOEF_RPAGE]; break; + case REQ_OP_ZONE_APPEND: case REQ_OP_WRITE: *costp = pages * ioc->params.lcoefs[LCOEF_WPAGE]; break; @@ -2708,7 +2724,9 @@ static void ioc_rqos_throttle(struct rq_qos *rqos, struct bio *bio) if (!iocg_activate(iocg, &now)) return; - iocg->cursor = bio_end_sector(bio); + /* dataless bios have no meaningful position for seq/rand detection */ + if (bio->bi_iter.bi_size) + iocg->cursor = bio_end_sector(bio); vtime = atomic64_read(&iocg->vtime); cost = adjust_inuse_and_calc_cost(iocg, vtime, abs_cost, &now); @@ -2725,10 +2743,11 @@ static void ioc_rqos_throttle(struct rq_qos *rqos, struct bio *bio) /* * We're over budget. This can be handled in two ways. IOs which may - * cause priority inversions are punted to @ioc->aux_iocg and charged as - * debt. Otherwise, the issuer is blocked on @iocg->waitq. Debt handling - * requires @ioc->lock, waitq handling @iocg->waitq.lock. Determine - * whether debt handling is needed and acquire locks accordingly. + * cause priority inversions are issued regardless and charged against + * @iocg->abs_vdebt as debt. Otherwise, the issuer is blocked on + * @iocg->waitq. Debt handling requires @ioc->lock, waitq handling + * @iocg->waitq.lock. Determine whether debt handling is needed and + * acquire locks accordingly. */ use_debt = bio_issue_as_root_blkg(bio) || fatal_signal_pending(current); ioc_locked = use_debt || READ_ONCE(iocg->abs_vdebt); @@ -2854,6 +2873,7 @@ static void ioc_rqos_done(struct rq_qos *rqos, struct request *rq) pidx = QOS_RLAT; rw = READ; break; + case REQ_OP_ZONE_APPEND: case REQ_OP_WRITE: pidx = QOS_WLAT; rw = WRITE; @@ -3440,10 +3460,12 @@ static u64 ioc_cost_model_prfill(struct seq_file *sf, spin_lock_irq(&ioc->lock); seq_printf(sf, "%s ctrl=%s model=linear " "rbps=%llu rseqiops=%llu rrandiops=%llu " - "wbps=%llu wseqiops=%llu wrandiops=%llu\n", + "wbps=%llu wseqiops=%llu wrandiops=%llu " + "flushiops=%llu\n", dname, ioc->user_cost_model ? "user" : "auto", u[I_LCOEF_RBPS], u[I_LCOEF_RSEQIOPS], u[I_LCOEF_RRANDIOPS], - u[I_LCOEF_WBPS], u[I_LCOEF_WSEQIOPS], u[I_LCOEF_WRANDIOPS]); + u[I_LCOEF_WBPS], u[I_LCOEF_WSEQIOPS], u[I_LCOEF_WRANDIOPS], + u[I_LCOEF_FLUSHIOPS]); spin_unlock_irq(&ioc->lock); return 0; } @@ -3470,6 +3492,7 @@ static const match_table_t i_lcoef_tokens = { { I_LCOEF_WBPS, "wbps=%u" }, { I_LCOEF_WSEQIOPS, "wseqiops=%u" }, { I_LCOEF_WRANDIOPS, "wrandiops=%u" }, + { I_LCOEF_FLUSHIOPS, "flushiops=%u" }, { NR_I_LCOEFS, NULL }, };