Files
android_kernel_fxtec_sm6115/block/elevator.c
Thomas Turner b49f61e801 Merge tag 'v4.19.325-cip134' of https://git.kernel.org/pub/scm/linux/kernel/git/cip/linux-cip into android13-4.19-kona
version 4.19.325-cip134

* tag 'v4.19.325-cip134' of https://git.kernel.org/pub/scm/linux/kernel/git/cip/linux-cip:
  CIP: Bump version suffix to -cip134 after merge from cip/linux-4.19.y-st tree
  Update localversion-st, tree is up-to-date with 5.10.258.
  tipc: fix double-free in tipc_buf_append()
  slip: reject VJ receive packets on instances with no rstate array
  netfilter: nfnetlink_osf: fix potential NULL dereference in ttl check
  wifi: brcmfmac: Fix error pointer dereference
  bpf: fix end-of-list detection in cgroup_storage_get_next_key()
  crypto: ccp - copy IV using skcipher ivsize
  netfilter: ipset: stop hash:* range iteration at end
  net/sched: netem: fix queue limit check to include reordered packets
  cdrom, scsi: sr: propagate read-only status to block layer via set_disk_ro()
  btrfs: fix double-decrement of bytes_may_use in submit_one_async_extent()
  crypto: authencesn - reject short ahash digests during instance creation
  crypto: ccp: Don't attempt to copy PDH cert to userspace if PSP command failed
  crypto: ccp: Don't attempt to copy CSR to userspace if PSP command failed
  x86/uprobes: Fix XOL allocation failure for 32-bit tasks
  blk-mq: use quiesced elevator switch when reinitializing queues
  scsi: ufs: core: Improve SCSI abort handling
  bcache: fix cached_dev.sb_bio use-after-free and crash
  rxrpc: proc: size address buffers for %pISpc output
  can: raw: fix ro->uniq use-after-free in raw_rcv()
  can: af_can: export can_sock_destruct()
  batman-adv: hold claim backbone gateways by reference
  l2tp: Drop large packets with UDP encap
  xsk: tighten UMEM headroom validation to account for tailroom and min frame
  can: mcp251x: add error handling for power enable in open and resume
  net: skbuff: propagate shared-frag marker through frag-transfer helpers
  Revert "i2c: fsi: Fix a potential leak in fsi_i2c_probe()"
  net: usb: lan78xx: Fix double free issue with interrupt buffer allocation
  string: add mem_is_zero() helper to check if memory area is all zeros
  tracing: Avoid NULL return from hist_field_name() on truncation
  platform/x86: intel-hid: Check ACPI_HANDLE() against NULL
  HID: quirks: really enable the intended work around for appledisplay
  net: ethernet: cs89x0: remove stale CONFIG_MACH_MX31ADS reference
  net: ethernet: cortina: Carry over frag counter
  net: ethernet: cortina: Drop half-assembled SKB
  net: ethernet: cortina: Make RX SKB per-port
  irqchip/ath79-cpu: Remove unused function
  ARM: integrator: Fix early initialization
  batman-adv: tt: fix negative tt_buff_len
  batman-adv: tt: fix negative last_changeset_len
  batman-adv: tp_meter: avoid use of uninit sender vars
  batman-adv: bla: fix report_work leak on backbone_gw purge
  batman-adv: frag: disallow unicast fragment in fragment
  batman-adv: fix tp_meter counter underflow during shutdown
  batman-adv: fix fragment reassembly length accounting
  batman-adv: dat: handle forward allocation error
  batman-adv: clear current gateway during teardown
  batman-adv: mcast: fix use-after-free in orig_node RCU release
  drm/amd/display: Fix integer overflow in bios_get_image()
  spi: ti-qspi: fix use-after-free after DMA setup failure
  scsi: isci: Fix use-after-free in device removal path
  tracing: Do not call map->ops->elt_free() if elt_alloc() fails
  ixgbevf: fix use-after-free in VEPA multicast source pruning
  ipv4: raw: reject IP_HDRINCL packets with ihl < 5
  vsock/vmci: fix UAF when peer resets connection during handshake
  netfilter: ip6t_hbh: reject oversized option lists
  net: bcmgenet: keep RBUF EEE/PM disabled
  phonet/pep: disable BH around forwarded sk_receive_skb()
  Bluetooth: hci_uart: fix UAFs and race conditions in close and init paths
  Bluetooth: bnep: Fix UAF read of dev->name
  ALSA: asihpi: Fix potential OOB array access at reading cache
  ALSA: ua101: Reject too-short USB descriptors
  sysfs: don't remove existing directory on update failure
  smb: client: reject userspace cifs.spnego descriptions
  Revert "s390/cio: Fix device lifecycle handling in css_alloc_subchannel()"
  selftests: lib.mk: Also install "config" and "settings"
  s390/debug: Reject zero-length input before trimming a newline
  net/rds: reset op_nents when zerocopy page pin fails
  drm/gma500/oaktrail_hdmi: fix i2c adapter leak on setup
  libceph: Fix potential out-of-bounds access in crush_decode()
  libceph: Fix potential null-ptr-deref in decode_choose_args()
  powerpc/warp: Fix error handling in pika_dtm_thread
  ceph: fix a buffer leak in __ceph_setxattr()
  ALSA: usb-audio: Bound MIDI endpoint descriptor scans
  audit: enforce AUDIT_LOCKED for AUDIT_TRIM and AUDIT_MAKE_EQUIV
  audit: fix incorrect inheritable capability in CAPSET records
  crypto: af_alg - Cap AEAD AD length to 0x80000000
  btrfs: tracepoints: fix sleep while in atomic context in btrfs_sync_file()
  drm/amd/display: Read EDID from VBIOS embedded panel info
  drm/amd/display: Allow DCE link encoder without AUX registers
  net/sched: sch_cake: annotate data-races in cake_dump_stats() (V)
  sctp: discard stale INIT after handshake completion
  ASoC: codecs: ab8500: Fix casting of private data
  NFC: trf7970a: Ignore antenna noise when checking for RF field
  net: usb: rtl8150: free skb on usb_submit_urb() failure in xmit
  net: usb: rtl8150: fix use-after-free in rtl8150_start_xmit()
  vrf: Fix a potential NPD when removing a port from a VRF
  net/sched: sch_choke: annotate data-races in choke_dump_stats()
  net: sched: choke: remove unused variables in struct choke_sched_data
  net/sched: netem: validate slot configuration
  net/sched: netem: fix probability gaps in 4-state loss model
  net: sched: sch_netem: Refactor code in 4-state loss generator
  scsi: sr: Add memory allocation failure handling for get_capabilities()
  netfilter: nf_conntrack_sip: don't use simple_strtoul
  netfilter: xt_policy: fix strict mode inbound policy matching
  drm/amdgpu/gfx6: Support harvested SI chips with disabled TCCs (v2)
  netfilter: arp_tables: fix IEEE1394 ARP payload parsing
  tracing: branch: Fix inverted check on stat tracer registration
  mailbox: mailbox-test: initialize struct earlier
  mailbox: mailbox-test: don't free the reused channel
  mailbox: add sanity check for channel array
  cgroup/rdma: fix integer overflow in rdmacg_try_charge()
  mailbox: mailbox-test: free channels on probe error
  fbdev: offb: fix PCI device reference leak on probe failure
  net/sched: sch_sfb: annotate data-races in sfb_dump_stats()
  net/sched: sch_fq_codel: remove data-races from fq_codel_dump_stats()
  net_sched: sch_hhf: annotate data-races in hhf_dump_stats()
  net/rds: zero per-item info buffer before handing it to visitors
  slip: bound decode() reads against the compressed packet length
  netfilter: nfnetlink_osf: fix out-of-bounds read on option matching
  ipvs: fix MTU check for GSO packets in tunnel mode
  netfilter: xtables: restrict several matches to inet family
  netfilter: conntrack: remove sprintf usage
  netfilter: nfnetlink_osf: fix divide-by-zero in OSF_WSS_MODULO
  netfilter: nft_osf: restrict it to ipv4
  openvswitch: cap upcall PID array size and pre-size vport replies
  dissector: do not set invalid PPP protocol
  sctp: fix OOB write to userspace in sctp_getsockopt_peer_auth_chunks
  ipv6: fix possible UAF in icmpv6_rcv()
  e1000e: Unroll PTP in probe error handling
  i40e: don't advertise IFF_SUPP_NOFCS
  PCMCIA: Fix garbled log messages for KERN_CONT
  clk: xgene: Fix mapping leak in xgene_pllclk_init()
  clk: qoriq: avoid format string warning
  clk: imx: imx6q: Fix device node reference leak in of_assigned_ldb_sels()
  scsi: target: core: Fix integer overflow in UNMAP bounds check
  scsi: sg: Resolve soft lockup issue when opening /dev/sgX
  RDMA/core: Prefer NLA_NUL_STRING
  nfs/blocklayout: Fix compilation error (`make W=1`) in bl_write_pagelist()
  mfd: mc13xxx-core: Fix memory leak in mc13xxx_add_subdevice_pdata()
  tty: hvc_iucv: fix off-by-one in number of supported devices
  tty: hvc: remove HVC_IUCV_MAGIC
  platform/surface: surfacepro3_button: Drop wakeup source on remove
  driver core: device.h: remove extern from function prototypes
  perf util: Kill die() prototype, dead for a long time
  pinctrl: abx500: Fix type of 'argument' variable
  pinctrl: pinctrl-pic32: Fix resource leak
  bpf: Fix precedence bug in convert_bpf_ld_abs alignment check
  HID: usbhid: fix deadlock in hid_post_reset()
  mtd: rawnand: sunxi: fix sunxi_nfc_hw_ecc_read_extra_oob
  HID: asus: do not abort probe when not necessary
  HID: asus: make asus_resume adhere to linux kernel coding standards
  ima: check return value of crypto_shash_final() in boot aggregate
  tracing: Rebuild full_name on each hist_field_name() call
  ocfs2: validate group add input before caching
  ocfs2: validate bg_bits during freefrag scan
  ocfs2: fix listxattr handling when the buffer is full
  ocfs2/dlm: fix off-by-one in dlm_match_regions() region comparison
  ocfs2/dlm: validate qr_numregions in dlm_match_regions()
  memory: tegra124-emc: Fix dll_change check
  ARM: dts: mediatek: mt7623: fix efuse fallback compatible
  efi/capsule-loader: fix incorrect sizeof in phys array reallocation
  quota: Fix race of dquot_scan_active() with quota deactivation
  ktest: Honor empty per-test option overrides
  ktest: Avoid undef warning when WARNINGS_FILE is unset
  selftest: memcg: skip memcg_sock test if address family not supported
  Documentation: fix a hugetlbfs reservation statement
  ALSA: hda/realtek: fix code style (ERROR: else should follow close brace '}')
  ALSA: hda/realtek: Whitespace fix
  drm/amd/pm/ci: Clear EnabledForActivity field for memory levels
  drm/amd/pm/ci: Fix powertune defaults for Hawaii 0x67B0
  drm/amd/pm/ci: Use highest MCLK on CI when MCLK DPM is disabled
  ALSA: core: Validate compress device numbers without dynamic minors
  ALSA: compress: Drop unused functions
  fbdev: matroxfb: Mark variable with __maybe_unused to avoid W=1 build break
  drm/sun4i: Fix resource leaks
  dm log: fix out-of-bounds write due to region_count overflow
  dm cache metadata: fix memory leak on metadata abort retry
  dm cache: fix concurrent write failure in passthrough mode
  dm cache policy smq: fix missing locks in invalidating cache blocks
  dm cache: fix write path cache coherency in passthrough mode
  dm cache: fix null-deref with concurrent writes in passthrough mode
  ASoC: sti: use managed regmap_field allocations
  ASoC: sti: Return errors from regmap_field_alloc()
  Bluetooth: hci_ldisc: Clear HCI_UART_PROTO_INIT on error
  Bluetooth: L2CAP: Fix printing wrong information if SDU length exceeds MTU
  ppp: require CAP_NET_ADMIN in target netns for unattached ioctls
  net/rds: Optimize rds_ib_laddr_check
  net: hamradio: 6pack: fix uninit-value in sixpack_receive_buf
  6pack: propagage new tty types
  netfilter: nft_fwd_netdev: check ttl/hl before forwarding
  net: bcmgenet: fix off-by-one in bcmgenet_put_txcb
  wifi: rtlwifi: pci: fix possible use-after-free caused by unfinished irq_prepare_bcn_tasklet
  firmware: dmi: Correct an indexing error in dmi.h
  locking: Fix rwlock support in <linux/spinlock_up.h>
  irqchip/irq-pic32-evic: Address warning related to wrong printf() formatter
  thermal/drivers/spear: Fix error condition for reading st,thermal-flags
  pstore/ram: fix resource leak when ioremap() fails
  nilfs2: reject zero bd_oblocknr in nilfs_ioctl_mark_blocks_dirty()
  drbd: Balance RCU calls in drbd_adm_dump_devices()
  fs/omfs: reject s_sys_blocksize smaller than OMFS_DIR_START
  Bluetooth: L2CAP: Fix null-ptr-deref in l2cap_sock_get_sndtimeo_cb()
  batman-adv: bla: put backbone reference on failed claim hash insert
  batman-adv: bla: only purge non-released claims
  batman-adv: bla: prevent use-after-free when deleting claims
  batman-adv: reject new tp_meter sessions during teardown
  batman-adv: fix integer overflow on buff_pos
  sctp: revalidate list cursor after sctp_sendmsg_to_asoc() in SCTP_SENDALL
  drm/amdgpu/pm: align Hawaii mclk workaround with radeon
  drm/amdgpu/pm: add missing revision check for CI
  drm/amdgpu/sdma4: replace BUG_ON with WARN_ON in fence emission
  drm/amdgpu/gfx9: drop unnecessary 64-bit fence flag check in KIQ
  drm/radeon: add missing revision check for CI
  spi: mpc52xx: fix use-after-free on unbind
  media: dib8000: avoid division by 0 in dib8000_set_dds()
  media: rc: streamzap: Error handling in probe
  media: uvcvideo: Enable VB2_DMABUF for metadata stream
  RDMA/rxe: Reject unknown opcodes before ICRC processing
  RDMA/ocrdma: Don't NULL deref uctx on errors in ocrdma_copy_pd_uresp()
  power: supply: max17042: avoid overflow when determining health
  PCI/AER: Stop ruling out unbound devices as error source
  s390/debug: Reject zero-length input in debug_input_flush_fn()
  nvmet: avoid recursive nvmet-wq flush in nvmet_ctrl_free
  md/raid10: fix divide-by-zero in setup_geo() with zero far_copies
  isofs: validate block number from NFS file handle in isofs_export_iget
  isofs: validate Rock Ridge CE continuation extent against volume size
  dm-verity-fec: correctly reject too-small hash devices
  dm-verity-fec: correctly reject too-small FEC devices
  dm: fix a buffer overflow in ioctl processing
  dm: don't report warning when doing deferred remove
  cpuidle: powerpc: avoid double clear when breaking snooze
  spi: topcliff-pch: fix use-after-free on unbind
  udf: reject descriptors with oversized CRC length
  ibmveth: Disable GSO for packets with small MSS
  parisc: Fix IRQ leak in LASI driver
  net/rds: handle zerocopy send cleanup before the message is queued
  ip6_gre: Use cached t->net in ip6erspan_changelink().
  sound: ua101: fix division by zero at probe
  Bluetooth: L2CAP: Fix null-ptr-deref in l2cap_sock_state_change_cb()
  Bluetooth: L2CAP: Fix null-ptr-deref in l2cap_sock_new_connection_cb()
  usb: ulpi: fix memory leak on ulpi_register() error paths
  USB: serial: option: add Telit Cinterion LE910Cx compositions
  USB: omap_udc: DMA: Don't enable burst 4 mode
  ALSA: usb-audio: Fix UAC3 cluster descriptor size check
  ALSA: usb-audio: Avoid potential endless loop in convert_chmap_v3()
  usb: usblp: fix uninitialized heap leak via LPGETSTATUS ioctl
  usb: usblp: fix heap leak in IEEE 1284 device ID via short response
  wifi: b43: enforce bounds check on firmware key index in b43_rx()
  wifi: ath5k: do not access array OOB
  wifi: rsi: fix kthread lifetime race between self-exit and external-stop
  wifi: b43legacy: enforce bounds check on firmware key index in RX path
  ipmi:ssif: NULL thread on error
  ipmi:ssif: Remove unnecessary indention
  ipmi:ssif: Clean up kthread on errors
  ipmi:ssif: Fix a shutdown race
  net/sched: sch_red: Replace direct dequeue call with peek and qdisc_dequeue_peeked
  fbdev: udlfb: add vm_ops to dlfb_ops_mmap to prevent use-after-free
  ipmi:si: Return state to normal if message allocation fails
  ipmi: Check event message buffer response for bad data
  ipmi: Add limits to event and receive message requests
  scsi: target: configfs: Bound snprintf() return in tg_pt_gp_members_show()
  ALSA: caiaq: fix usb_dev refcount leak on probe failure
  ALSA: caiaq: Don't abort when no input device is available
  ALSA: caiaq: Fix potentially leftover ep1_in_urb at error path
  dm mirror: fix integer overflow in create_dirty_log()
  crypto: atmel-tdes - fix DMA sync direction
  crypto: ccree - fix a memory leak in cc_mac_digest()
  crypto: hisilicon - Fix dma_unmap_single() direction
  crypto: atmel-ecc - Release client on allocation failure
  crypto: atmel-aes - Fix 3-page memory leak in atmel_aes_buff_cleanup
  taskstats: set version in TGID exit notifications
  inotify: fix watch count leak when fsnotify_add_inode_mark_locked() fails
  md/raid5: validate payload size before accessing journal metadata
  md/raid5: fix soft lockup in retry_aligned_read()
  ext4: fix missing brelse() in ext4_xattr_inode_dec_ref_all()
  userfaultfd: allow registration of ranges below mmap_min_addr
  tpm: tpm_tis: add error logging for data transfer
  mmc: block: use single block write in retry
  RDMA/rxe: Validate pad and ICRC before payload_size() in rxe_rcv
  ALSA: 6fire: Fix input volume change detection
  ALSA: caiaq: Handle probe errors properly
  ALSA: caiaq: Fix control_put() result and cache rollback
  ALSA: seq_oss: return full count for successful SEQ_FULLSIZE writes
  ALSA: ctxfi: Add fallback to default RSR for S/PDIF
  lib/ts_kmp: fix integer overflow in pattern length calculation
  Revert "ALSA: usb: Increase volume range that triggers a warning"
  net: strparser: fix skb_head leak in strp_abort_strp()
  net: caif: clear client service pointer on teardown
  ALSA: control: Validate buf_len before strnlen() in snd_ctl_elem_init_enum_names()
  crypto: pcrypt - Fix handling of MAY_BACKLOG requests
  um: drivers: call kernel_strrchr() explicitly in cow_user.c
  firmware: google: framebuffer: Do not mark framebuffer as busy
  ibmasm: fix heap over-read in ibmasm_send_i2o_message()
  ibmasm: fix OOB reads in command_file_write due to missing size checks
  misc: ibmasm: fix OOB MMIO read in ibmasm_handle_mouse_interrupt()
  ALSA: usb-audio: Fix Audio Advantage Micro II SPDIF switch
  ALSA: usb-audio: Avoid false E-MU sample-rate notifications
  ALSA: usb-audio: stop parsing UAC2 rates at MAX_NR_RATES
  rxrpc: Fix missing validation of ticket length in non-XDR key preparsing
  ALSA: caiaq: take a reference on the USB device in create_card()
  ALSA: usb-audio: apply quirk for MOONDROP JU Jiu
  rxrpc: Fix anonymous key handling
  scripts/dtc: Remove unused dts_version in dtc-lexer.l
  gfs2: Validate i_depth for exhash directories
  mm: blk-cgroup: fix use-after-free in cgwb_release_workfn()
  ocfs2: fix possible deadlock between unlink and dio_end_io_write
  fs/ocfs2: fix comments mentioning i_mutex
  rxrpc: reject undecryptable rxkad response tickets
  ocfs2: fix out-of-bounds write in ocfs2_write_end_inline
  ocfs2: validate inline data i_size during inode read
  ocfs2: add inline inode consistency check to ocfs2_validate_inode_block()
  xfrm: clear trailing padding in build_polexpire()
  rxrpc: fix reference count leak in rxrpc_server_keyring()
  mailbox: Prevent out-of-bounds access in of_mbox_index_xlate()
  wifi: mac80211: always free skb on ieee80211_tx_prepare_skb() failure
  ipv6: add NULL checks for idev in SRv6 paths
  net: tap: NULL pointer derefence in dev_parse_header_protocol when skb->dev is null
  media: hackrf: fix to not free memory after the device is registered in hackrf_probe()
  nilfs2: fix NULL i_assoc_inode dereference in nilfs_mdt_save_to_shadow_map
  media: as102: fix to not free memory after the device is registered in as102_usb_probe()
  ALSA: 6fire: fix use-after-free on disconnect
  media: em28xx: fix use-after-free in em28xx_v4l2_open()
  mm/kasan: fix double free for kasan pXds
  KVM: x86: Use scratch field in MMIO fragment to hold small write values
  media: uvcvideo: Allow extra entities
  Revert "wifi: cfg80211: stop NAN and P2P in cfg80211_leave"
  rxrpc: Fix call removal to use RCU safe deletion
  ACPI: property: Constify stubs for CONFIG_ACPI=n case
  ocfs2: handle invalid dinode in ocfs2_group_extend
  ocfs2: fix use-after-free in ocfs2_fault() when VM_FAULT_RETRY
  ALSA: ctxfi: Limit PTP to a single page
  USB: serial: option: add Telit Cinterion FN990A MBIM composition
  fbdev: udlfb: avoid divide-by-zero on FBIOPUT_VSCREENINFO
  usb: storage: Expand range of matched versions for VL817 quirks entry
  usbip: validate number_of_packets in usbip_pack_ret_submit()
  usb: gadget: renesas_usb3: validate endpoint index in standard request handlers
  usb: gadget: f_phonet: fix skb frags[] overflow in pn_rx_complete()
  usb: gadget: f_ncm: validate minimum block_len in ncm_unwrap_ntb()
  fbdev: tdfxfb: avoid divide-by-zero on FBIOPUT_VSCREENINFO
  ALSA: fireworks: bound device-supplied status before string array lookup
  NFC: digital: Bounds check NFC-A cascade depth in SDD response handler
  net: usb: cdc-phonet: fix skb frags[] overflow in rx_complete()
  HID: core: clamp report_size in s32ton() to avoid undefined shift
  HID: alps: fix NULL pointer dereference in alps_raw_event()
  staging: rtl8723bs: initialize le_tmp64 in rtw_BIP_verify()
  i2c: s3c24xx: check the size of the SMBUS message before using it
  nfc: llcp: add missing return after LLCP_CLOSED checks
  MIPS: mm: Suppress TLB uniquification on EHINV hardware
  af_unix: read UNIX_DIAG_VFS data under unix_state_lock
  netfilter: ip6t_eui64: reject invalid MAC header for all packets
  netfilter: xt_multiport: validate range encoding in checkentry
  netfilter: nfnetlink_log: initialize nfgenmsg in NLMSG_DONE terminator
  xfrm_user: fix info leak in build_mapping()
  e1000: check return value of e1000_read_eeprom
  net: lapbether: replace comparison to NULL with "lapbeth_get_x25_dev"
  net: lapbether: Close the LAPB device before its underlying Ethernet device closes
  net: sched: act_csum: validate nested VLAN headers
  drm/vc4: Protect madv read in vc4_gem_object_mmap() with madv_lock
  drm/vc4: Fix a memory leak in hang state error path
  drm/vc4: Fix memory leak of BO array in hang state
  ASoC: stm32_sai: fix incorrect BCLK polarity for DSP_A/B, LEFT_J
  wifi: brcmfmac: validate bsscfg indices in IF events
  ata: ahci: force 32-bit DMA for JMicron JMB582/JMB585
  HID: roccat: fix use-after-free in roccat_report_event
  HID: quirks: add HID_QUIRK_ALWAYS_POLL for 8BitDo Pro 3
  wifi: wl1251: validate packet IDs before indexing tx_frames
  btrfs: tracepoints: get correct superblock from dentry in event btrfs_sync_file()
  ALSA: asihpi: avoid write overflow check warning
  net: skbuff: preserve shared-frag marker during coalescing

Change-Id: I535ecdb1a77312a12564125fedb9f99b3a5533d1
2026-07-09 22:06:07 +01:00

1202 lines
28 KiB
C

/*
* Block device elevator/IO-scheduler.
*
* Copyright (C) 2000 Andrea Arcangeli <andrea@suse.de> SuSE
*
* 30042000 Jens Axboe <axboe@kernel.dk> :
*
* Split the elevator a bit so that it is possible to choose a different
* one or even write a new "plug in". There are three pieces:
* - elevator_fn, inserts a new request in the queue list
* - elevator_merge_fn, decides whether a new buffer can be merged with
* an existing request
* - elevator_dequeue_fn, called when a request is taken off the active list
*
* 20082000 Dave Jones <davej@suse.de> :
* Removed tests for max-bomb-segments, which was breaking elvtune
* when run without -bN
*
* Jens:
* - Rework again to work with bio instead of buffer_heads
* - loose bi_dev comparisons, partition handling is right now
* - completely modularize elevator setup and teardown
*
*/
#include <linux/kernel.h>
#include <linux/fs.h>
#include <linux/blkdev.h>
#include <linux/elevator.h>
#include <linux/bio.h>
#include <linux/module.h>
#include <linux/slab.h>
#include <linux/init.h>
#include <linux/compiler.h>
#include <linux/blktrace_api.h>
#include <linux/hash.h>
#include <linux/uaccess.h>
#include <linux/pm_runtime.h>
#include <linux/blk-cgroup.h>
#include <trace/events/block.h>
#include "blk.h"
#include "blk-mq-sched.h"
#include "blk-wbt.h"
static DEFINE_SPINLOCK(elv_list_lock);
static LIST_HEAD(elv_list);
/*
* Merge hash stuff.
*/
#define rq_hash_key(rq) (blk_rq_pos(rq) + blk_rq_sectors(rq))
/*
* Query io scheduler to see if the current process issuing bio may be
* merged with rq.
*/
static int elv_iosched_allow_bio_merge(struct request *rq, struct bio *bio)
{
struct request_queue *q = rq->q;
struct elevator_queue *e = q->elevator;
if (e->uses_mq && e->type->ops.mq.allow_merge)
return e->type->ops.mq.allow_merge(q, rq, bio);
else if (!e->uses_mq && e->type->ops.sq.elevator_allow_bio_merge_fn)
return e->type->ops.sq.elevator_allow_bio_merge_fn(q, rq, bio);
return 1;
}
/*
* can we safely merge with this request?
*/
bool elv_bio_merge_ok(struct request *rq, struct bio *bio)
{
if (!blk_rq_merge_ok(rq, bio))
return false;
if (!elv_iosched_allow_bio_merge(rq, bio))
return false;
return true;
}
EXPORT_SYMBOL(elv_bio_merge_ok);
static bool elevator_match(const struct elevator_type *e, const char *name)
{
if (!strcmp(e->elevator_name, name))
return true;
if (e->elevator_alias && !strcmp(e->elevator_alias, name))
return true;
return false;
}
/*
* Return scheduler with name 'name' and with matching 'mq capability
*/
static struct elevator_type *elevator_find(const char *name, bool mq)
{
struct elevator_type *e;
list_for_each_entry(e, &elv_list, list) {
if (elevator_match(e, name) && (mq == e->uses_mq))
return e;
}
return NULL;
}
static void elevator_put(struct elevator_type *e)
{
module_put(e->elevator_owner);
}
static struct elevator_type *elevator_get(struct request_queue *q,
const char *name, bool try_loading)
{
struct elevator_type *e;
spin_lock(&elv_list_lock);
e = elevator_find(name, q->mq_ops != NULL);
if (!e && try_loading) {
spin_unlock(&elv_list_lock);
request_module("%s-iosched", name);
spin_lock(&elv_list_lock);
e = elevator_find(name, q->mq_ops != NULL);
}
if (e && !try_module_get(e->elevator_owner))
e = NULL;
spin_unlock(&elv_list_lock);
return e;
}
static char chosen_elevator[ELV_NAME_MAX];
static int __init elevator_setup(char *str)
{
/*
* Be backwards-compatible with previous kernels, so users
* won't get the wrong elevator.
*/
strncpy(chosen_elevator, str, sizeof(chosen_elevator) - 1);
return 1;
}
__setup("elevator=", elevator_setup);
/* called during boot to load the elevator chosen by the elevator param */
void __init load_default_elevator_module(void)
{
struct elevator_type *e;
if (!chosen_elevator[0])
return;
/*
* Boot parameter is deprecated, we haven't supported that for MQ.
* Only look for non-mq schedulers from here.
*/
spin_lock(&elv_list_lock);
e = elevator_find(chosen_elevator, false);
spin_unlock(&elv_list_lock);
if (!e)
request_module("%s-iosched", chosen_elevator);
}
static struct kobj_type elv_ktype;
struct elevator_queue *elevator_alloc(struct request_queue *q,
struct elevator_type *e)
{
struct elevator_queue *eq;
eq = kzalloc_node(sizeof(*eq), GFP_KERNEL, q->node);
if (unlikely(!eq))
return NULL;
eq->type = e;
kobject_init(&eq->kobj, &elv_ktype);
mutex_init(&eq->sysfs_lock);
hash_init(eq->hash);
eq->uses_mq = e->uses_mq;
return eq;
}
EXPORT_SYMBOL(elevator_alloc);
static void elevator_release(struct kobject *kobj)
{
struct elevator_queue *e;
e = container_of(kobj, struct elevator_queue, kobj);
elevator_put(e->type);
kfree(e);
}
/*
* Use the default elevator specified by config boot param for non-mq devices,
* or by config option. Don't try to load modules as we could be running off
* async and request_module() isn't allowed from async.
*/
int elevator_init(struct request_queue *q)
{
struct elevator_type *e = NULL;
int err = 0;
/*
* q->sysfs_lock must be held to provide mutual exclusion between
* elevator_switch() and here.
*/
mutex_lock(&q->sysfs_lock);
if (unlikely(q->elevator))
goto out_unlock;
if (*chosen_elevator) {
e = elevator_get(q, chosen_elevator, false);
if (!e)
printk(KERN_ERR "I/O scheduler %s not found\n",
chosen_elevator);
}
if (!e)
e = elevator_get(q, CONFIG_DEFAULT_IOSCHED, false);
if (!e) {
printk(KERN_ERR
"Default I/O scheduler not found. Using noop.\n");
e = elevator_get(q, "noop", false);
}
err = e->ops.sq.elevator_init_fn(q, e);
if (err)
elevator_put(e);
out_unlock:
mutex_unlock(&q->sysfs_lock);
return err;
}
void elevator_exit(struct request_queue *q, struct elevator_queue *e)
{
mutex_lock(&e->sysfs_lock);
if (e->uses_mq && e->type->ops.mq.exit_sched)
blk_mq_exit_sched(q, e);
else if (!e->uses_mq && e->type->ops.sq.elevator_exit_fn)
e->type->ops.sq.elevator_exit_fn(e);
mutex_unlock(&e->sysfs_lock);
kobject_put(&e->kobj);
}
static inline void __elv_rqhash_del(struct request *rq)
{
hash_del(&rq->hash);
rq->rq_flags &= ~RQF_HASHED;
}
void elv_rqhash_del(struct request_queue *q, struct request *rq)
{
if (ELV_ON_HASH(rq))
__elv_rqhash_del(rq);
}
EXPORT_SYMBOL_GPL(elv_rqhash_del);
void elv_rqhash_add(struct request_queue *q, struct request *rq)
{
struct elevator_queue *e = q->elevator;
BUG_ON(ELV_ON_HASH(rq));
hash_add(e->hash, &rq->hash, rq_hash_key(rq));
rq->rq_flags |= RQF_HASHED;
}
EXPORT_SYMBOL_GPL(elv_rqhash_add);
void elv_rqhash_reposition(struct request_queue *q, struct request *rq)
{
__elv_rqhash_del(rq);
elv_rqhash_add(q, rq);
}
struct request *elv_rqhash_find(struct request_queue *q, sector_t offset)
{
struct elevator_queue *e = q->elevator;
struct hlist_node *next;
struct request *rq;
hash_for_each_possible_safe(e->hash, rq, next, hash, offset) {
BUG_ON(!ELV_ON_HASH(rq));
if (unlikely(!rq_mergeable(rq))) {
__elv_rqhash_del(rq);
continue;
}
if (rq_hash_key(rq) == offset)
return rq;
}
return NULL;
}
/*
* RB-tree support functions for inserting/lookup/removal of requests
* in a sorted RB tree.
*/
void elv_rb_add(struct rb_root *root, struct request *rq)
{
struct rb_node **p = &root->rb_node;
struct rb_node *parent = NULL;
struct request *__rq;
while (*p) {
parent = *p;
__rq = rb_entry(parent, struct request, rb_node);
if (blk_rq_pos(rq) < blk_rq_pos(__rq))
p = &(*p)->rb_left;
else if (blk_rq_pos(rq) >= blk_rq_pos(__rq))
p = &(*p)->rb_right;
}
rb_link_node(&rq->rb_node, parent, p);
rb_insert_color(&rq->rb_node, root);
}
EXPORT_SYMBOL(elv_rb_add);
void elv_rb_del(struct rb_root *root, struct request *rq)
{
BUG_ON(RB_EMPTY_NODE(&rq->rb_node));
rb_erase(&rq->rb_node, root);
RB_CLEAR_NODE(&rq->rb_node);
}
EXPORT_SYMBOL(elv_rb_del);
struct request *elv_rb_find(struct rb_root *root, sector_t sector)
{
struct rb_node *n = root->rb_node;
struct request *rq;
while (n) {
rq = rb_entry(n, struct request, rb_node);
if (sector < blk_rq_pos(rq))
n = n->rb_left;
else if (sector > blk_rq_pos(rq))
n = n->rb_right;
else
return rq;
}
return NULL;
}
EXPORT_SYMBOL(elv_rb_find);
/*
* Insert rq into dispatch queue of q. Queue lock must be held on
* entry. rq is sort instead into the dispatch queue. To be used by
* specific elevators.
*/
void elv_dispatch_sort(struct request_queue *q, struct request *rq)
{
sector_t boundary;
struct list_head *entry;
if (q->last_merge == rq)
q->last_merge = NULL;
elv_rqhash_del(q, rq);
q->nr_sorted--;
boundary = q->end_sector;
list_for_each_prev(entry, &q->queue_head) {
struct request *pos = list_entry_rq(entry);
if (req_op(rq) != req_op(pos))
break;
if (rq_data_dir(rq) != rq_data_dir(pos))
break;
if (pos->rq_flags & (RQF_STARTED | RQF_SOFTBARRIER))
break;
if (blk_rq_pos(rq) >= boundary) {
if (blk_rq_pos(pos) < boundary)
continue;
} else {
if (blk_rq_pos(pos) >= boundary)
break;
}
if (blk_rq_pos(rq) >= blk_rq_pos(pos))
break;
}
list_add(&rq->queuelist, entry);
}
EXPORT_SYMBOL(elv_dispatch_sort);
/*
* Insert rq into dispatch queue of q. Queue lock must be held on
* entry. rq is added to the back of the dispatch queue. To be used by
* specific elevators.
*/
void elv_dispatch_add_tail(struct request_queue *q, struct request *rq)
{
if (q->last_merge == rq)
q->last_merge = NULL;
elv_rqhash_del(q, rq);
q->nr_sorted--;
q->end_sector = rq_end_sector(rq);
q->boundary_rq = rq;
list_add_tail(&rq->queuelist, &q->queue_head);
}
EXPORT_SYMBOL(elv_dispatch_add_tail);
enum elv_merge elv_merge(struct request_queue *q, struct request **req,
struct bio *bio)
{
struct elevator_queue *e = q->elevator;
struct request *__rq;
/*
* Levels of merges:
* nomerges: No merges at all attempted
* noxmerges: Only simple one-hit cache try
* merges: All merge tries attempted
*/
if (blk_queue_nomerges(q) || !bio_mergeable(bio))
return ELEVATOR_NO_MERGE;
/*
* First try one-hit cache.
*/
if (q->last_merge && elv_bio_merge_ok(q->last_merge, bio)) {
enum elv_merge ret = blk_try_merge(q->last_merge, bio);
if (ret != ELEVATOR_NO_MERGE) {
*req = q->last_merge;
return ret;
}
}
if (blk_queue_noxmerges(q))
return ELEVATOR_NO_MERGE;
/*
* See if our hash lookup can find a potential backmerge.
*/
__rq = elv_rqhash_find(q, bio->bi_iter.bi_sector);
if (__rq && elv_bio_merge_ok(__rq, bio)) {
*req = __rq;
return ELEVATOR_BACK_MERGE;
}
if (e->uses_mq && e->type->ops.mq.request_merge)
return e->type->ops.mq.request_merge(q, req, bio);
else if (!e->uses_mq && e->type->ops.sq.elevator_merge_fn)
return e->type->ops.sq.elevator_merge_fn(q, req, bio);
return ELEVATOR_NO_MERGE;
}
/*
* Attempt to do an insertion back merge. Only check for the case where
* we can append 'rq' to an existing request, so we can throw 'rq' away
* afterwards.
*
* Returns true if we merged, false otherwise
*/
bool elv_attempt_insert_merge(struct request_queue *q, struct request *rq)
{
struct request *__rq;
bool ret;
if (blk_queue_nomerges(q))
return false;
/*
* First try one-hit cache.
*/
if (q->last_merge && blk_attempt_req_merge(q, q->last_merge, rq))
return true;
if (blk_queue_noxmerges(q))
return false;
ret = false;
/*
* See if our hash lookup can find a potential backmerge.
*/
while (1) {
__rq = elv_rqhash_find(q, blk_rq_pos(rq));
if (!__rq || !blk_attempt_req_merge(q, __rq, rq))
break;
/* The merged request could be merged with others, try again */
ret = true;
rq = __rq;
}
return ret;
}
void elv_merged_request(struct request_queue *q, struct request *rq,
enum elv_merge type)
{
struct elevator_queue *e = q->elevator;
if (e->uses_mq && e->type->ops.mq.request_merged)
e->type->ops.mq.request_merged(q, rq, type);
else if (!e->uses_mq && e->type->ops.sq.elevator_merged_fn)
e->type->ops.sq.elevator_merged_fn(q, rq, type);
if (type == ELEVATOR_BACK_MERGE)
elv_rqhash_reposition(q, rq);
q->last_merge = rq;
}
void elv_merge_requests(struct request_queue *q, struct request *rq,
struct request *next)
{
struct elevator_queue *e = q->elevator;
bool next_sorted = false;
if (e->uses_mq && e->type->ops.mq.requests_merged)
e->type->ops.mq.requests_merged(q, rq, next);
else if (e->type->ops.sq.elevator_merge_req_fn) {
next_sorted = (__force bool)(next->rq_flags & RQF_SORTED);
if (next_sorted)
e->type->ops.sq.elevator_merge_req_fn(q, rq, next);
}
elv_rqhash_reposition(q, rq);
if (next_sorted) {
elv_rqhash_del(q, next);
q->nr_sorted--;
}
q->last_merge = rq;
}
void elv_bio_merged(struct request_queue *q, struct request *rq,
struct bio *bio)
{
struct elevator_queue *e = q->elevator;
if (WARN_ON_ONCE(e->uses_mq))
return;
if (e->type->ops.sq.elevator_bio_merged_fn)
e->type->ops.sq.elevator_bio_merged_fn(q, rq, bio);
}
#ifdef CONFIG_PM
static void blk_pm_requeue_request(struct request *rq)
{
if (rq->q->dev && !(rq->rq_flags & RQF_PM) &&
(rq->rq_flags & (RQF_PM_ADDED | RQF_FLUSH_SEQ))) {
rq->rq_flags &= ~RQF_PM_ADDED;
rq->q->nr_pending--;
}
}
static void blk_pm_add_request(struct request_queue *q, struct request *rq)
{
if (q->dev && !(rq->rq_flags & RQF_PM)) {
rq->rq_flags |= RQF_PM_ADDED;
if (q->nr_pending++ == 0 &&
(q->rpm_status == RPM_SUSPENDED ||
q->rpm_status == RPM_SUSPENDING))
pm_request_resume(q->dev);
}
}
#else
static inline void blk_pm_requeue_request(struct request *rq) {}
static inline void blk_pm_add_request(struct request_queue *q,
struct request *rq)
{
}
#endif
void elv_requeue_request(struct request_queue *q, struct request *rq)
{
/*
* it already went through dequeue, we need to decrement the
* in_flight count again
*/
if (blk_account_rq(rq)) {
q->in_flight[rq_is_sync(rq)]--;
if (rq->rq_flags & RQF_SORTED)
elv_deactivate_rq(q, rq);
}
rq->rq_flags &= ~RQF_STARTED;
blk_pm_requeue_request(rq);
__elv_add_request(q, rq, ELEVATOR_INSERT_REQUEUE);
}
void elv_drain_elevator(struct request_queue *q)
{
struct elevator_queue *e = q->elevator;
static int printed;
if (WARN_ON_ONCE(e->uses_mq))
return;
lockdep_assert_held(q->queue_lock);
while (e->type->ops.sq.elevator_dispatch_fn(q, 1))
;
if (q->nr_sorted && !blk_queue_is_zoned(q) && printed++ < 10 ) {
printk(KERN_ERR "%s: forced dispatching is broken "
"(nr_sorted=%u), please report this\n",
q->elevator->type->elevator_name, q->nr_sorted);
}
}
void __elv_add_request(struct request_queue *q, struct request *rq, int where)
{
trace_block_rq_insert(q, rq);
blk_pm_add_request(q, rq);
rq->q = q;
if (rq->rq_flags & RQF_SOFTBARRIER) {
/* barriers are scheduling boundary, update end_sector */
if (!blk_rq_is_passthrough(rq)) {
q->end_sector = rq_end_sector(rq);
q->boundary_rq = rq;
}
} else if (!(rq->rq_flags & RQF_ELVPRIV) &&
(where == ELEVATOR_INSERT_SORT ||
where == ELEVATOR_INSERT_SORT_MERGE))
where = ELEVATOR_INSERT_BACK;
switch (where) {
case ELEVATOR_INSERT_REQUEUE:
case ELEVATOR_INSERT_FRONT:
rq->rq_flags |= RQF_SOFTBARRIER;
list_add(&rq->queuelist, &q->queue_head);
break;
case ELEVATOR_INSERT_BACK:
rq->rq_flags |= RQF_SOFTBARRIER;
elv_drain_elevator(q);
list_add_tail(&rq->queuelist, &q->queue_head);
/*
* We kick the queue here for the following reasons.
* - The elevator might have returned NULL previously
* to delay requests and returned them now. As the
* queue wasn't empty before this request, ll_rw_blk
* won't run the queue on return, resulting in hang.
* - Usually, back inserted requests won't be merged
* with anything. There's no point in delaying queue
* processing.
*/
__blk_run_queue(q);
break;
case ELEVATOR_INSERT_SORT_MERGE:
/*
* If we succeed in merging this request with one in the
* queue already, we are done - rq has now been freed,
* so no need to do anything further.
*/
if (elv_attempt_insert_merge(q, rq))
break;
/* fall through */
case ELEVATOR_INSERT_SORT:
BUG_ON(blk_rq_is_passthrough(rq));
rq->rq_flags |= RQF_SORTED;
q->nr_sorted++;
if (rq_mergeable(rq)) {
elv_rqhash_add(q, rq);
if (!q->last_merge)
q->last_merge = rq;
}
/*
* Some ioscheds (cfq) run q->request_fn directly, so
* rq cannot be accessed after calling
* elevator_add_req_fn.
*/
q->elevator->type->ops.sq.elevator_add_req_fn(q, rq);
break;
case ELEVATOR_INSERT_FLUSH:
rq->rq_flags |= RQF_SOFTBARRIER;
blk_insert_flush(rq);
break;
default:
printk(KERN_ERR "%s: bad insertion point %d\n",
__func__, where);
BUG();
}
}
EXPORT_SYMBOL(__elv_add_request);
void elv_add_request(struct request_queue *q, struct request *rq, int where)
{
unsigned long flags;
spin_lock_irqsave(q->queue_lock, flags);
__elv_add_request(q, rq, where);
spin_unlock_irqrestore(q->queue_lock, flags);
}
EXPORT_SYMBOL(elv_add_request);
struct request *elv_latter_request(struct request_queue *q, struct request *rq)
{
struct elevator_queue *e = q->elevator;
if (e->uses_mq && e->type->ops.mq.next_request)
return e->type->ops.mq.next_request(q, rq);
else if (!e->uses_mq && e->type->ops.sq.elevator_latter_req_fn)
return e->type->ops.sq.elevator_latter_req_fn(q, rq);
return NULL;
}
struct request *elv_former_request(struct request_queue *q, struct request *rq)
{
struct elevator_queue *e = q->elevator;
if (e->uses_mq && e->type->ops.mq.former_request)
return e->type->ops.mq.former_request(q, rq);
if (!e->uses_mq && e->type->ops.sq.elevator_former_req_fn)
return e->type->ops.sq.elevator_former_req_fn(q, rq);
return NULL;
}
int elv_set_request(struct request_queue *q, struct request *rq,
struct bio *bio, gfp_t gfp_mask)
{
struct elevator_queue *e = q->elevator;
if (WARN_ON_ONCE(e->uses_mq))
return 0;
if (e->type->ops.sq.elevator_set_req_fn)
return e->type->ops.sq.elevator_set_req_fn(q, rq, bio, gfp_mask);
return 0;
}
void elv_put_request(struct request_queue *q, struct request *rq)
{
struct elevator_queue *e = q->elevator;
if (WARN_ON_ONCE(e->uses_mq))
return;
if (e->type->ops.sq.elevator_put_req_fn)
e->type->ops.sq.elevator_put_req_fn(rq);
}
int elv_may_queue(struct request_queue *q, unsigned int op)
{
struct elevator_queue *e = q->elevator;
if (WARN_ON_ONCE(e->uses_mq))
return 0;
if (e->type->ops.sq.elevator_may_queue_fn)
return e->type->ops.sq.elevator_may_queue_fn(q, op);
return ELV_MQUEUE_MAY;
}
void elv_completed_request(struct request_queue *q, struct request *rq)
{
struct elevator_queue *e = q->elevator;
if (WARN_ON_ONCE(e->uses_mq))
return;
/*
* request is released from the driver, io must be done
*/
if (blk_account_rq(rq)) {
q->in_flight[rq_is_sync(rq)]--;
if ((rq->rq_flags & RQF_SORTED) &&
e->type->ops.sq.elevator_completed_req_fn)
e->type->ops.sq.elevator_completed_req_fn(q, rq);
}
}
#define to_elv(atr) container_of((atr), struct elv_fs_entry, attr)
static ssize_t
elv_attr_show(struct kobject *kobj, struct attribute *attr, char *page)
{
struct elv_fs_entry *entry = to_elv(attr);
struct elevator_queue *e;
ssize_t error;
if (!entry->show)
return -EIO;
e = container_of(kobj, struct elevator_queue, kobj);
mutex_lock(&e->sysfs_lock);
error = e->type ? entry->show(e, page) : -ENOENT;
mutex_unlock(&e->sysfs_lock);
return error;
}
static ssize_t
elv_attr_store(struct kobject *kobj, struct attribute *attr,
const char *page, size_t length)
{
struct elv_fs_entry *entry = to_elv(attr);
struct elevator_queue *e;
ssize_t error;
if (!entry->store)
return -EIO;
e = container_of(kobj, struct elevator_queue, kobj);
mutex_lock(&e->sysfs_lock);
error = e->type ? entry->store(e, page, length) : -ENOENT;
mutex_unlock(&e->sysfs_lock);
return error;
}
static const struct sysfs_ops elv_sysfs_ops = {
.show = elv_attr_show,
.store = elv_attr_store,
};
static struct kobj_type elv_ktype = {
.sysfs_ops = &elv_sysfs_ops,
.release = elevator_release,
};
int elv_register_queue(struct request_queue *q)
{
struct elevator_queue *e = q->elevator;
int error;
lockdep_assert_held(&q->sysfs_lock);
error = kobject_add(&e->kobj, &q->kobj, "%s", "iosched");
if (!error) {
struct elv_fs_entry *attr = e->type->elevator_attrs;
if (attr) {
while (attr->attr.name) {
if (sysfs_create_file(&e->kobj, &attr->attr))
break;
attr++;
}
}
kobject_uevent(&e->kobj, KOBJ_ADD);
e->registered = 1;
if (!e->uses_mq && e->type->ops.sq.elevator_registered_fn)
e->type->ops.sq.elevator_registered_fn(q);
else if (e->uses_mq && e->type->ops.mq.elevator_registered_fn)
e->type->ops.mq.elevator_registered_fn(q);
}
return error;
}
void elv_unregister_queue(struct request_queue *q)
{
lockdep_assert_held(&q->sysfs_lock);
if (q) {
struct elevator_queue *e = q->elevator;
kobject_uevent(&e->kobj, KOBJ_REMOVE);
kobject_del(&e->kobj);
e->registered = 0;
}
}
int elv_register(struct elevator_type *e)
{
char *def = "";
/* create icq_cache if requested */
if (e->icq_size) {
if (WARN_ON(e->icq_size < sizeof(struct io_cq)) ||
WARN_ON(e->icq_align < __alignof__(struct io_cq)))
return -EINVAL;
snprintf(e->icq_cache_name, sizeof(e->icq_cache_name),
"%s_io_cq", e->elevator_name);
e->icq_cache = kmem_cache_create(e->icq_cache_name, e->icq_size,
e->icq_align, 0, NULL);
if (!e->icq_cache)
return -ENOMEM;
}
/* register, don't allow duplicate names */
spin_lock(&elv_list_lock);
if (elevator_find(e->elevator_name, e->uses_mq)) {
spin_unlock(&elv_list_lock);
kmem_cache_destroy(e->icq_cache);
return -EBUSY;
}
list_add_tail(&e->list, &elv_list);
spin_unlock(&elv_list_lock);
/* print pretty message */
if (elevator_match(e, chosen_elevator) ||
(!*chosen_elevator &&
elevator_match(e, CONFIG_DEFAULT_IOSCHED)))
def = " (default)";
printk(KERN_INFO "io scheduler %s registered%s\n", e->elevator_name,
def);
return 0;
}
EXPORT_SYMBOL_GPL(elv_register);
void elv_unregister(struct elevator_type *e)
{
/* unregister */
spin_lock(&elv_list_lock);
list_del_init(&e->list);
spin_unlock(&elv_list_lock);
/*
* Destroy icq_cache if it exists. icq's are RCU managed. Make
* sure all RCU operations are complete before proceeding.
*/
if (e->icq_cache) {
rcu_barrier();
kmem_cache_destroy(e->icq_cache);
e->icq_cache = NULL;
}
}
EXPORT_SYMBOL_GPL(elv_unregister);
static int elevator_switch_mq(struct request_queue *q,
struct elevator_type *new_e)
{
int ret;
lockdep_assert_held(&q->sysfs_lock);
if (q->elevator) {
if (q->elevator->registered)
elv_unregister_queue(q);
ioc_clear_queue(q);
elevator_exit(q, q->elevator);
}
ret = blk_mq_init_sched(q, new_e);
if (ret)
goto out;
if (new_e) {
ret = elv_register_queue(q);
if (ret) {
elevator_exit(q, q->elevator);
goto out;
}
}
if (new_e)
blk_add_trace_msg(q, "elv switch: %s", new_e->elevator_name);
else
blk_add_trace_msg(q, "elv switch: none");
out:
return ret;
}
/*
* For blk-mq devices, we default to using mq-deadline, if available, for single
* queue devices. If deadline isn't available OR we have multiple queues,
* default to "none".
*/
int elevator_init_mq(struct request_queue *q)
{
struct elevator_type *e;
int err = 0;
if (q->nr_hw_queues != 1)
return 0;
WARN_ON_ONCE(test_bit(QUEUE_FLAG_REGISTERED, &q->queue_flags));
if (unlikely(q->elevator))
goto out;
if (IS_ENABLED(CONFIG_IOSCHED_BFQ)) {
e = elevator_get(q, "bfq", false);
if (!e)
goto out;
} else {
e = elevator_get(q, "mq-deadline", false);
if (!e)
goto out;
}
err = blk_mq_init_sched(q, e);
if (err)
elevator_put(e);
out:
return err;
}
/*
* switch to new_e io scheduler. be careful not to introduce deadlocks -
* we don't free the old io scheduler, before we have allocated what we
* need for the new one. this way we have a chance of going back to the old
* one, if the new one fails init for some reason.
*/
int elevator_switch(struct request_queue *q, struct elevator_type *new_e)
{
struct elevator_queue *old = q->elevator;
bool old_registered = false;
int err;
lockdep_assert_held(&q->sysfs_lock);
if (q->mq_ops) {
blk_mq_freeze_queue(q);
blk_mq_quiesce_queue(q);
err = elevator_switch_mq(q, new_e);
blk_mq_unquiesce_queue(q);
blk_mq_unfreeze_queue(q);
return err;
}
/*
* Turn on BYPASS and drain all requests w/ elevator private data.
* Block layer doesn't call into a quiesced elevator - all requests
* are directly put on the dispatch list without elevator data
* using INSERT_BACK. All requests have SOFTBARRIER set and no
* merge happens either.
*/
if (old) {
old_registered = old->registered;
blk_queue_bypass_start(q);
/* unregister and clear all auxiliary data of the old elevator */
if (old_registered)
elv_unregister_queue(q);
ioc_clear_queue(q);
}
/* allocate, init and register new elevator */
err = new_e->ops.sq.elevator_init_fn(q, new_e);
if (err)
goto fail_init;
err = elv_register_queue(q);
if (err)
goto fail_register;
/* done, kill the old one and finish */
if (old) {
elevator_exit(q, old);
blk_queue_bypass_end(q);
}
blk_add_trace_msg(q, "elv switch: %s", new_e->elevator_name);
return 0;
fail_register:
elevator_exit(q, q->elevator);
fail_init:
/* switch failed, restore and re-register old elevator */
if (old) {
q->elevator = old;
elv_register_queue(q);
blk_queue_bypass_end(q);
}
return err;
}
/*
* Switch this queue to the given IO scheduler.
*/
static int __elevator_change(struct request_queue *q, const char *name)
{
char elevator_name[ELV_NAME_MAX];
struct elevator_type *e;
/* Make sure queue is not in the middle of being removed */
if (!blk_queue_registered(q))
return -ENOENT;
/*
* Special case for mq, turn off scheduling
*/
if (q->mq_ops && !strncmp(name, "none", 4))
return elevator_switch(q, NULL);
strlcpy(elevator_name, name, sizeof(elevator_name));
e = elevator_get(q, strstrip(elevator_name), true);
if (!e)
return -EINVAL;
if (q->elevator && elevator_match(q->elevator->type, elevator_name)) {
elevator_put(e);
return 0;
}
return elevator_switch(q, e);
}
static inline bool elv_support_iosched(struct request_queue *q)
{
if (q->mq_ops && q->tag_set && (q->tag_set->flags &
BLK_MQ_F_NO_SCHED))
return false;
return true;
}
ssize_t elv_iosched_store(struct request_queue *q, const char *name,
size_t count)
{
int ret;
if (!(q->mq_ops || q->request_fn) || !elv_support_iosched(q))
return count;
ret = __elevator_change(q, name);
if (!ret)
return count;
return ret;
}
ssize_t elv_iosched_show(struct request_queue *q, char *name)
{
struct elevator_queue *e = q->elevator;
struct elevator_type *elv = NULL;
struct elevator_type *__e;
bool uses_mq = q->mq_ops != NULL;
int len = 0;
if (!queue_is_rq_based(q))
return sprintf(name, "none\n");
if (!q->elevator)
len += sprintf(name+len, "[none] ");
else
elv = e->type;
spin_lock(&elv_list_lock);
list_for_each_entry(__e, &elv_list, list) {
if (elv && elevator_match(elv, __e->elevator_name) &&
(__e->uses_mq == uses_mq)) {
len += sprintf(name+len, "[%s] ", elv->elevator_name);
continue;
}
if (__e->uses_mq && q->mq_ops && elv_support_iosched(q))
len += sprintf(name+len, "%s ", __e->elevator_name);
else if (!__e->uses_mq && !q->mq_ops)
len += sprintf(name+len, "%s ", __e->elevator_name);
}
spin_unlock(&elv_list_lock);
if (q->mq_ops && q->elevator)
len += sprintf(name+len, "none");
len += sprintf(len+name, "\n");
return len;
}
struct request *elv_rb_former_request(struct request_queue *q,
struct request *rq)
{
struct rb_node *rbprev = rb_prev(&rq->rb_node);
if (rbprev)
return rb_entry_rq(rbprev);
return NULL;
}
EXPORT_SYMBOL(elv_rb_former_request);
struct request *elv_rb_latter_request(struct request_queue *q,
struct request *rq)
{
struct rb_node *rbnext = rb_next(&rq->rb_node);
if (rbnext)
return rb_entry_rq(rbnext);
return NULL;
}
EXPORT_SYMBOL(elv_rb_latter_request);