resctrl.rst 72 KB

123456789101112131415161718192021222324252627282930313233343536373839404142434445464748495051525354555657585960616263646566676869707172737475767778798081828384858687888990919293949596979899100101102103104105106107108109110111112113114115116117118119120121122123124125126127128129130131132133134135136137138139140141142143144145146147148149150151152153154155156157158159160161162163164165166167168169170171172173174175176177178179180181182183184185186187188189190191192193194195196197198199200201202203204205206207208209210211212213214215216217218219220221222223224225226227228229230231232233234235236237238239240241242243244245246247248249250251252253254255256257258259260261262263264265266267268269270271272273274275276277278279280281282283284285286287288289290291292293294295296297298299300301302303304305306307308309310311312313314315316317318319320321322323324325326327328329330331332333334335336337338339340341342343344345346347348349350351352353354355356357358359360361362363364365366367368369370371372373374375376377378379380381382383384385386387388389390391392393394395396397398399400401402403404405406407408409410411412413414415416417418419420421422423424425426427428429430431432433434435436437438439440441442443444445446447448449450451452453454455456457458459460461462463464465466467468469470471472473474475476477478479480481482483484485486487488489490491492493494495496497498499500501502503504505506507508509510511512513514515516517518519520521522523524525526527528529530531532533534535536537538539540541542543544545546547548549550551552553554555556557558559560561562563564565566567568569570571572573574575576577578579580581582583584585586587588589590591592593594595596597598599600601602603604605606607608609610611612613614615616617618619620621622623624625626627628629630631632633634635636637638639640641642643644645646647648649650651652653654655656657658659660661662663664665666667668669670671672673674675676677678679680681682683684685686687688689690691692693694695696697698699700701702703704705706707708709710711712713714715716717718719720721722723724725726727728729730731732733734735736737738739740741742743744745746747748749750751752753754755756757758759760761762763764765766767768769770771772773774775776777778779780781782783784785786787788789790791792793794795796797798799800801802803804805806807808809810811812813814815816817818819820821822823824825826827828829830831832833834835836837838839840841842843844845846847848849850851852853854855856857858859860861862863864865866867868869870871872873874875876877878879880881882883884885886887888889890891892893894895896897898899900901902903904905906907908909910911912913914915916917918919920921922923924925926927928929930931932933934935936937938939940941942943944945946947948949950951952953954955956957958959960961962963964965966967968969970971972973974975976977978979980981982983984985986987988989990991992993994995996997998999100010011002100310041005100610071008100910101011101210131014101510161017101810191020102110221023102410251026102710281029103010311032103310341035103610371038103910401041104210431044104510461047104810491050105110521053105410551056105710581059106010611062106310641065106610671068106910701071107210731074107510761077107810791080108110821083108410851086108710881089109010911092109310941095109610971098109911001101110211031104110511061107110811091110111111121113111411151116111711181119112011211122112311241125112611271128112911301131113211331134113511361137113811391140114111421143114411451146114711481149115011511152115311541155115611571158115911601161116211631164116511661167116811691170117111721173117411751176117711781179118011811182118311841185118611871188118911901191119211931194119511961197119811991200120112021203120412051206120712081209121012111212121312141215121612171218121912201221122212231224122512261227122812291230123112321233123412351236123712381239124012411242124312441245124612471248124912501251125212531254125512561257125812591260126112621263126412651266126712681269127012711272127312741275127612771278127912801281128212831284128512861287128812891290129112921293129412951296129712981299130013011302130313041305130613071308130913101311131213131314131513161317131813191320132113221323132413251326132713281329133013311332133313341335133613371338133913401341134213431344134513461347134813491350135113521353135413551356135713581359136013611362136313641365136613671368136913701371137213731374137513761377137813791380138113821383138413851386138713881389139013911392139313941395139613971398139914001401140214031404140514061407140814091410141114121413141414151416141714181419142014211422142314241425142614271428142914301431143214331434143514361437143814391440144114421443144414451446144714481449145014511452145314541455145614571458145914601461146214631464146514661467146814691470147114721473147414751476147714781479148014811482148314841485148614871488148914901491149214931494149514961497149814991500150115021503150415051506150715081509151015111512151315141515151615171518151915201521152215231524152515261527152815291530153115321533153415351536153715381539154015411542154315441545154615471548154915501551155215531554155515561557155815591560156115621563156415651566156715681569157015711572157315741575157615771578157915801581158215831584158515861587158815891590159115921593159415951596159715981599160016011602160316041605160616071608160916101611161216131614161516161617161816191620162116221623162416251626162716281629163016311632163316341635163616371638163916401641164216431644164516461647164816491650165116521653165416551656165716581659166016611662166316641665166616671668166916701671167216731674167516761677167816791680168116821683168416851686168716881689169016911692169316941695169616971698169917001701170217031704170517061707170817091710171117121713171417151716171717181719172017211722172317241725172617271728172917301731173217331734173517361737173817391740174117421743174417451746174717481749175017511752175317541755175617571758175917601761176217631764176517661767176817691770177117721773177417751776177717781779178017811782178317841785178617871788178917901791179217931794179517961797179817991800180118021803180418051806180718081809181018111812181318141815181618171818181918201821182218231824182518261827182818291830183118321833183418351836183718381839184018411842184318441845184618471848184918501851185218531854185518561857185818591860186118621863186418651866186718681869187018711872187318741875187618771878187918801881188218831884188518861887188818891890189118921893189418951896189718981899190019011902190319041905190619071908190919101911191219131914191519161917191819191920192119221923192419251926192719281929193019311932193319341935193619371938193919401941194219431944194519461947194819491950195119521953195419551956195719581959196019611962196319641965196619671968196919701971197219731974
  1. .. SPDX-License-Identifier: GPL-2.0
  2. .. include:: <isonum.txt>
  3. =====================================================
  4. User Interface for Resource Control feature (resctrl)
  5. =====================================================
  6. :Copyright: |copy| 2016 Intel Corporation
  7. :Authors: - Fenghua Yu <fenghua.yu@intel.com>
  8. - Tony Luck <tony.luck@intel.com>
  9. - Vikas Shivappa <vikas.shivappa@intel.com>
  10. Intel refers to this feature as Intel Resource Director Technology(Intel(R) RDT).
  11. AMD refers to this feature as AMD Platform Quality of Service(AMD QoS).
  12. This feature is enabled by the CONFIG_X86_CPU_RESCTRL and the x86 /proc/cpuinfo
  13. flag bits:
  14. =============================================================== ================================
  15. RDT (Resource Director Technology) Allocation "rdt_a"
  16. CAT (Cache Allocation Technology) "cat_l3", "cat_l2"
  17. CDP (Code and Data Prioritization) "cdp_l3", "cdp_l2"
  18. CQM (Cache QoS Monitoring) "cqm_llc", "cqm_occup_llc"
  19. MBM (Memory Bandwidth Monitoring) "cqm_mbm_total", "cqm_mbm_local"
  20. MBA (Memory Bandwidth Allocation) "mba"
  21. SMBA (Slow Memory Bandwidth Allocation) ""
  22. BMEC (Bandwidth Monitoring Event Configuration) ""
  23. ABMC (Assignable Bandwidth Monitoring Counters) ""
  24. SDCIAE (Smart Data Cache Injection Allocation Enforcement) ""
  25. =============================================================== ================================
  26. Historically, new features were made visible by default in /proc/cpuinfo. This
  27. resulted in the feature flags becoming hard to parse by humans. Adding a new
  28. flag to /proc/cpuinfo should be avoided if user space can obtain information
  29. about the feature from resctrl's info directory.
  30. To use the feature mount the file system::
  31. # mount -t resctrl resctrl [-o cdp[,cdpl2][,mba_MBps][,debug]] /sys/fs/resctrl
  32. mount options are:
  33. "cdp":
  34. Enable code/data prioritization in L3 cache allocations.
  35. "cdpl2":
  36. Enable code/data prioritization in L2 cache allocations.
  37. "mba_MBps":
  38. Enable the MBA Software Controller(mba_sc) to specify MBA
  39. bandwidth in MiBps
  40. "debug":
  41. Make debug files accessible. Available debug files are annotated with
  42. "Available only with debug option".
  43. L2 and L3 CDP are controlled separately.
  44. RDT features are orthogonal. A particular system may support only
  45. monitoring, only control, or both monitoring and control. Cache
  46. pseudo-locking is a unique way of using cache control to "pin" or
  47. "lock" data in the cache. Details can be found in
  48. "Cache Pseudo-Locking".
  49. The mount succeeds if either of allocation or monitoring is present, but
  50. only those files and directories supported by the system will be created.
  51. For more details on the behavior of the interface during monitoring
  52. and allocation, see the "Resource alloc and monitor groups" section.
  53. Info directory
  54. ==============
  55. The 'info' directory contains information about the enabled
  56. resources. Each resource has its own subdirectory. The subdirectory
  57. names reflect the resource names.
  58. Most of the files in the resource's subdirectory are read-only, and
  59. describe properties of the resource. Resources that support global
  60. configuration options also include writable files that can be used
  61. to modify those settings.
  62. Each subdirectory contains the following files with respect to
  63. allocation:
  64. Cache resource(L3/L2) subdirectory contains the following files
  65. related to allocation:
  66. "num_closids":
  67. The number of CLOSIDs which are valid for this
  68. resource. The kernel uses the smallest number of
  69. CLOSIDs of all enabled resources as limit.
  70. "cbm_mask":
  71. The bitmask which is valid for this resource.
  72. This mask is equivalent to 100%.
  73. "min_cbm_bits":
  74. The minimum number of consecutive bits which
  75. must be set when writing a mask.
  76. "shareable_bits":
  77. Bitmask of shareable resource with other executing entities
  78. (e.g. I/O). Applies to all instances of this resource. User
  79. can use this when setting up exclusive cache partitions.
  80. Note that some platforms support devices that have their
  81. own settings for cache use which can over-ride these bits.
  82. When "io_alloc" is enabled, a portion of each cache instance can
  83. be configured for shared use between hardware and software.
  84. "bit_usage" should be used to see which portions of each cache
  85. instance is configured for hardware use via "io_alloc" feature
  86. because every cache instance can have its "io_alloc" bitmask
  87. configured independently via "io_alloc_cbm".
  88. "bit_usage":
  89. Annotated capacity bitmasks showing how all
  90. instances of the resource are used. The legend is:
  91. "0":
  92. Corresponding region is unused. When the system's
  93. resources have been allocated and a "0" is found
  94. in "bit_usage" it is a sign that resources are
  95. wasted.
  96. "H":
  97. Corresponding region is used by hardware only
  98. but available for software use. If a resource
  99. has bits set in "shareable_bits" or "io_alloc_cbm"
  100. but not all of these bits appear in the resource
  101. groups' schemata then the bits appearing in
  102. "shareable_bits" or "io_alloc_cbm" but no
  103. resource group will be marked as "H".
  104. "X":
  105. Corresponding region is available for sharing and
  106. used by hardware and software. These are the bits
  107. that appear in "shareable_bits" or "io_alloc_cbm"
  108. as well as a resource group's allocation.
  109. "S":
  110. Corresponding region is used by software
  111. and available for sharing.
  112. "E":
  113. Corresponding region is used exclusively by
  114. one resource group. No sharing allowed.
  115. "P":
  116. Corresponding region is pseudo-locked. No
  117. sharing allowed.
  118. "sparse_masks":
  119. Indicates if non-contiguous 1s value in CBM is supported.
  120. "0":
  121. Only contiguous 1s value in CBM is supported.
  122. "1":
  123. Non-contiguous 1s value in CBM is supported.
  124. "io_alloc":
  125. "io_alloc" enables system software to configure the portion of
  126. the cache allocated for I/O traffic. File may only exist if the
  127. system supports this feature on some of its cache resources.
  128. "disabled":
  129. Resource supports "io_alloc" but the feature is disabled.
  130. Portions of cache used for allocation of I/O traffic cannot
  131. be configured.
  132. "enabled":
  133. Portions of cache used for allocation of I/O traffic
  134. can be configured using "io_alloc_cbm".
  135. "not supported":
  136. Support not available for this resource.
  137. The feature can be modified by writing to the interface, for example:
  138. To enable::
  139. # echo 1 > /sys/fs/resctrl/info/L3/io_alloc
  140. To disable::
  141. # echo 0 > /sys/fs/resctrl/info/L3/io_alloc
  142. The underlying implementation may reduce resources available to
  143. general (CPU) cache allocation. See architecture specific notes
  144. below. Depending on usage requirements the feature can be enabled
  145. or disabled.
  146. On AMD systems, io_alloc feature is supported by the L3 Smart
  147. Data Cache Injection Allocation Enforcement (SDCIAE). The CLOSID for
  148. io_alloc is the highest CLOSID supported by the resource. When
  149. io_alloc is enabled, the highest CLOSID is dedicated to io_alloc and
  150. no longer available for general (CPU) cache allocation. When CDP is
  151. enabled, io_alloc routes I/O traffic using the highest CLOSID allocated
  152. for the instruction cache (CDP_CODE), making this CLOSID no longer
  153. available for general (CPU) cache allocation for both the CDP_CODE
  154. and CDP_DATA resources.
  155. "io_alloc_cbm":
  156. Capacity bitmasks that describe the portions of cache instances to
  157. which I/O traffic from supported I/O devices are routed when "io_alloc"
  158. is enabled.
  159. CBMs are displayed in the following format:
  160. <cache_id0>=<cbm>;<cache_id1>=<cbm>;...
  161. Example::
  162. # cat /sys/fs/resctrl/info/L3/io_alloc_cbm
  163. 0=ffff;1=ffff
  164. CBMs can be configured by writing to the interface.
  165. Example::
  166. # echo 1=ff > /sys/fs/resctrl/info/L3/io_alloc_cbm
  167. # cat /sys/fs/resctrl/info/L3/io_alloc_cbm
  168. 0=ffff;1=00ff
  169. # echo "0=ff;1=f" > /sys/fs/resctrl/info/L3/io_alloc_cbm
  170. # cat /sys/fs/resctrl/info/L3/io_alloc_cbm
  171. 0=00ff;1=000f
  172. When CDP is enabled "io_alloc_cbm" associated with the CDP_DATA and CDP_CODE
  173. resources may reflect the same values. For example, values read from and
  174. written to /sys/fs/resctrl/info/L3DATA/io_alloc_cbm may be reflected by
  175. /sys/fs/resctrl/info/L3CODE/io_alloc_cbm and vice versa.
  176. Memory bandwidth(MB) subdirectory contains the following files
  177. with respect to allocation:
  178. "min_bandwidth":
  179. The minimum memory bandwidth percentage which
  180. user can request.
  181. "bandwidth_gran":
  182. The granularity in which the memory bandwidth
  183. percentage is allocated. The allocated
  184. b/w percentage is rounded off to the next
  185. control step available on the hardware. The
  186. available bandwidth control steps are:
  187. min_bandwidth + N * bandwidth_gran.
  188. "delay_linear":
  189. Indicates if the delay scale is linear or
  190. non-linear. This field is purely informational
  191. only.
  192. "thread_throttle_mode":
  193. Indicator on Intel systems of how tasks running on threads
  194. of a physical core are throttled in cases where they
  195. request different memory bandwidth percentages:
  196. "max":
  197. the smallest percentage is applied
  198. to all threads
  199. "per-thread":
  200. bandwidth percentages are directly applied to
  201. the threads running on the core
  202. If L3 monitoring is available there will be an "L3_MON" directory
  203. with the following files:
  204. "num_rmids":
  205. The number of RMIDs supported by hardware for
  206. L3 monitoring events.
  207. "mon_features":
  208. Lists the monitoring events if
  209. monitoring is enabled for the resource.
  210. Example::
  211. # cat /sys/fs/resctrl/info/L3_MON/mon_features
  212. llc_occupancy
  213. mbm_total_bytes
  214. mbm_local_bytes
  215. If the system supports Bandwidth Monitoring Event
  216. Configuration (BMEC), then the bandwidth events will
  217. be configurable. The output will be::
  218. # cat /sys/fs/resctrl/info/L3_MON/mon_features
  219. llc_occupancy
  220. mbm_total_bytes
  221. mbm_total_bytes_config
  222. mbm_local_bytes
  223. mbm_local_bytes_config
  224. "mbm_total_bytes_config", "mbm_local_bytes_config":
  225. Read/write files containing the configuration for the mbm_total_bytes
  226. and mbm_local_bytes events, respectively, when the Bandwidth
  227. Monitoring Event Configuration (BMEC) feature is supported.
  228. The event configuration settings are domain specific and affect
  229. all the CPUs in the domain. When either event configuration is
  230. changed, the bandwidth counters for all RMIDs of both events
  231. (mbm_total_bytes as well as mbm_local_bytes) are cleared for that
  232. domain. The next read for every RMID will report "Unavailable"
  233. and subsequent reads will report the valid value.
  234. Following are the types of events supported:
  235. ==== ========================================================
  236. Bits Description
  237. ==== ========================================================
  238. 6 Dirty Victims from the QOS domain to all types of memory
  239. 5 Reads to slow memory in the non-local NUMA domain
  240. 4 Reads to slow memory in the local NUMA domain
  241. 3 Non-temporal writes to non-local NUMA domain
  242. 2 Non-temporal writes to local NUMA domain
  243. 1 Reads to memory in the non-local NUMA domain
  244. 0 Reads to memory in the local NUMA domain
  245. ==== ========================================================
  246. By default, the mbm_total_bytes configuration is set to 0x7f to count
  247. all the event types and the mbm_local_bytes configuration is set to
  248. 0x15 to count all the local memory events.
  249. Examples:
  250. * To view the current configuration::
  251. ::
  252. # cat /sys/fs/resctrl/info/L3_MON/mbm_total_bytes_config
  253. 0=0x7f;1=0x7f;2=0x7f;3=0x7f
  254. # cat /sys/fs/resctrl/info/L3_MON/mbm_local_bytes_config
  255. 0=0x15;1=0x15;3=0x15;4=0x15
  256. * To change the mbm_total_bytes to count only reads on domain 0,
  257. the bits 0, 1, 4 and 5 needs to be set, which is 110011b in binary
  258. (in hexadecimal 0x33):
  259. ::
  260. # echo "0=0x33" > /sys/fs/resctrl/info/L3_MON/mbm_total_bytes_config
  261. # cat /sys/fs/resctrl/info/L3_MON/mbm_total_bytes_config
  262. 0=0x33;1=0x7f;2=0x7f;3=0x7f
  263. * To change the mbm_local_bytes to count all the slow memory reads on
  264. domain 0 and 1, the bits 4 and 5 needs to be set, which is 110000b
  265. in binary (in hexadecimal 0x30):
  266. ::
  267. # echo "0=0x30;1=0x30" > /sys/fs/resctrl/info/L3_MON/mbm_local_bytes_config
  268. # cat /sys/fs/resctrl/info/L3_MON/mbm_local_bytes_config
  269. 0=0x30;1=0x30;3=0x15;4=0x15
  270. "mbm_assign_mode":
  271. The supported counter assignment modes. The enclosed brackets indicate which mode
  272. is enabled. The MBM events associated with counters may reset when "mbm_assign_mode"
  273. is changed.
  274. ::
  275. # cat /sys/fs/resctrl/info/L3_MON/mbm_assign_mode
  276. [mbm_event]
  277. default
  278. "mbm_event":
  279. mbm_event mode allows users to assign a hardware counter to an RMID, event
  280. pair and monitor the bandwidth usage as long as it is assigned. The hardware
  281. continues to track the assigned counter until it is explicitly unassigned by
  282. the user. Each event within a resctrl group can be assigned independently.
  283. In this mode, a monitoring event can only accumulate data while it is backed
  284. by a hardware counter. Use "mbm_L3_assignments" found in each CTRL_MON and MON
  285. group to specify which of the events should have a counter assigned. The number
  286. of counters available is described in the "num_mbm_cntrs" file. Changing the
  287. mode may cause all counters on the resource to reset.
  288. Moving to mbm_event counter assignment mode requires users to assign the counters
  289. to the events. Otherwise, the MBM event counters will return 'Unassigned' when read.
  290. The mode is beneficial for AMD platforms that support more CTRL_MON
  291. and MON groups than available hardware counters. By default, this
  292. feature is enabled on AMD platforms with the ABMC (Assignable Bandwidth
  293. Monitoring Counters) capability, ensuring counters remain assigned even
  294. when the corresponding RMID is not actively used by any processor.
  295. "default":
  296. In default mode, resctrl assumes there is a hardware counter for each
  297. event within every CTRL_MON and MON group. On AMD platforms, it is
  298. recommended to use the mbm_event mode, if supported, to prevent reset of MBM
  299. events between reads resulting from hardware re-allocating counters. This can
  300. result in misleading values or display "Unavailable" if no counter is assigned
  301. to the event.
  302. * To enable "mbm_event" counter assignment mode:
  303. ::
  304. # echo "mbm_event" > /sys/fs/resctrl/info/L3_MON/mbm_assign_mode
  305. * To enable "default" monitoring mode:
  306. ::
  307. # echo "default" > /sys/fs/resctrl/info/L3_MON/mbm_assign_mode
  308. "num_mbm_cntrs":
  309. The maximum number of counters (total of available and assigned counters) in
  310. each domain when the system supports mbm_event mode.
  311. For example, on a system with maximum of 32 memory bandwidth monitoring
  312. counters in each of its L3 domains:
  313. ::
  314. # cat /sys/fs/resctrl/info/L3_MON/num_mbm_cntrs
  315. 0=32;1=32
  316. "available_mbm_cntrs":
  317. The number of counters available for assignment in each domain when mbm_event
  318. mode is enabled on the system.
  319. For example, on a system with 30 available [hardware] assignable counters
  320. in each of its L3 domains:
  321. ::
  322. # cat /sys/fs/resctrl/info/L3_MON/available_mbm_cntrs
  323. 0=30;1=30
  324. "event_configs":
  325. Directory that exists when "mbm_event" counter assignment mode is supported.
  326. Contains a sub-directory for each MBM event that can be assigned to a counter.
  327. Two MBM events are supported by default: mbm_local_bytes and mbm_total_bytes.
  328. Each MBM event's sub-directory contains a file named "event_filter" that is
  329. used to view and modify which memory transactions the MBM event is configured
  330. with. The file is accessible only when "mbm_event" counter assignment mode is
  331. enabled.
  332. List of memory transaction types supported:
  333. ========================== ========================================================
  334. Name Description
  335. ========================== ========================================================
  336. dirty_victim_writes_all Dirty Victims from the QOS domain to all types of memory
  337. remote_reads_slow_memory Reads to slow memory in the non-local NUMA domain
  338. local_reads_slow_memory Reads to slow memory in the local NUMA domain
  339. remote_non_temporal_writes Non-temporal writes to non-local NUMA domain
  340. local_non_temporal_writes Non-temporal writes to local NUMA domain
  341. remote_reads Reads to memory in the non-local NUMA domain
  342. local_reads Reads to memory in the local NUMA domain
  343. ========================== ========================================================
  344. For example::
  345. # cat /sys/fs/resctrl/info/L3_MON/event_configs/mbm_total_bytes/event_filter
  346. local_reads,remote_reads,local_non_temporal_writes,remote_non_temporal_writes,
  347. local_reads_slow_memory,remote_reads_slow_memory,dirty_victim_writes_all
  348. # cat /sys/fs/resctrl/info/L3_MON/event_configs/mbm_local_bytes/event_filter
  349. local_reads,local_non_temporal_writes,local_reads_slow_memory
  350. Modify the event configuration by writing to the "event_filter" file within
  351. the "event_configs" directory. The read/write "event_filter" file contains the
  352. configuration of the event that reflects which memory transactions are counted by it.
  353. For example::
  354. # echo "local_reads, local_non_temporal_writes" >
  355. /sys/fs/resctrl/info/L3_MON/event_configs/mbm_total_bytes/event_filter
  356. # cat /sys/fs/resctrl/info/L3_MON/event_configs/mbm_total_bytes/event_filter
  357. local_reads,local_non_temporal_writes
  358. "mbm_assign_on_mkdir":
  359. Exists when "mbm_event" counter assignment mode is supported. Accessible
  360. only when "mbm_event" counter assignment mode is enabled.
  361. Determines if a counter will automatically be assigned to an RMID, MBM event
  362. pair when its associated monitor group is created via mkdir. Enabled by default
  363. on boot, also when switched from "default" mode to "mbm_event" counter assignment
  364. mode. Users can disable this capability by writing to the interface.
  365. "0":
  366. Auto assignment is disabled.
  367. "1":
  368. Auto assignment is enabled.
  369. Example::
  370. # echo 0 > /sys/fs/resctrl/info/L3_MON/mbm_assign_on_mkdir
  371. # cat /sys/fs/resctrl/info/L3_MON/mbm_assign_on_mkdir
  372. 0
  373. "max_threshold_occupancy":
  374. Read/write file provides the largest value (in
  375. bytes) at which a previously used LLC_occupancy
  376. counter can be considered for reuse.
  377. If telemetry monitoring is available there will be a "PERF_PKG_MON" directory
  378. with the following files:
  379. "num_rmids":
  380. The number of RMIDs for telemetry monitoring events.
  381. On Intel resctrl will not enable telemetry events if the number of
  382. RMIDs that can be tracked concurrently is lower than the total number
  383. of RMIDs supported. Telemetry events can be force-enabled with the
  384. "rdt=" kernel parameter, but this may reduce the number of
  385. monitoring groups that can be created.
  386. "mon_features":
  387. Lists the telemetry monitoring events that are enabled on this system.
  388. The upper bound for how many "CTRL_MON" + "MON" can be created
  389. is the smaller of the L3_MON and PERF_PKG_MON "num_rmids" values.
  390. Finally, in the top level of the "info" directory there is a file
  391. named "last_cmd_status". This is reset with every "command" issued
  392. via the file system (making new directories or writing to any of the
  393. control files). If the command was successful, it will read as "ok".
  394. If the command failed, it will provide more information that can be
  395. conveyed in the error returns from file operations. E.g.
  396. ::
  397. # echo L3:0=f7 > schemata
  398. bash: echo: write error: Invalid argument
  399. # cat info/last_cmd_status
  400. mask f7 has non-consecutive 1-bits
  401. Resource alloc and monitor groups
  402. =================================
  403. Resource groups are represented as directories in the resctrl file
  404. system. The default group is the root directory which, immediately
  405. after mounting, owns all the tasks and cpus in the system and can make
  406. full use of all resources.
  407. On a system with RDT control features additional directories can be
  408. created in the root directory that specify different amounts of each
  409. resource (see "schemata" below). The root and these additional top level
  410. directories are referred to as "CTRL_MON" groups below.
  411. On a system with RDT monitoring the root directory and other top level
  412. directories contain a directory named "mon_groups" in which additional
  413. directories can be created to monitor subsets of tasks in the CTRL_MON
  414. group that is their ancestor. These are called "MON" groups in the rest
  415. of this document.
  416. Removing a directory will move all tasks and cpus owned by the group it
  417. represents to the parent. Removing one of the created CTRL_MON groups
  418. will automatically remove all MON groups below it.
  419. Moving MON group directories to a new parent CTRL_MON group is supported
  420. for the purpose of changing the resource allocations of a MON group
  421. without impacting its monitoring data or assigned tasks. This operation
  422. is not allowed for MON groups which monitor CPUs. No other move
  423. operation is currently allowed other than simply renaming a CTRL_MON or
  424. MON group.
  425. All groups contain the following files:
  426. "tasks":
  427. Reading this file shows the list of all tasks that belong to
  428. this group. Writing a task id to the file will add a task to the
  429. group. Multiple tasks can be added by separating the task ids
  430. with commas. Tasks will be assigned sequentially. Multiple
  431. failures are not supported. A single failure encountered while
  432. attempting to assign a task will cause the operation to abort and
  433. already added tasks before the failure will remain in the group.
  434. Failures will be logged to /sys/fs/resctrl/info/last_cmd_status.
  435. If the group is a CTRL_MON group the task is removed from
  436. whichever previous CTRL_MON group owned the task and also from
  437. any MON group that owned the task. If the group is a MON group,
  438. then the task must already belong to the CTRL_MON parent of this
  439. group. The task is removed from any previous MON group.
  440. "cpus":
  441. Reading this file shows a bitmask of the logical CPUs owned by
  442. this group. Writing a mask to this file will add and remove
  443. CPUs to/from this group. As with the tasks file a hierarchy is
  444. maintained where MON groups may only include CPUs owned by the
  445. parent CTRL_MON group.
  446. When the resource group is in pseudo-locked mode this file will
  447. only be readable, reflecting the CPUs associated with the
  448. pseudo-locked region.
  449. "cpus_list":
  450. Just like "cpus", only using ranges of CPUs instead of bitmasks.
  451. When control is enabled all CTRL_MON groups will also contain:
  452. "schemata":
  453. A list of all the resources available to this group.
  454. Each resource has its own line and format - see below for details.
  455. "size":
  456. Mirrors the display of the "schemata" file to display the size in
  457. bytes of each allocation instead of the bits representing the
  458. allocation.
  459. "mode":
  460. The "mode" of the resource group dictates the sharing of its
  461. allocations. A "shareable" resource group allows sharing of its
  462. allocations while an "exclusive" resource group does not. A
  463. cache pseudo-locked region is created by first writing
  464. "pseudo-locksetup" to the "mode" file before writing the cache
  465. pseudo-locked region's schemata to the resource group's "schemata"
  466. file. On successful pseudo-locked region creation the mode will
  467. automatically change to "pseudo-locked".
  468. "ctrl_hw_id":
  469. Available only with debug option. The identifier used by hardware
  470. for the control group. On x86 this is the CLOSID.
  471. When monitoring is enabled all MON groups will also contain:
  472. "mon_data":
  473. This contains directories for each monitor domain.
  474. If L3 monitoring is enabled, there will be a "mon_L3_XX" directory for
  475. each instance of an L3 cache. Each directory contains files for the enabled
  476. L3 events (e.g. "llc_occupancy", "mbm_total_bytes", and "mbm_local_bytes").
  477. If telemetry monitoring is enabled, there will be a "mon_PERF_PKG_YY"
  478. directory for each physical processor package. Each directory contains
  479. files for the enabled telemetry events (e.g. "core_energy". "activity",
  480. "uops_retired", etc.)
  481. The info/`*`/mon_features files provide the full list of enabled
  482. event/file names.
  483. "core energy" reports a floating point number for the energy (in Joules)
  484. consumed by cores (registers, arithmetic units, TLB and L1/L2 caches)
  485. during execution of instructions summed across all logical CPUs on a
  486. package for the current monitoring group.
  487. "activity" also reports a floating point value (in Farads). This provides
  488. an estimate of work done independent of the frequency that the CPUs used
  489. for execution.
  490. Note that "core energy" and "activity" only measure energy/activity in the
  491. "core" of the CPU (arithmetic units, TLB, L1 and L2 caches, etc.). They
  492. do not include L3 cache, memory, I/O devices etc.
  493. All other events report decimal integer values.
  494. In a MON group these files provide a read out of the current value of
  495. the event for all tasks in the group. In CTRL_MON groups these files
  496. provide the sum for all tasks in the CTRL_MON group and all tasks in
  497. MON groups. Please see example section for more details on usage.
  498. On systems with Sub-NUMA Cluster (SNC) enabled there are extra
  499. directories for each node (located within the "mon_L3_XX" directory
  500. for the L3 cache they occupy). These are named "mon_sub_L3_YY"
  501. where "YY" is the node number.
  502. When the 'mbm_event' counter assignment mode is enabled, reading
  503. an MBM event of a MON group returns 'Unassigned' if no hardware
  504. counter is assigned to it. For CTRL_MON groups, 'Unassigned' is
  505. returned if the MBM event does not have an assigned counter in the
  506. CTRL_MON group nor in any of its associated MON groups.
  507. "mon_hw_id":
  508. Available only with debug option. The identifier used by hardware
  509. for the monitor group. On x86 this is the RMID.
  510. When monitoring is enabled all MON groups may also contain:
  511. "mbm_L3_assignments":
  512. Exists when "mbm_event" counter assignment mode is supported and lists the
  513. counter assignment states of the group.
  514. The assignment list is displayed in the following format:
  515. <Event>:<Domain ID>=<Assignment state>;<Domain ID>=<Assignment state>
  516. Event: A valid MBM event in the
  517. /sys/fs/resctrl/info/L3_MON/event_configs directory.
  518. Domain ID: A valid domain ID. When writing, '*' applies the changes
  519. to all the domains.
  520. Assignment states:
  521. _ : No counter assigned.
  522. e : Counter assigned exclusively.
  523. Example:
  524. To display the counter assignment states for the default group.
  525. ::
  526. # cd /sys/fs/resctrl
  527. # cat /sys/fs/resctrl/mbm_L3_assignments
  528. mbm_total_bytes:0=e;1=e
  529. mbm_local_bytes:0=e;1=e
  530. Assignments can be modified by writing to the interface.
  531. Examples:
  532. To unassign the counter associated with the mbm_total_bytes event on domain 0:
  533. ::
  534. # echo "mbm_total_bytes:0=_" > /sys/fs/resctrl/mbm_L3_assignments
  535. # cat /sys/fs/resctrl/mbm_L3_assignments
  536. mbm_total_bytes:0=_;1=e
  537. mbm_local_bytes:0=e;1=e
  538. To unassign the counter associated with the mbm_total_bytes event on all the domains:
  539. ::
  540. # echo "mbm_total_bytes:*=_" > /sys/fs/resctrl/mbm_L3_assignments
  541. # cat /sys/fs/resctrl/mbm_L3_assignments
  542. mbm_total_bytes:0=_;1=_
  543. mbm_local_bytes:0=e;1=e
  544. To assign a counter associated with the mbm_total_bytes event on all domains in
  545. exclusive mode:
  546. ::
  547. # echo "mbm_total_bytes:*=e" > /sys/fs/resctrl/mbm_L3_assignments
  548. # cat /sys/fs/resctrl/mbm_L3_assignments
  549. mbm_total_bytes:0=e;1=e
  550. mbm_local_bytes:0=e;1=e
  551. When the "mba_MBps" mount option is used all CTRL_MON groups will also contain:
  552. "mba_MBps_event":
  553. Reading this file shows which memory bandwidth event is used
  554. as input to the software feedback loop that keeps memory bandwidth
  555. below the value specified in the schemata file. Writing the
  556. name of one of the supported memory bandwidth events found in
  557. /sys/fs/resctrl/info/L3_MON/mon_features changes the input
  558. event.
  559. Resource allocation rules
  560. -------------------------
  561. When a task is running the following rules define which resources are
  562. available to it:
  563. 1) If the task is a member of a non-default group, then the schemata
  564. for that group is used.
  565. 2) Else if the task belongs to the default group, but is running on a
  566. CPU that is assigned to some specific group, then the schemata for the
  567. CPU's group is used.
  568. 3) Otherwise the schemata for the default group is used.
  569. Resource monitoring rules
  570. -------------------------
  571. 1) If a task is a member of a MON group, or non-default CTRL_MON group
  572. then RDT events for the task will be reported in that group.
  573. 2) If a task is a member of the default CTRL_MON group, but is running
  574. on a CPU that is assigned to some specific group, then the RDT events
  575. for the task will be reported in that group.
  576. 3) Otherwise RDT events for the task will be reported in the root level
  577. "mon_data" group.
  578. Notes on cache occupancy monitoring and control
  579. ===============================================
  580. When moving a task from one group to another you should remember that
  581. this only affects *new* cache allocations by the task. E.g. you may have
  582. a task in a monitor group showing 3 MB of cache occupancy. If you move
  583. to a new group and immediately check the occupancy of the old and new
  584. groups you will likely see that the old group is still showing 3 MB and
  585. the new group zero. When the task accesses locations still in cache from
  586. before the move, the h/w does not update any counters. On a busy system
  587. you will likely see the occupancy in the old group go down as cache lines
  588. are evicted and re-used while the occupancy in the new group rises as
  589. the task accesses memory and loads into the cache are counted based on
  590. membership in the new group.
  591. The same applies to cache allocation control. Moving a task to a group
  592. with a smaller cache partition will not evict any cache lines. The
  593. process may continue to use them from the old partition.
  594. Hardware uses CLOSid(Class of service ID) and an RMID(Resource monitoring ID)
  595. to identify a control group and a monitoring group respectively. Each of
  596. the resource groups are mapped to these IDs based on the kind of group. The
  597. number of CLOSid and RMID are limited by the hardware and hence the creation of
  598. a "CTRL_MON" directory may fail if we run out of either CLOSID or RMID
  599. and creation of "MON" group may fail if we run out of RMIDs.
  600. max_threshold_occupancy - generic concepts
  601. ------------------------------------------
  602. Note that an RMID once freed may not be immediately available for use as
  603. the RMID is still tagged the cache lines of the previous user of RMID.
  604. Hence such RMIDs are placed on limbo list and checked back if the cache
  605. occupancy has gone down. If there is a time when system has a lot of
  606. limbo RMIDs but which are not ready to be used, user may see an -EBUSY
  607. during mkdir.
  608. max_threshold_occupancy is a user configurable value to determine the
  609. occupancy at which an RMID can be freed.
  610. The mon_llc_occupancy_limbo tracepoint gives the precise occupancy in bytes
  611. for a subset of RMID that are not immediately available for allocation.
  612. This can't be relied on to produce output every second, it may be necessary
  613. to attempt to create an empty monitor group to force an update. Output may
  614. only be produced if creation of a control or monitor group fails.
  615. Schemata files - general concepts
  616. ---------------------------------
  617. Each line in the file describes one resource. The line starts with
  618. the name of the resource, followed by specific values to be applied
  619. in each of the instances of that resource on the system.
  620. Cache IDs
  621. ---------
  622. On current generation systems there is one L3 cache per socket and L2
  623. caches are generally just shared by the hyperthreads on a core, but this
  624. isn't an architectural requirement. We could have multiple separate L3
  625. caches on a socket, multiple cores could share an L2 cache. So instead
  626. of using "socket" or "core" to define the set of logical cpus sharing
  627. a resource we use a "Cache ID". At a given cache level this will be a
  628. unique number across the whole system (but it isn't guaranteed to be a
  629. contiguous sequence, there may be gaps). To find the ID for each logical
  630. CPU look in /sys/devices/system/cpu/cpu*/cache/index*/id
  631. Cache Bit Masks (CBM)
  632. ---------------------
  633. For cache resources we describe the portion of the cache that is available
  634. for allocation using a bitmask. The maximum value of the mask is defined
  635. by each cpu model (and may be different for different cache levels). It
  636. is found using CPUID, but is also provided in the "info" directory of
  637. the resctrl file system in "info/{resource}/cbm_mask". Some Intel hardware
  638. requires that these masks have all the '1' bits in a contiguous block. So
  639. 0x3, 0x6 and 0xC are legal 4-bit masks with two bits set, but 0x5, 0x9
  640. and 0xA are not. Check /sys/fs/resctrl/info/{resource}/sparse_masks
  641. if non-contiguous 1s value is supported. On a system with a 20-bit mask
  642. each bit represents 5% of the capacity of the cache. You could partition
  643. the cache into four equal parts with masks: 0x1f, 0x3e0, 0x7c00, 0xf8000.
  644. Notes on Sub-NUMA Cluster mode
  645. ==============================
  646. When SNC mode is enabled, Linux may load balance tasks between Sub-NUMA
  647. nodes much more readily than between regular NUMA nodes since the CPUs
  648. on Sub-NUMA nodes share the same L3 cache and the system may report
  649. the NUMA distance between Sub-NUMA nodes with a lower value than used
  650. for regular NUMA nodes.
  651. The top-level monitoring files in each "mon_L3_XX" directory provide
  652. the sum of data across all SNC nodes sharing an L3 cache instance.
  653. Users who bind tasks to the CPUs of a specific Sub-NUMA node can read
  654. the "llc_occupancy", "mbm_total_bytes", and "mbm_local_bytes" in the
  655. "mon_sub_L3_YY" directories to get node local data.
  656. Memory bandwidth allocation is still performed at the L3 cache
  657. level. I.e. throttling controls are applied to all SNC nodes.
  658. L3 cache allocation bitmaps also apply to all SNC nodes. But note that
  659. the amount of L3 cache represented by each bit is divided by the number
  660. of SNC nodes per L3 cache. E.g. with a 100MB cache on a system with 10-bit
  661. allocation masks each bit normally represents 10MB. With SNC mode enabled
  662. with two SNC nodes per L3 cache, each bit only represents 5MB.
  663. Memory bandwidth Allocation and monitoring
  664. ==========================================
  665. For Memory bandwidth resource, by default the user controls the resource
  666. by indicating the percentage of total memory bandwidth.
  667. The minimum bandwidth percentage value for each cpu model is predefined
  668. and can be looked up through "info/MB/min_bandwidth". The bandwidth
  669. granularity that is allocated is also dependent on the cpu model and can
  670. be looked up at "info/MB/bandwidth_gran". The available bandwidth
  671. control steps are: min_bw + N * bw_gran. Intermediate values are rounded
  672. to the next control step available on the hardware.
  673. The bandwidth throttling is a core specific mechanism on some of Intel
  674. SKUs. Using a high bandwidth and a low bandwidth setting on two threads
  675. sharing a core may result in both threads being throttled to use the
  676. low bandwidth (see "thread_throttle_mode").
  677. The fact that Memory bandwidth allocation(MBA) may be a core
  678. specific mechanism where as memory bandwidth monitoring(MBM) is done at
  679. the package level may lead to confusion when users try to apply control
  680. via the MBA and then monitor the bandwidth to see if the controls are
  681. effective. Below are such scenarios:
  682. 1. User may *not* see increase in actual bandwidth when percentage
  683. values are increased:
  684. This can occur when aggregate L2 external bandwidth is more than L3
  685. external bandwidth. Consider an SKL SKU with 24 cores on a package and
  686. where L2 external is 10GBps (hence aggregate L2 external bandwidth is
  687. 240GBps) and L3 external bandwidth is 100GBps. Now a workload with '20
  688. threads, having 50% bandwidth, each consuming 5GBps' consumes the max L3
  689. bandwidth of 100GBps although the percentage value specified is only 50%
  690. << 100%. Hence increasing the bandwidth percentage will not yield any
  691. more bandwidth. This is because although the L2 external bandwidth still
  692. has capacity, the L3 external bandwidth is fully used. Also note that
  693. this would be dependent on number of cores the benchmark is run on.
  694. 2. Same bandwidth percentage may mean different actual bandwidth
  695. depending on # of threads:
  696. For the same SKU in #1, a 'single thread, with 10% bandwidth' and '4
  697. thread, with 10% bandwidth' can consume up to 10GBps and 40GBps although
  698. they have same percentage bandwidth of 10%. This is simply because as
  699. threads start using more cores in an rdtgroup, the actual bandwidth may
  700. increase or vary although user specified bandwidth percentage is same.
  701. In order to mitigate this and make the interface more user friendly,
  702. resctrl added support for specifying the bandwidth in MiBps as well. The
  703. kernel underneath would use a software feedback mechanism or a "Software
  704. Controller(mba_sc)" which reads the actual bandwidth using MBM counters
  705. and adjust the memory bandwidth percentages to ensure::
  706. "actual bandwidth < user specified bandwidth".
  707. By default, the schemata would take the bandwidth percentage values
  708. where as user can switch to the "MBA software controller" mode using
  709. a mount option 'mba_MBps'. The schemata format is specified in the below
  710. sections.
  711. L3 schemata file details (code and data prioritization disabled)
  712. ----------------------------------------------------------------
  713. With CDP disabled the L3 schemata format is::
  714. L3:<cache_id0>=<cbm>;<cache_id1>=<cbm>;...
  715. L3 schemata file details (CDP enabled via mount option to resctrl)
  716. ------------------------------------------------------------------
  717. When CDP is enabled L3 control is split into two separate resources
  718. so you can specify independent masks for code and data like this::
  719. L3DATA:<cache_id0>=<cbm>;<cache_id1>=<cbm>;...
  720. L3CODE:<cache_id0>=<cbm>;<cache_id1>=<cbm>;...
  721. L2 schemata file details
  722. ------------------------
  723. CDP is supported at L2 using the 'cdpl2' mount option. The schemata
  724. format is either::
  725. L2:<cache_id0>=<cbm>;<cache_id1>=<cbm>;...
  726. or
  727. L2DATA:<cache_id0>=<cbm>;<cache_id1>=<cbm>;...
  728. L2CODE:<cache_id0>=<cbm>;<cache_id1>=<cbm>;...
  729. Memory bandwidth Allocation (default mode)
  730. ------------------------------------------
  731. Memory b/w domain is L3 cache.
  732. ::
  733. MB:<cache_id0>=bandwidth0;<cache_id1>=bandwidth1;...
  734. Memory bandwidth Allocation specified in MiBps
  735. ----------------------------------------------
  736. Memory bandwidth domain is L3 cache.
  737. ::
  738. MB:<cache_id0>=bw_MiBps0;<cache_id1>=bw_MiBps1;...
  739. Slow Memory Bandwidth Allocation (SMBA)
  740. ---------------------------------------
  741. AMD hardware supports Slow Memory Bandwidth Allocation (SMBA).
  742. CXL.memory is the only supported "slow" memory device. With the
  743. support of SMBA, the hardware enables bandwidth allocation on
  744. the slow memory devices. If there are multiple such devices in
  745. the system, the throttling logic groups all the slow sources
  746. together and applies the limit on them as a whole.
  747. The presence of SMBA (with CXL.memory) is independent of slow memory
  748. devices presence. If there are no such devices on the system, then
  749. configuring SMBA will have no impact on the performance of the system.
  750. The bandwidth domain for slow memory is L3 cache. Its schemata file
  751. is formatted as:
  752. ::
  753. SMBA:<cache_id0>=bandwidth0;<cache_id1>=bandwidth1;...
  754. Reading/writing the schemata file
  755. ---------------------------------
  756. Reading the schemata file will show the state of all resources
  757. on all domains. When writing you only need to specify those values
  758. which you wish to change. E.g.
  759. ::
  760. # cat schemata
  761. L3DATA:0=fffff;1=fffff;2=fffff;3=fffff
  762. L3CODE:0=fffff;1=fffff;2=fffff;3=fffff
  763. # echo "L3DATA:2=3c0;" > schemata
  764. # cat schemata
  765. L3DATA:0=fffff;1=fffff;2=3c0;3=fffff
  766. L3CODE:0=fffff;1=fffff;2=fffff;3=fffff
  767. Reading/writing the schemata file (on AMD systems)
  768. --------------------------------------------------
  769. Reading the schemata file will show the current bandwidth limit on all
  770. domains. The allocated resources are in multiples of one eighth GB/s.
  771. When writing to the file, you need to specify what cache id you wish to
  772. configure the bandwidth limit.
  773. For example, to allocate 2GB/s limit on the first cache id:
  774. ::
  775. # cat schemata
  776. MB:0=2048;1=2048;2=2048;3=2048
  777. L3:0=ffff;1=ffff;2=ffff;3=ffff
  778. # echo "MB:1=16" > schemata
  779. # cat schemata
  780. MB:0=2048;1= 16;2=2048;3=2048
  781. L3:0=ffff;1=ffff;2=ffff;3=ffff
  782. Reading/writing the schemata file (on AMD systems) with SMBA feature
  783. --------------------------------------------------------------------
  784. Reading and writing the schemata file is the same as without SMBA in
  785. above section.
  786. For example, to allocate 8GB/s limit on the first cache id:
  787. ::
  788. # cat schemata
  789. SMBA:0=2048;1=2048;2=2048;3=2048
  790. MB:0=2048;1=2048;2=2048;3=2048
  791. L3:0=ffff;1=ffff;2=ffff;3=ffff
  792. # echo "SMBA:1=64" > schemata
  793. # cat schemata
  794. SMBA:0=2048;1= 64;2=2048;3=2048
  795. MB:0=2048;1=2048;2=2048;3=2048
  796. L3:0=ffff;1=ffff;2=ffff;3=ffff
  797. Cache Pseudo-Locking
  798. ====================
  799. CAT enables a user to specify the amount of cache space that an
  800. application can fill. Cache pseudo-locking builds on the fact that a
  801. CPU can still read and write data pre-allocated outside its current
  802. allocated area on a cache hit. With cache pseudo-locking, data can be
  803. preloaded into a reserved portion of cache that no application can
  804. fill, and from that point on will only serve cache hits. The cache
  805. pseudo-locked memory is made accessible to user space where an
  806. application can map it into its virtual address space and thus have
  807. a region of memory with reduced average read latency.
  808. The creation of a cache pseudo-locked region is triggered by a request
  809. from the user to do so that is accompanied by a schemata of the region
  810. to be pseudo-locked. The cache pseudo-locked region is created as follows:
  811. - Create a CAT allocation CLOSNEW with a CBM matching the schemata
  812. from the user of the cache region that will contain the pseudo-locked
  813. memory. This region must not overlap with any current CAT allocation/CLOS
  814. on the system and no future overlap with this cache region is allowed
  815. while the pseudo-locked region exists.
  816. - Create a contiguous region of memory of the same size as the cache
  817. region.
  818. - Flush the cache, disable hardware prefetchers, disable preemption.
  819. - Make CLOSNEW the active CLOS and touch the allocated memory to load
  820. it into the cache.
  821. - Set the previous CLOS as active.
  822. - At this point the closid CLOSNEW can be released - the cache
  823. pseudo-locked region is protected as long as its CBM does not appear in
  824. any CAT allocation. Even though the cache pseudo-locked region will from
  825. this point on not appear in any CBM of any CLOS an application running with
  826. any CLOS will be able to access the memory in the pseudo-locked region since
  827. the region continues to serve cache hits.
  828. - The contiguous region of memory loaded into the cache is exposed to
  829. user-space as a character device.
  830. Cache pseudo-locking increases the probability that data will remain
  831. in the cache via carefully configuring the CAT feature and controlling
  832. application behavior. There is no guarantee that data is placed in
  833. cache. Instructions like INVD, WBINVD, CLFLUSH, etc. can still evict
  834. “locked” data from cache. Power management C-states may shrink or
  835. power off cache. Deeper C-states will automatically be restricted on
  836. pseudo-locked region creation.
  837. It is required that an application using a pseudo-locked region runs
  838. with affinity to the cores (or a subset of the cores) associated
  839. with the cache on which the pseudo-locked region resides. A sanity check
  840. within the code will not allow an application to map pseudo-locked memory
  841. unless it runs with affinity to cores associated with the cache on which the
  842. pseudo-locked region resides. The sanity check is only done during the
  843. initial mmap() handling, there is no enforcement afterwards and the
  844. application self needs to ensure it remains affine to the correct cores.
  845. Pseudo-locking is accomplished in two stages:
  846. 1) During the first stage the system administrator allocates a portion
  847. of cache that should be dedicated to pseudo-locking. At this time an
  848. equivalent portion of memory is allocated, loaded into allocated
  849. cache portion, and exposed as a character device.
  850. 2) During the second stage a user-space application maps (mmap()) the
  851. pseudo-locked memory into its address space.
  852. Cache Pseudo-Locking Interface
  853. ------------------------------
  854. A pseudo-locked region is created using the resctrl interface as follows:
  855. 1) Create a new resource group by creating a new directory in /sys/fs/resctrl.
  856. 2) Change the new resource group's mode to "pseudo-locksetup" by writing
  857. "pseudo-locksetup" to the "mode" file.
  858. 3) Write the schemata of the pseudo-locked region to the "schemata" file. All
  859. bits within the schemata should be "unused" according to the "bit_usage"
  860. file.
  861. On successful pseudo-locked region creation the "mode" file will contain
  862. "pseudo-locked" and a new character device with the same name as the resource
  863. group will exist in /dev/pseudo_lock. This character device can be mmap()'ed
  864. by user space in order to obtain access to the pseudo-locked memory region.
  865. An example of cache pseudo-locked region creation and usage can be found below.
  866. Cache Pseudo-Locking Debugging Interface
  867. ----------------------------------------
  868. The pseudo-locking debugging interface is enabled by default (if
  869. CONFIG_DEBUG_FS is enabled) and can be found in /sys/kernel/debug/resctrl.
  870. There is no explicit way for the kernel to test if a provided memory
  871. location is present in the cache. The pseudo-locking debugging interface uses
  872. the tracing infrastructure to provide two ways to measure cache residency of
  873. the pseudo-locked region:
  874. 1) Memory access latency using the pseudo_lock_mem_latency tracepoint. Data
  875. from these measurements are best visualized using a hist trigger (see
  876. example below). In this test the pseudo-locked region is traversed at
  877. a stride of 32 bytes while hardware prefetchers and preemption
  878. are disabled. This also provides a substitute visualization of cache
  879. hits and misses.
  880. 2) Cache hit and miss measurements using model specific precision counters if
  881. available. Depending on the levels of cache on the system the pseudo_lock_l2
  882. and pseudo_lock_l3 tracepoints are available.
  883. When a pseudo-locked region is created a new debugfs directory is created for
  884. it in debugfs as /sys/kernel/debug/resctrl/<newdir>. A single
  885. write-only file, pseudo_lock_measure, is present in this directory. The
  886. measurement of the pseudo-locked region depends on the number written to this
  887. debugfs file:
  888. 1:
  889. writing "1" to the pseudo_lock_measure file will trigger the latency
  890. measurement captured in the pseudo_lock_mem_latency tracepoint. See
  891. example below.
  892. 2:
  893. writing "2" to the pseudo_lock_measure file will trigger the L2 cache
  894. residency (cache hits and misses) measurement captured in the
  895. pseudo_lock_l2 tracepoint. See example below.
  896. 3:
  897. writing "3" to the pseudo_lock_measure file will trigger the L3 cache
  898. residency (cache hits and misses) measurement captured in the
  899. pseudo_lock_l3 tracepoint.
  900. All measurements are recorded with the tracing infrastructure. This requires
  901. the relevant tracepoints to be enabled before the measurement is triggered.
  902. Example of latency debugging interface
  903. ~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~
  904. In this example a pseudo-locked region named "newlock" was created. Here is
  905. how we can measure the latency in cycles of reading from this region and
  906. visualize this data with a histogram that is available if CONFIG_HIST_TRIGGERS
  907. is set::
  908. # :> /sys/kernel/tracing/trace
  909. # echo 'hist:keys=latency' > /sys/kernel/tracing/events/resctrl/pseudo_lock_mem_latency/trigger
  910. # echo 1 > /sys/kernel/tracing/events/resctrl/pseudo_lock_mem_latency/enable
  911. # echo 1 > /sys/kernel/debug/resctrl/newlock/pseudo_lock_measure
  912. # echo 0 > /sys/kernel/tracing/events/resctrl/pseudo_lock_mem_latency/enable
  913. # cat /sys/kernel/tracing/events/resctrl/pseudo_lock_mem_latency/hist
  914. # event histogram
  915. #
  916. # trigger info: hist:keys=latency:vals=hitcount:sort=hitcount:size=2048 [active]
  917. #
  918. { latency: 456 } hitcount: 1
  919. { latency: 50 } hitcount: 83
  920. { latency: 36 } hitcount: 96
  921. { latency: 44 } hitcount: 174
  922. { latency: 48 } hitcount: 195
  923. { latency: 46 } hitcount: 262
  924. { latency: 42 } hitcount: 693
  925. { latency: 40 } hitcount: 3204
  926. { latency: 38 } hitcount: 3484
  927. Totals:
  928. Hits: 8192
  929. Entries: 9
  930. Dropped: 0
  931. Example of cache hits/misses debugging
  932. ~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~
  933. In this example a pseudo-locked region named "newlock" was created on the L2
  934. cache of a platform. Here is how we can obtain details of the cache hits
  935. and misses using the platform's precision counters.
  936. ::
  937. # :> /sys/kernel/tracing/trace
  938. # echo 1 > /sys/kernel/tracing/events/resctrl/pseudo_lock_l2/enable
  939. # echo 2 > /sys/kernel/debug/resctrl/newlock/pseudo_lock_measure
  940. # echo 0 > /sys/kernel/tracing/events/resctrl/pseudo_lock_l2/enable
  941. # cat /sys/kernel/tracing/trace
  942. # tracer: nop
  943. #
  944. # _-----=> irqs-off
  945. # / _----=> need-resched
  946. # | / _---=> hardirq/softirq
  947. # || / _--=> preempt-depth
  948. # ||| / delay
  949. # TASK-PID CPU# |||| TIMESTAMP FUNCTION
  950. # | | | |||| | |
  951. pseudo_lock_mea-1672 [002] .... 3132.860500: pseudo_lock_l2: hits=4097 miss=0
  952. Examples for RDT allocation usage
  953. ~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~
  954. 1) Example 1
  955. On a two socket machine (one L3 cache per socket) with just four bits
  956. for cache bit masks, minimum b/w of 10% with a memory bandwidth
  957. granularity of 10%.
  958. ::
  959. # mount -t resctrl resctrl /sys/fs/resctrl
  960. # cd /sys/fs/resctrl
  961. # mkdir p0 p1
  962. # echo "L3:0=3;1=c\nMB:0=50;1=50" > /sys/fs/resctrl/p0/schemata
  963. # echo "L3:0=3;1=3\nMB:0=50;1=50" > /sys/fs/resctrl/p1/schemata
  964. The default resource group is unmodified, so we have access to all parts
  965. of all caches (its schemata file reads "L3:0=f;1=f").
  966. Tasks that are under the control of group "p0" may only allocate from the
  967. "lower" 50% on cache ID 0, and the "upper" 50% of cache ID 1.
  968. Tasks in group "p1" use the "lower" 50% of cache on both sockets.
  969. Similarly, tasks that are under the control of group "p0" may use a
  970. maximum memory b/w of 50% on socket0 and 50% on socket 1.
  971. Tasks in group "p1" may also use 50% memory b/w on both sockets.
  972. Note that unlike cache masks, memory b/w cannot specify whether these
  973. allocations can overlap or not. The allocations specifies the maximum
  974. b/w that the group may be able to use and the system admin can configure
  975. the b/w accordingly.
  976. If resctrl is using the software controller (mba_sc) then user can enter the
  977. max b/w in MB rather than the percentage values.
  978. ::
  979. # echo "L3:0=3;1=c\nMB:0=1024;1=500" > /sys/fs/resctrl/p0/schemata
  980. # echo "L3:0=3;1=3\nMB:0=1024;1=500" > /sys/fs/resctrl/p1/schemata
  981. In the above example the tasks in "p1" and "p0" on socket 0 would use a max b/w
  982. of 1024MB where as on socket 1 they would use 500MB.
  983. 2) Example 2
  984. Again two sockets, but this time with a more realistic 20-bit mask.
  985. Two real time tasks pid=1234 running on processor 0 and pid=5678 running on
  986. processor 1 on socket 0 on a 2-socket and dual core machine. To avoid noisy
  987. neighbors, each of the two real-time tasks exclusively occupies one quarter
  988. of L3 cache on socket 0.
  989. ::
  990. # mount -t resctrl resctrl /sys/fs/resctrl
  991. # cd /sys/fs/resctrl
  992. First we reset the schemata for the default group so that the "upper"
  993. 50% of the L3 cache on socket 0 and 50% of memory b/w cannot be used by
  994. ordinary tasks::
  995. # echo "L3:0=3ff;1=fffff\nMB:0=50;1=100" > schemata
  996. Next we make a resource group for our first real time task and give
  997. it access to the "top" 25% of the cache on socket 0.
  998. ::
  999. # mkdir p0
  1000. # echo "L3:0=f8000;1=fffff" > p0/schemata
  1001. Finally we move our first real time task into this resource group. We
  1002. also use taskset(1) to ensure the task always runs on a dedicated CPU
  1003. on socket 0. Most uses of resource groups will also constrain which
  1004. processors tasks run on.
  1005. ::
  1006. # echo 1234 > p0/tasks
  1007. # taskset -cp 1 1234
  1008. Ditto for the second real time task (with the remaining 25% of cache)::
  1009. # mkdir p1
  1010. # echo "L3:0=7c00;1=fffff" > p1/schemata
  1011. # echo 5678 > p1/tasks
  1012. # taskset -cp 2 5678
  1013. For the same 2 socket system with memory b/w resource and CAT L3 the
  1014. schemata would look like(Assume min_bandwidth 10 and bandwidth_gran is
  1015. 10):
  1016. For our first real time task this would request 20% memory b/w on socket 0.
  1017. ::
  1018. # echo -e "L3:0=f8000;1=fffff\nMB:0=20;1=100" > p0/schemata
  1019. For our second real time task this would request an other 20% memory b/w
  1020. on socket 0.
  1021. ::
  1022. # echo -e "L3:0=f8000;1=fffff\nMB:0=20;1=100" > p0/schemata
  1023. 3) Example 3
  1024. A single socket system which has real-time tasks running on core 4-7 and
  1025. non real-time workload assigned to core 0-3. The real-time tasks share text
  1026. and data, so a per task association is not required and due to interaction
  1027. with the kernel it's desired that the kernel on these cores shares L3 with
  1028. the tasks.
  1029. ::
  1030. # mount -t resctrl resctrl /sys/fs/resctrl
  1031. # cd /sys/fs/resctrl
  1032. First we reset the schemata for the default group so that the "upper"
  1033. 50% of the L3 cache on socket 0, and 50% of memory bandwidth on socket 0
  1034. cannot be used by ordinary tasks::
  1035. # echo "L3:0=3ff\nMB:0=50" > schemata
  1036. Next we make a resource group for our real time cores and give it access
  1037. to the "top" 50% of the cache on socket 0 and 50% of memory bandwidth on
  1038. socket 0.
  1039. ::
  1040. # mkdir p0
  1041. # echo "L3:0=ffc00\nMB:0=50" > p0/schemata
  1042. Finally we move core 4-7 over to the new group and make sure that the
  1043. kernel and the tasks running there get 50% of the cache. They should
  1044. also get 50% of memory bandwidth assuming that the cores 4-7 are SMT
  1045. siblings and only the real time threads are scheduled on the cores 4-7.
  1046. ::
  1047. # echo F0 > p0/cpus
  1048. 4) Example 4
  1049. The resource groups in previous examples were all in the default "shareable"
  1050. mode allowing sharing of their cache allocations. If one resource group
  1051. configures a cache allocation then nothing prevents another resource group
  1052. to overlap with that allocation.
  1053. In this example a new exclusive resource group will be created on a L2 CAT
  1054. system with two L2 cache instances that can be configured with an 8-bit
  1055. capacity bitmask. The new exclusive resource group will be configured to use
  1056. 25% of each cache instance.
  1057. ::
  1058. # mount -t resctrl resctrl /sys/fs/resctrl/
  1059. # cd /sys/fs/resctrl
  1060. First, we observe that the default group is configured to allocate to all L2
  1061. cache::
  1062. # cat schemata
  1063. L2:0=ff;1=ff
  1064. We could attempt to create the new resource group at this point, but it will
  1065. fail because of the overlap with the schemata of the default group::
  1066. # mkdir p0
  1067. # echo 'L2:0=0x3;1=0x3' > p0/schemata
  1068. # cat p0/mode
  1069. shareable
  1070. # echo exclusive > p0/mode
  1071. -sh: echo: write error: Invalid argument
  1072. # cat info/last_cmd_status
  1073. schemata overlaps
  1074. To ensure that there is no overlap with another resource group the default
  1075. resource group's schemata has to change, making it possible for the new
  1076. resource group to become exclusive.
  1077. ::
  1078. # echo 'L2:0=0xfc;1=0xfc' > schemata
  1079. # echo exclusive > p0/mode
  1080. # grep . p0/*
  1081. p0/cpus:0
  1082. p0/mode:exclusive
  1083. p0/schemata:L2:0=03;1=03
  1084. p0/size:L2:0=262144;1=262144
  1085. A new resource group will on creation not overlap with an exclusive resource
  1086. group::
  1087. # mkdir p1
  1088. # grep . p1/*
  1089. p1/cpus:0
  1090. p1/mode:shareable
  1091. p1/schemata:L2:0=fc;1=fc
  1092. p1/size:L2:0=786432;1=786432
  1093. The bit_usage will reflect how the cache is used::
  1094. # cat info/L2/bit_usage
  1095. 0=SSSSSSEE;1=SSSSSSEE
  1096. A resource group cannot be forced to overlap with an exclusive resource group::
  1097. # echo 'L2:0=0x1;1=0x1' > p1/schemata
  1098. -sh: echo: write error: Invalid argument
  1099. # cat info/last_cmd_status
  1100. overlaps with exclusive group
  1101. Example of Cache Pseudo-Locking
  1102. ~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~
  1103. Lock portion of L2 cache from cache id 1 using CBM 0x3. Pseudo-locked
  1104. region is exposed at /dev/pseudo_lock/newlock that can be provided to
  1105. application for argument to mmap().
  1106. ::
  1107. # mount -t resctrl resctrl /sys/fs/resctrl/
  1108. # cd /sys/fs/resctrl
  1109. Ensure that there are bits available that can be pseudo-locked, since only
  1110. unused bits can be pseudo-locked the bits to be pseudo-locked needs to be
  1111. removed from the default resource group's schemata::
  1112. # cat info/L2/bit_usage
  1113. 0=SSSSSSSS;1=SSSSSSSS
  1114. # echo 'L2:1=0xfc' > schemata
  1115. # cat info/L2/bit_usage
  1116. 0=SSSSSSSS;1=SSSSSS00
  1117. Create a new resource group that will be associated with the pseudo-locked
  1118. region, indicate that it will be used for a pseudo-locked region, and
  1119. configure the requested pseudo-locked region capacity bitmask::
  1120. # mkdir newlock
  1121. # echo pseudo-locksetup > newlock/mode
  1122. # echo 'L2:1=0x3' > newlock/schemata
  1123. On success the resource group's mode will change to pseudo-locked, the
  1124. bit_usage will reflect the pseudo-locked region, and the character device
  1125. exposing the pseudo-locked region will exist::
  1126. # cat newlock/mode
  1127. pseudo-locked
  1128. # cat info/L2/bit_usage
  1129. 0=SSSSSSSS;1=SSSSSSPP
  1130. # ls -l /dev/pseudo_lock/newlock
  1131. crw------- 1 root root 243, 0 Apr 3 05:01 /dev/pseudo_lock/newlock
  1132. ::
  1133. /*
  1134. * Example code to access one page of pseudo-locked cache region
  1135. * from user space.
  1136. */
  1137. #define _GNU_SOURCE
  1138. #include <fcntl.h>
  1139. #include <sched.h>
  1140. #include <stdio.h>
  1141. #include <stdlib.h>
  1142. #include <unistd.h>
  1143. #include <sys/mman.h>
  1144. /*
  1145. * It is required that the application runs with affinity to only
  1146. * cores associated with the pseudo-locked region. Here the cpu
  1147. * is hardcoded for convenience of example.
  1148. */
  1149. static int cpuid = 2;
  1150. int main(int argc, char *argv[])
  1151. {
  1152. cpu_set_t cpuset;
  1153. long page_size;
  1154. void *mapping;
  1155. int dev_fd;
  1156. int ret;
  1157. page_size = sysconf(_SC_PAGESIZE);
  1158. CPU_ZERO(&cpuset);
  1159. CPU_SET(cpuid, &cpuset);
  1160. ret = sched_setaffinity(0, sizeof(cpuset), &cpuset);
  1161. if (ret < 0) {
  1162. perror("sched_setaffinity");
  1163. exit(EXIT_FAILURE);
  1164. }
  1165. dev_fd = open("/dev/pseudo_lock/newlock", O_RDWR);
  1166. if (dev_fd < 0) {
  1167. perror("open");
  1168. exit(EXIT_FAILURE);
  1169. }
  1170. mapping = mmap(0, page_size, PROT_READ | PROT_WRITE, MAP_SHARED,
  1171. dev_fd, 0);
  1172. if (mapping == MAP_FAILED) {
  1173. perror("mmap");
  1174. close(dev_fd);
  1175. exit(EXIT_FAILURE);
  1176. }
  1177. /* Application interacts with pseudo-locked memory @mapping */
  1178. ret = munmap(mapping, page_size);
  1179. if (ret < 0) {
  1180. perror("munmap");
  1181. close(dev_fd);
  1182. exit(EXIT_FAILURE);
  1183. }
  1184. close(dev_fd);
  1185. exit(EXIT_SUCCESS);
  1186. }
  1187. Locking between applications
  1188. ----------------------------
  1189. Certain operations on the resctrl filesystem, composed of read/writes
  1190. to/from multiple files, must be atomic.
  1191. As an example, the allocation of an exclusive reservation of L3 cache
  1192. involves:
  1193. 1. Read the cbmmasks from each directory or the per-resource "bit_usage"
  1194. 2. Find a contiguous set of bits in the global CBM bitmask that is clear
  1195. in any of the directory cbmmasks
  1196. 3. Create a new directory
  1197. 4. Set the bits found in step 2 to the new directory "schemata" file
  1198. If two applications attempt to allocate space concurrently then they can
  1199. end up allocating the same bits so the reservations are shared instead of
  1200. exclusive.
  1201. To coordinate atomic operations on the resctrlfs and to avoid the problem
  1202. above, the following locking procedure is recommended:
  1203. Locking is based on flock, which is available in libc and also as a shell
  1204. script command
  1205. Write lock:
  1206. A) Take flock(LOCK_EX) on /sys/fs/resctrl
  1207. B) Read/write the directory structure.
  1208. C) funlock
  1209. Read lock:
  1210. A) Take flock(LOCK_SH) on /sys/fs/resctrl
  1211. B) If success read the directory structure.
  1212. C) funlock
  1213. Example with bash::
  1214. # Atomically read directory structure
  1215. $ flock -s /sys/fs/resctrl/ find /sys/fs/resctrl
  1216. # Read directory contents and create new subdirectory
  1217. $ cat create-dir.sh
  1218. find /sys/fs/resctrl/ > output.txt
  1219. mask = function-of(output.txt)
  1220. mkdir /sys/fs/resctrl/newres/
  1221. echo mask > /sys/fs/resctrl/newres/schemata
  1222. $ flock /sys/fs/resctrl/ ./create-dir.sh
  1223. Example with C::
  1224. /*
  1225. * Example code do take advisory locks
  1226. * before accessing resctrl filesystem
  1227. */
  1228. #include <sys/file.h>
  1229. #include <stdlib.h>
  1230. void resctrl_take_shared_lock(int fd)
  1231. {
  1232. int ret;
  1233. /* take shared lock on resctrl filesystem */
  1234. ret = flock(fd, LOCK_SH);
  1235. if (ret) {
  1236. perror("flock");
  1237. exit(-1);
  1238. }
  1239. }
  1240. void resctrl_take_exclusive_lock(int fd)
  1241. {
  1242. int ret;
  1243. /* release lock on resctrl filesystem */
  1244. ret = flock(fd, LOCK_EX);
  1245. if (ret) {
  1246. perror("flock");
  1247. exit(-1);
  1248. }
  1249. }
  1250. void resctrl_release_lock(int fd)
  1251. {
  1252. int ret;
  1253. /* take shared lock on resctrl filesystem */
  1254. ret = flock(fd, LOCK_UN);
  1255. if (ret) {
  1256. perror("flock");
  1257. exit(-1);
  1258. }
  1259. }
  1260. void main(void)
  1261. {
  1262. int fd, ret;
  1263. fd = open("/sys/fs/resctrl", O_DIRECTORY);
  1264. if (fd == -1) {
  1265. perror("open");
  1266. exit(-1);
  1267. }
  1268. resctrl_take_shared_lock(fd);
  1269. /* code to read directory contents */
  1270. resctrl_release_lock(fd);
  1271. resctrl_take_exclusive_lock(fd);
  1272. /* code to read and write directory contents */
  1273. resctrl_release_lock(fd);
  1274. }
  1275. Examples for RDT Monitoring along with allocation usage
  1276. =======================================================
  1277. Reading monitored data
  1278. ----------------------
  1279. Reading an event file (for ex: mon_data/mon_L3_00/llc_occupancy) would
  1280. show the current snapshot of LLC occupancy of the corresponding MON
  1281. group or CTRL_MON group.
  1282. Example 1 (Monitor CTRL_MON group and subset of tasks in CTRL_MON group)
  1283. ------------------------------------------------------------------------
  1284. On a two socket machine (one L3 cache per socket) with just four bits
  1285. for cache bit masks::
  1286. # mount -t resctrl resctrl /sys/fs/resctrl
  1287. # cd /sys/fs/resctrl
  1288. # mkdir p0 p1
  1289. # echo "L3:0=3;1=c" > /sys/fs/resctrl/p0/schemata
  1290. # echo "L3:0=3;1=3" > /sys/fs/resctrl/p1/schemata
  1291. # echo 5678 > p1/tasks
  1292. # echo 5679 > p1/tasks
  1293. The default resource group is unmodified, so we have access to all parts
  1294. of all caches (its schemata file reads "L3:0=f;1=f").
  1295. Tasks that are under the control of group "p0" may only allocate from the
  1296. "lower" 50% on cache ID 0, and the "upper" 50% of cache ID 1.
  1297. Tasks in group "p1" use the "lower" 50% of cache on both sockets.
  1298. Create monitor groups and assign a subset of tasks to each monitor group.
  1299. ::
  1300. # cd /sys/fs/resctrl/p1/mon_groups
  1301. # mkdir m11 m12
  1302. # echo 5678 > m11/tasks
  1303. # echo 5679 > m12/tasks
  1304. fetch data (data shown in bytes)
  1305. ::
  1306. # cat m11/mon_data/mon_L3_00/llc_occupancy
  1307. 16234000
  1308. # cat m11/mon_data/mon_L3_01/llc_occupancy
  1309. 14789000
  1310. # cat m12/mon_data/mon_L3_00/llc_occupancy
  1311. 16789000
  1312. The parent ctrl_mon group shows the aggregated data.
  1313. ::
  1314. # cat /sys/fs/resctrl/p1/mon_data/mon_l3_00/llc_occupancy
  1315. 31234000
  1316. Example 2 (Monitor a task from its creation)
  1317. --------------------------------------------
  1318. On a two socket machine (one L3 cache per socket)::
  1319. # mount -t resctrl resctrl /sys/fs/resctrl
  1320. # cd /sys/fs/resctrl
  1321. # mkdir p0 p1
  1322. An RMID is allocated to the group once its created and hence the <cmd>
  1323. below is monitored from its creation.
  1324. ::
  1325. # echo $$ > /sys/fs/resctrl/p1/tasks
  1326. # <cmd>
  1327. Fetch the data::
  1328. # cat /sys/fs/resctrl/p1/mon_data/mon_l3_00/llc_occupancy
  1329. 31789000
  1330. Example 3 (Monitor without CAT support or before creating CAT groups)
  1331. ---------------------------------------------------------------------
  1332. Assume a system like HSW has only CQM and no CAT support. In this case
  1333. the resctrl will still mount but cannot create CTRL_MON directories.
  1334. But user can create different MON groups within the root group thereby
  1335. able to monitor all tasks including kernel threads.
  1336. This can also be used to profile jobs cache size footprint before being
  1337. able to allocate them to different allocation groups.
  1338. ::
  1339. # mount -t resctrl resctrl /sys/fs/resctrl
  1340. # cd /sys/fs/resctrl
  1341. # mkdir mon_groups/m01
  1342. # mkdir mon_groups/m02
  1343. # echo 3478 > /sys/fs/resctrl/mon_groups/m01/tasks
  1344. # echo 2467 > /sys/fs/resctrl/mon_groups/m02/tasks
  1345. Monitor the groups separately and also get per domain data. From the
  1346. below its apparent that the tasks are mostly doing work on
  1347. domain(socket) 0.
  1348. ::
  1349. # cat /sys/fs/resctrl/mon_groups/m01/mon_L3_00/llc_occupancy
  1350. 31234000
  1351. # cat /sys/fs/resctrl/mon_groups/m01/mon_L3_01/llc_occupancy
  1352. 34555
  1353. # cat /sys/fs/resctrl/mon_groups/m02/mon_L3_00/llc_occupancy
  1354. 31234000
  1355. # cat /sys/fs/resctrl/mon_groups/m02/mon_L3_01/llc_occupancy
  1356. 32789
  1357. Example 4 (Monitor real time tasks)
  1358. -----------------------------------
  1359. A single socket system which has real time tasks running on cores 4-7
  1360. and non real time tasks on other cpus. We want to monitor the cache
  1361. occupancy of the real time threads on these cores.
  1362. ::
  1363. # mount -t resctrl resctrl /sys/fs/resctrl
  1364. # cd /sys/fs/resctrl
  1365. # mkdir p1
  1366. Move the cpus 4-7 over to p1::
  1367. # echo f0 > p1/cpus
  1368. View the llc occupancy snapshot::
  1369. # cat /sys/fs/resctrl/p1/mon_data/mon_L3_00/llc_occupancy
  1370. 11234000
  1371. Examples on working with mbm_assign_mode
  1372. ========================================
  1373. a. Check if MBM counter assignment mode is supported.
  1374. ::
  1375. # mount -t resctrl resctrl /sys/fs/resctrl/
  1376. # cat /sys/fs/resctrl/info/L3_MON/mbm_assign_mode
  1377. [mbm_event]
  1378. default
  1379. The "mbm_event" mode is detected and enabled.
  1380. b. Check how many assignable counters are supported.
  1381. ::
  1382. # cat /sys/fs/resctrl/info/L3_MON/num_mbm_cntrs
  1383. 0=32;1=32
  1384. c. Check how many assignable counters are available for assignment in each domain.
  1385. ::
  1386. # cat /sys/fs/resctrl/info/L3_MON/available_mbm_cntrs
  1387. 0=30;1=30
  1388. d. To list the default group's assign states.
  1389. ::
  1390. # cat /sys/fs/resctrl/mbm_L3_assignments
  1391. mbm_total_bytes:0=e;1=e
  1392. mbm_local_bytes:0=e;1=e
  1393. e. To unassign the counter associated with the mbm_total_bytes event on domain 0.
  1394. ::
  1395. # echo "mbm_total_bytes:0=_" > /sys/fs/resctrl/mbm_L3_assignments
  1396. # cat /sys/fs/resctrl/mbm_L3_assignments
  1397. mbm_total_bytes:0=_;1=e
  1398. mbm_local_bytes:0=e;1=e
  1399. f. To unassign the counter associated with the mbm_total_bytes event on all domains.
  1400. ::
  1401. # echo "mbm_total_bytes:*=_" > /sys/fs/resctrl/mbm_L3_assignments
  1402. # cat /sys/fs/resctrl/mbm_L3_assignment
  1403. mbm_total_bytes:0=_;1=_
  1404. mbm_local_bytes:0=e;1=e
  1405. g. To assign a counter associated with the mbm_total_bytes event on all domains in
  1406. exclusive mode.
  1407. ::
  1408. # echo "mbm_total_bytes:*=e" > /sys/fs/resctrl/mbm_L3_assignments
  1409. # cat /sys/fs/resctrl/mbm_L3_assignments
  1410. mbm_total_bytes:0=e;1=e
  1411. mbm_local_bytes:0=e;1=e
  1412. h. Read the events mbm_total_bytes and mbm_local_bytes of the default group. There is
  1413. no change in reading the events with the assignment.
  1414. ::
  1415. # cat /sys/fs/resctrl/mon_data/mon_L3_00/mbm_total_bytes
  1416. 779247936
  1417. # cat /sys/fs/resctrl/mon_data/mon_L3_01/mbm_total_bytes
  1418. 562324232
  1419. # cat /sys/fs/resctrl/mon_data/mon_L3_00/mbm_local_bytes
  1420. 212122123
  1421. # cat /sys/fs/resctrl/mon_data/mon_L3_01/mbm_local_bytes
  1422. 121212144
  1423. i. Check the event configurations.
  1424. ::
  1425. # cat /sys/fs/resctrl/info/L3_MON/event_configs/mbm_total_bytes/event_filter
  1426. local_reads,remote_reads,local_non_temporal_writes,remote_non_temporal_writes,
  1427. local_reads_slow_memory,remote_reads_slow_memory,dirty_victim_writes_all
  1428. # cat /sys/fs/resctrl/info/L3_MON/event_configs/mbm_local_bytes/event_filter
  1429. local_reads,local_non_temporal_writes,local_reads_slow_memory
  1430. j. Change the event configuration for mbm_local_bytes.
  1431. ::
  1432. # echo "local_reads, local_non_temporal_writes, local_reads_slow_memory, remote_reads" >
  1433. /sys/fs/resctrl/info/L3_MON/event_configs/mbm_local_bytes/event_filter
  1434. # cat /sys/fs/resctrl/info/L3_MON/event_configs/mbm_local_bytes/event_filter
  1435. local_reads,local_non_temporal_writes,local_reads_slow_memory,remote_reads
  1436. k. Now read the local events again. The first read may come back with "Unavailable"
  1437. status. The subsequent read of mbm_local_bytes will display the current value.
  1438. ::
  1439. # cat /sys/fs/resctrl/mon_data/mon_L3_00/mbm_local_bytes
  1440. Unavailable
  1441. # cat /sys/fs/resctrl/mon_data/mon_L3_00/mbm_local_bytes
  1442. 2252323
  1443. # cat /sys/fs/resctrl/mon_data/mon_L3_01/mbm_local_bytes
  1444. Unavailable
  1445. # cat /sys/fs/resctrl/mon_data/mon_L3_01/mbm_local_bytes
  1446. 1566565
  1447. l. Users have the option to go back to 'default' mbm_assign_mode if required. This can be
  1448. done using the following command. Note that switching the mbm_assign_mode may reset all
  1449. the MBM counters (and thus all MBM events) of all the resctrl groups.
  1450. ::
  1451. # echo "default" > /sys/fs/resctrl/info/L3_MON/mbm_assign_mode
  1452. # cat /sys/fs/resctrl/info/L3_MON/mbm_assign_mode
  1453. mbm_event
  1454. [default]
  1455. m. Unmount the resctrl filesystem.
  1456. ::
  1457. # umount /sys/fs/resctrl/
  1458. Intel RDT Errata
  1459. ================
  1460. Intel MBM Counters May Report System Memory Bandwidth Incorrectly
  1461. -----------------------------------------------------------------
  1462. Errata SKX99 for Skylake server and BDF102 for Broadwell server.
  1463. Problem: Intel Memory Bandwidth Monitoring (MBM) counters track metrics
  1464. according to the assigned Resource Monitor ID (RMID) for that logical
  1465. core. The IA32_QM_CTR register (MSR 0xC8E), used to report these
  1466. metrics, may report incorrect system bandwidth for certain RMID values.
  1467. Implication: Due to the errata, system memory bandwidth may not match
  1468. what is reported.
  1469. Workaround: MBM total and local readings are corrected according to the
  1470. following correction factor table:
  1471. +---------------+---------------+---------------+-----------------+
  1472. |core count |rmid count |rmid threshold |correction factor|
  1473. +---------------+---------------+---------------+-----------------+
  1474. |1 |8 |0 |1.000000 |
  1475. +---------------+---------------+---------------+-----------------+
  1476. |2 |16 |0 |1.000000 |
  1477. +---------------+---------------+---------------+-----------------+
  1478. |3 |24 |15 |0.969650 |
  1479. +---------------+---------------+---------------+-----------------+
  1480. |4 |32 |0 |1.000000 |
  1481. +---------------+---------------+---------------+-----------------+
  1482. |6 |48 |31 |0.969650 |
  1483. +---------------+---------------+---------------+-----------------+
  1484. |7 |56 |47 |1.142857 |
  1485. +---------------+---------------+---------------+-----------------+
  1486. |8 |64 |0 |1.000000 |
  1487. +---------------+---------------+---------------+-----------------+
  1488. |9 |72 |63 |1.185115 |
  1489. +---------------+---------------+---------------+-----------------+
  1490. |10 |80 |63 |1.066553 |
  1491. +---------------+---------------+---------------+-----------------+
  1492. |11 |88 |79 |1.454545 |
  1493. +---------------+---------------+---------------+-----------------+
  1494. |12 |96 |0 |1.000000 |
  1495. +---------------+---------------+---------------+-----------------+
  1496. |13 |104 |95 |1.230769 |
  1497. +---------------+---------------+---------------+-----------------+
  1498. |14 |112 |95 |1.142857 |
  1499. +---------------+---------------+---------------+-----------------+
  1500. |15 |120 |95 |1.066667 |
  1501. +---------------+---------------+---------------+-----------------+
  1502. |16 |128 |0 |1.000000 |
  1503. +---------------+---------------+---------------+-----------------+
  1504. |17 |136 |127 |1.254863 |
  1505. +---------------+---------------+---------------+-----------------+
  1506. |18 |144 |127 |1.185255 |
  1507. +---------------+---------------+---------------+-----------------+
  1508. |19 |152 |0 |1.000000 |
  1509. +---------------+---------------+---------------+-----------------+
  1510. |20 |160 |127 |1.066667 |
  1511. +---------------+---------------+---------------+-----------------+
  1512. |21 |168 |0 |1.000000 |
  1513. +---------------+---------------+---------------+-----------------+
  1514. |22 |176 |159 |1.454334 |
  1515. +---------------+---------------+---------------+-----------------+
  1516. |23 |184 |0 |1.000000 |
  1517. +---------------+---------------+---------------+-----------------+
  1518. |24 |192 |127 |0.969744 |
  1519. +---------------+---------------+---------------+-----------------+
  1520. |25 |200 |191 |1.280246 |
  1521. +---------------+---------------+---------------+-----------------+
  1522. |26 |208 |191 |1.230921 |
  1523. +---------------+---------------+---------------+-----------------+
  1524. |27 |216 |0 |1.000000 |
  1525. +---------------+---------------+---------------+-----------------+
  1526. |28 |224 |191 |1.143118 |
  1527. +---------------+---------------+---------------+-----------------+
  1528. If rmid > rmid threshold, MBM total and local values should be multiplied
  1529. by the correction factor.
  1530. See:
  1531. 1. Erratum SKX99 in Intel Xeon Processor Scalable Family Specification Update:
  1532. http://web.archive.org/web/20200716124958/https://www.intel.com/content/www/us/en/processors/xeon/scalable/xeon-scalable-spec-update.html
  1533. 2. Erratum BDF102 in Intel Xeon E5-2600 v4 Processor Product Family Specification Update:
  1534. http://web.archive.org/web/20191125200531/https://www.intel.com/content/dam/www/public/us/en/documents/specification-updates/xeon-e5-v4-spec-update.pdf
  1535. 3. The errata in Intel Resource Director Technology (Intel RDT) on 2nd Generation Intel Xeon Scalable Processors Reference Manual:
  1536. https://software.intel.com/content/www/us/en/develop/articles/intel-resource-director-technology-rdt-reference-manual.html
  1537. for further information.