glock.c 70 KB

12345678910111213141516171819202122232425262728293031323334353637383940414243444546474849505152535455565758596061626364656667686970717273747576777879808182838485868788899091929394959697989910010110210310410510610710810911011111211311411511611711811912012112212312412512612712812913013113213313413513613713813914014114214314414514614714814915015115215315415515615715815916016116216316416516616716816917017117217317417517617717817918018118218318418518618718818919019119219319419519619719819920020120220320420520620720820921021121221321421521621721821922022122222322422522622722822923023123223323423523623723823924024124224324424524624724824925025125225325425525625725825926026126226326426526626726826927027127227327427527627727827928028128228328428528628728828929029129229329429529629729829930030130230330430530630730830931031131231331431531631731831932032132232332432532632732832933033133233333433533633733833934034134234334434534634734834935035135235335435535635735835936036136236336436536636736836937037137237337437537637737837938038138238338438538638738838939039139239339439539639739839940040140240340440540640740840941041141241341441541641741841942042142242342442542642742842943043143243343443543643743843944044144244344444544644744844945045145245345445545645745845946046146246346446546646746846947047147247347447547647747847948048148248348448548648748848949049149249349449549649749849950050150250350450550650750850951051151251351451551651751851952052152252352452552652752852953053153253353453553653753853954054154254354454554654754854955055155255355455555655755855956056156256356456556656756856957057157257357457557657757857958058158258358458558658758858959059159259359459559659759859960060160260360460560660760860961061161261361461561661761861962062162262362462562662762862963063163263363463563663763863964064164264364464564664764864965065165265365465565665765865966066166266366466566666766866967067167267367467567667767867968068168268368468568668768868969069169269369469569669769869970070170270370470570670770870971071171271371471571671771871972072172272372472572672772872973073173273373473573673773873974074174274374474574674774874975075175275375475575675775875976076176276376476576676776876977077177277377477577677777877978078178278378478578678778878979079179279379479579679779879980080180280380480580680780880981081181281381481581681781881982082182282382482582682782882983083183283383483583683783883984084184284384484584684784884985085185285385485585685785885986086186286386486586686786886987087187287387487587687787887988088188288388488588688788888989089189289389489589689789889990090190290390490590690790890991091191291391491591691791891992092192292392492592692792892993093193293393493593693793893994094194294394494594694794894995095195295395495595695795895996096196296396496596696796896997097197297397497597697797897998098198298398498598698798898999099199299399499599699799899910001001100210031004100510061007100810091010101110121013101410151016101710181019102010211022102310241025102610271028102910301031103210331034103510361037103810391040104110421043104410451046104710481049105010511052105310541055105610571058105910601061106210631064106510661067106810691070107110721073107410751076107710781079108010811082108310841085108610871088108910901091109210931094109510961097109810991100110111021103110411051106110711081109111011111112111311141115111611171118111911201121112211231124112511261127112811291130113111321133113411351136113711381139114011411142114311441145114611471148114911501151115211531154115511561157115811591160116111621163116411651166116711681169117011711172117311741175117611771178117911801181118211831184118511861187118811891190119111921193119411951196119711981199120012011202120312041205120612071208120912101211121212131214121512161217121812191220122112221223122412251226122712281229123012311232123312341235123612371238123912401241124212431244124512461247124812491250125112521253125412551256125712581259126012611262126312641265126612671268126912701271127212731274127512761277127812791280128112821283128412851286128712881289129012911292129312941295129612971298129913001301130213031304130513061307130813091310131113121313131413151316131713181319132013211322132313241325132613271328132913301331133213331334133513361337133813391340134113421343134413451346134713481349135013511352135313541355135613571358135913601361136213631364136513661367136813691370137113721373137413751376137713781379138013811382138313841385138613871388138913901391139213931394139513961397139813991400140114021403140414051406140714081409141014111412141314141415141614171418141914201421142214231424142514261427142814291430143114321433143414351436143714381439144014411442144314441445144614471448144914501451145214531454145514561457145814591460146114621463146414651466146714681469147014711472147314741475147614771478147914801481148214831484148514861487148814891490149114921493149414951496149714981499150015011502150315041505150615071508150915101511151215131514151515161517151815191520152115221523152415251526152715281529153015311532153315341535153615371538153915401541154215431544154515461547154815491550155115521553155415551556155715581559156015611562156315641565156615671568156915701571157215731574157515761577157815791580158115821583158415851586158715881589159015911592159315941595159615971598159916001601160216031604160516061607160816091610161116121613161416151616161716181619162016211622162316241625162616271628162916301631163216331634163516361637163816391640164116421643164416451646164716481649165016511652165316541655165616571658165916601661166216631664166516661667166816691670167116721673167416751676167716781679168016811682168316841685168616871688168916901691169216931694169516961697169816991700170117021703170417051706170717081709171017111712171317141715171617171718171917201721172217231724172517261727172817291730173117321733173417351736173717381739174017411742174317441745174617471748174917501751175217531754175517561757175817591760176117621763176417651766176717681769177017711772177317741775177617771778177917801781178217831784178517861787178817891790179117921793179417951796179717981799180018011802180318041805180618071808180918101811181218131814181518161817181818191820182118221823182418251826182718281829183018311832183318341835183618371838183918401841184218431844184518461847184818491850185118521853185418551856185718581859186018611862186318641865186618671868186918701871187218731874187518761877187818791880188118821883188418851886188718881889189018911892189318941895189618971898189919001901190219031904190519061907190819091910191119121913191419151916191719181919192019211922192319241925192619271928192919301931193219331934193519361937193819391940194119421943194419451946194719481949195019511952195319541955195619571958195919601961196219631964196519661967196819691970197119721973197419751976197719781979198019811982198319841985198619871988198919901991199219931994199519961997199819992000200120022003200420052006200720082009201020112012201320142015201620172018201920202021202220232024202520262027202820292030203120322033203420352036203720382039204020412042204320442045204620472048204920502051205220532054205520562057205820592060206120622063206420652066206720682069207020712072207320742075207620772078207920802081208220832084208520862087208820892090209120922093209420952096209720982099210021012102210321042105210621072108210921102111211221132114211521162117211821192120212121222123212421252126212721282129213021312132213321342135213621372138213921402141214221432144214521462147214821492150215121522153215421552156215721582159216021612162216321642165216621672168216921702171217221732174217521762177217821792180218121822183218421852186218721882189219021912192219321942195219621972198219922002201220222032204220522062207220822092210221122122213221422152216221722182219222022212222222322242225222622272228222922302231223222332234223522362237223822392240224122422243224422452246224722482249225022512252225322542255225622572258225922602261226222632264226522662267226822692270227122722273227422752276227722782279228022812282228322842285228622872288228922902291229222932294229522962297229822992300230123022303230423052306230723082309231023112312231323142315231623172318231923202321232223232324232523262327232823292330233123322333233423352336233723382339234023412342234323442345234623472348234923502351235223532354235523562357235823592360236123622363236423652366236723682369237023712372237323742375237623772378237923802381238223832384238523862387238823892390239123922393239423952396239723982399240024012402240324042405240624072408240924102411241224132414241524162417241824192420242124222423242424252426242724282429243024312432243324342435243624372438243924402441244224432444244524462447244824492450245124522453245424552456245724582459246024612462246324642465246624672468246924702471247224732474247524762477247824792480248124822483248424852486248724882489249024912492249324942495249624972498249925002501250225032504250525062507250825092510251125122513251425152516251725182519252025212522252325242525252625272528252925302531253225332534253525362537253825392540254125422543254425452546254725482549255025512552255325542555255625572558255925602561256225632564256525662567256825692570257125722573257425752576257725782579258025812582258325842585258625872588258925902591259225932594259525962597259825992600260126022603260426052606260726082609261026112612261326142615261626172618261926202621262226232624262526262627262826292630263126322633263426352636263726382639264026412642264326442645264626472648264926502651265226532654265526562657265826592660266126622663266426652666266726682669267026712672267326742675267626772678267926802681268226832684268526862687268826892690269126922693269426952696269726982699270027012702270327042705270627072708270927102711271227132714271527162717271827192720272127222723272427252726272727282729273027312732273327342735273627372738273927402741274227432744274527462747274827492750275127522753275427552756275727582759276027612762276327642765276627672768276927702771277227732774277527762777277827792780278127822783278427852786278727882789279027912792279327942795279627972798279928002801280228032804280528062807280828092810281128122813281428152816281728182819282028212822282328242825282628272828282928302831283228332834283528362837283828392840284128422843284428452846
  1. // SPDX-License-Identifier: GPL-2.0-only
  2. /*
  3. * Copyright (C) Sistina Software, Inc. 1997-2003 All rights reserved.
  4. * Copyright (C) 2004-2008 Red Hat, Inc. All rights reserved.
  5. */
  6. #define pr_fmt(fmt) KBUILD_MODNAME ": " fmt
  7. #include <linux/sched.h>
  8. #include <linux/slab.h>
  9. #include <linux/spinlock.h>
  10. #include <linux/buffer_head.h>
  11. #include <linux/delay.h>
  12. #include <linux/sort.h>
  13. #include <linux/hash.h>
  14. #include <linux/jhash.h>
  15. #include <linux/kallsyms.h>
  16. #include <linux/gfs2_ondisk.h>
  17. #include <linux/list.h>
  18. #include <linux/wait.h>
  19. #include <linux/module.h>
  20. #include <linux/uaccess.h>
  21. #include <linux/seq_file.h>
  22. #include <linux/debugfs.h>
  23. #include <linux/kthread.h>
  24. #include <linux/freezer.h>
  25. #include <linux/workqueue.h>
  26. #include <linux/jiffies.h>
  27. #include <linux/rcupdate.h>
  28. #include <linux/rculist_bl.h>
  29. #include <linux/bit_spinlock.h>
  30. #include <linux/percpu.h>
  31. #include <linux/list_sort.h>
  32. #include <linux/lockref.h>
  33. #include <linux/rhashtable.h>
  34. #include <linux/pid_namespace.h>
  35. #include <linux/file.h>
  36. #include <linux/random.h>
  37. #include "gfs2.h"
  38. #include "incore.h"
  39. #include "glock.h"
  40. #include "glops.h"
  41. #include "inode.h"
  42. #include "lops.h"
  43. #include "meta_io.h"
  44. #include "quota.h"
  45. #include "super.h"
  46. #include "util.h"
  47. #include "bmap.h"
  48. #define CREATE_TRACE_POINTS
  49. #include "trace_gfs2.h"
  50. struct gfs2_glock_iter {
  51. struct gfs2_sbd *sdp; /* incore superblock */
  52. struct rhashtable_iter hti; /* rhashtable iterator */
  53. struct gfs2_glock *gl; /* current glock struct */
  54. loff_t last_pos; /* last position */
  55. };
  56. typedef void (*glock_examiner) (struct gfs2_glock * gl);
  57. static void do_xmote(struct gfs2_glock *gl, struct gfs2_holder *gh,
  58. unsigned int target, bool may_cancel);
  59. static void request_demote(struct gfs2_glock *gl, unsigned int state,
  60. unsigned long delay, bool remote);
  61. static struct dentry *gfs2_root;
  62. static LIST_HEAD(lru_list);
  63. static atomic_t lru_count = ATOMIC_INIT(0);
  64. static DEFINE_SPINLOCK(lru_lock);
  65. #define GFS2_GL_HASH_SHIFT 15
  66. #define GFS2_GL_HASH_SIZE BIT(GFS2_GL_HASH_SHIFT)
  67. static const struct rhashtable_params ht_parms = {
  68. .nelem_hint = GFS2_GL_HASH_SIZE * 3 / 4,
  69. .key_len = offsetofend(struct lm_lockname, ln_type),
  70. .key_offset = offsetof(struct gfs2_glock, gl_name),
  71. .head_offset = offsetof(struct gfs2_glock, gl_node),
  72. };
  73. static struct rhashtable gl_hash_table;
  74. #define GLOCK_WAIT_TABLE_BITS 12
  75. #define GLOCK_WAIT_TABLE_SIZE (1 << GLOCK_WAIT_TABLE_BITS)
  76. static wait_queue_head_t glock_wait_table[GLOCK_WAIT_TABLE_SIZE] __cacheline_aligned;
  77. struct wait_glock_queue {
  78. struct lm_lockname *name;
  79. wait_queue_entry_t wait;
  80. };
  81. static int glock_wake_function(wait_queue_entry_t *wait, unsigned int mode,
  82. int sync, void *key)
  83. {
  84. struct wait_glock_queue *wait_glock =
  85. container_of(wait, struct wait_glock_queue, wait);
  86. struct lm_lockname *wait_name = wait_glock->name;
  87. struct lm_lockname *wake_name = key;
  88. if (wake_name->ln_sbd != wait_name->ln_sbd ||
  89. wake_name->ln_number != wait_name->ln_number ||
  90. wake_name->ln_type != wait_name->ln_type)
  91. return 0;
  92. return autoremove_wake_function(wait, mode, sync, key);
  93. }
  94. static wait_queue_head_t *glock_waitqueue(struct lm_lockname *name)
  95. {
  96. u32 hash = jhash2((u32 *)name, ht_parms.key_len / 4, 0);
  97. return glock_wait_table + hash_32(hash, GLOCK_WAIT_TABLE_BITS);
  98. }
  99. /**
  100. * wake_up_glock - Wake up waiters on a glock
  101. * @gl: the glock
  102. */
  103. static void wake_up_glock(struct gfs2_glock *gl)
  104. {
  105. wait_queue_head_t *wq = glock_waitqueue(&gl->gl_name);
  106. if (waitqueue_active(wq))
  107. __wake_up(wq, TASK_NORMAL, 1, &gl->gl_name);
  108. }
  109. static void gfs2_glock_dealloc(struct rcu_head *rcu)
  110. {
  111. struct gfs2_glock *gl = container_of(rcu, struct gfs2_glock, gl_rcu);
  112. kfree(gl->gl_lksb.sb_lvbptr);
  113. if (gl->gl_ops->go_flags & GLOF_ASPACE) {
  114. struct gfs2_glock_aspace *gla =
  115. container_of(gl, struct gfs2_glock_aspace, glock);
  116. kmem_cache_free(gfs2_glock_aspace_cachep, gla);
  117. } else
  118. kmem_cache_free(gfs2_glock_cachep, gl);
  119. }
  120. static void __gfs2_glock_free(struct gfs2_glock *gl)
  121. {
  122. rhashtable_remove_fast(&gl_hash_table, &gl->gl_node, ht_parms);
  123. smp_mb();
  124. wake_up_glock(gl);
  125. call_rcu(&gl->gl_rcu, gfs2_glock_dealloc);
  126. }
  127. void gfs2_glock_free(struct gfs2_glock *gl) {
  128. struct gfs2_sbd *sdp = glock_sbd(gl);
  129. __gfs2_glock_free(gl);
  130. if (atomic_dec_and_test(&sdp->sd_glock_disposal))
  131. wake_up(&sdp->sd_kill_wait);
  132. }
  133. void gfs2_glock_free_later(struct gfs2_glock *gl) {
  134. struct gfs2_sbd *sdp = glock_sbd(gl);
  135. spin_lock(&lru_lock);
  136. list_add(&gl->gl_lru, &sdp->sd_dead_glocks);
  137. spin_unlock(&lru_lock);
  138. if (atomic_dec_and_test(&sdp->sd_glock_disposal))
  139. wake_up(&sdp->sd_kill_wait);
  140. }
  141. static void gfs2_free_dead_glocks(struct gfs2_sbd *sdp)
  142. {
  143. struct list_head *list = &sdp->sd_dead_glocks;
  144. while(!list_empty(list)) {
  145. struct gfs2_glock *gl;
  146. gl = list_first_entry(list, struct gfs2_glock, gl_lru);
  147. list_del_init(&gl->gl_lru);
  148. __gfs2_glock_free(gl);
  149. }
  150. }
  151. /**
  152. * gfs2_glock_hold() - increment reference count on glock
  153. * @gl: The glock to hold
  154. *
  155. */
  156. struct gfs2_glock *gfs2_glock_hold(struct gfs2_glock *gl)
  157. {
  158. if (!lockref_get_not_dead(&gl->gl_lockref))
  159. GLOCK_BUG_ON(gl, 1);
  160. return gl;
  161. }
  162. static void gfs2_glock_add_to_lru(struct gfs2_glock *gl)
  163. {
  164. spin_lock(&lru_lock);
  165. list_move_tail(&gl->gl_lru, &lru_list);
  166. if (!test_bit(GLF_LRU, &gl->gl_flags)) {
  167. set_bit(GLF_LRU, &gl->gl_flags);
  168. atomic_inc(&lru_count);
  169. }
  170. spin_unlock(&lru_lock);
  171. }
  172. static void gfs2_glock_remove_from_lru(struct gfs2_glock *gl)
  173. {
  174. spin_lock(&lru_lock);
  175. if (test_bit(GLF_LRU, &gl->gl_flags)) {
  176. list_del_init(&gl->gl_lru);
  177. atomic_dec(&lru_count);
  178. clear_bit(GLF_LRU, &gl->gl_flags);
  179. }
  180. spin_unlock(&lru_lock);
  181. }
  182. /*
  183. * Enqueue the glock on the work queue. Passes one glock reference on to the
  184. * work queue.
  185. */
  186. static void gfs2_glock_queue_work(struct gfs2_glock *gl, unsigned long delay) {
  187. struct gfs2_sbd *sdp = glock_sbd(gl);
  188. if (!queue_delayed_work(sdp->sd_glock_wq, &gl->gl_work, delay)) {
  189. /*
  190. * We are holding the lockref spinlock, and the work was still
  191. * queued above. The queued work (glock_work_func) takes that
  192. * spinlock before dropping its glock reference(s), so it
  193. * cannot have dropped them in the meantime.
  194. */
  195. GLOCK_BUG_ON(gl, gl->gl_lockref.count < 2);
  196. gl->gl_lockref.count--;
  197. }
  198. }
  199. static void __gfs2_glock_put(struct gfs2_glock *gl)
  200. {
  201. struct gfs2_sbd *sdp = glock_sbd(gl);
  202. struct address_space *mapping = gfs2_glock2aspace(gl);
  203. lockref_mark_dead(&gl->gl_lockref);
  204. spin_unlock(&gl->gl_lockref.lock);
  205. gfs2_glock_remove_from_lru(gl);
  206. GLOCK_BUG_ON(gl, !list_empty(&gl->gl_holders));
  207. if (mapping) {
  208. truncate_inode_pages_final(mapping);
  209. if (!gfs2_withdrawn(sdp))
  210. GLOCK_BUG_ON(gl, !mapping_empty(mapping));
  211. }
  212. trace_gfs2_glock_put(gl);
  213. sdp->sd_lockstruct.ls_ops->lm_put_lock(gl);
  214. }
  215. static bool __gfs2_glock_put_or_lock(struct gfs2_glock *gl)
  216. {
  217. if (lockref_put_or_lock(&gl->gl_lockref))
  218. return true;
  219. GLOCK_BUG_ON(gl, gl->gl_lockref.count != 1);
  220. if (gl->gl_state != LM_ST_UNLOCKED) {
  221. gl->gl_lockref.count--;
  222. gfs2_glock_add_to_lru(gl);
  223. spin_unlock(&gl->gl_lockref.lock);
  224. return true;
  225. }
  226. return false;
  227. }
  228. /**
  229. * gfs2_glock_put() - Decrement reference count on glock
  230. * @gl: The glock to put
  231. *
  232. */
  233. void gfs2_glock_put(struct gfs2_glock *gl)
  234. {
  235. if (__gfs2_glock_put_or_lock(gl))
  236. return;
  237. __gfs2_glock_put(gl);
  238. }
  239. /*
  240. * gfs2_glock_put_async - Decrement reference count without sleeping
  241. * @gl: The glock to put
  242. *
  243. * Decrement the reference count on glock immediately unless it is the last
  244. * reference. Defer putting the last reference to work queue context.
  245. */
  246. void gfs2_glock_put_async(struct gfs2_glock *gl)
  247. {
  248. if (__gfs2_glock_put_or_lock(gl))
  249. return;
  250. gfs2_glock_queue_work(gl, 0);
  251. spin_unlock(&gl->gl_lockref.lock);
  252. }
  253. /**
  254. * may_grant - check if it's ok to grant a new lock
  255. * @gl: The glock
  256. * @current_gh: One of the current holders of @gl
  257. * @gh: The lock request which we wish to grant
  258. *
  259. * With our current compatibility rules, if a glock has one or more active
  260. * holders (HIF_HOLDER flag set), any of those holders can be passed in as
  261. * @current_gh; they are all the same as far as compatibility with the new @gh
  262. * goes.
  263. *
  264. * Returns true if it's ok to grant the lock.
  265. */
  266. static inline bool may_grant(struct gfs2_glock *gl,
  267. struct gfs2_holder *current_gh,
  268. struct gfs2_holder *gh)
  269. {
  270. if (current_gh) {
  271. GLOCK_BUG_ON(gl, !test_bit(HIF_HOLDER, &current_gh->gh_iflags));
  272. switch(current_gh->gh_state) {
  273. case LM_ST_EXCLUSIVE:
  274. /*
  275. * Here we make a special exception to grant holders
  276. * who agree to share the EX lock with other holders
  277. * who also have the bit set. If the original holder
  278. * has the LM_FLAG_NODE_SCOPE bit set, we grant more
  279. * holders with the bit set.
  280. */
  281. return gh->gh_state == LM_ST_EXCLUSIVE &&
  282. (current_gh->gh_flags & LM_FLAG_NODE_SCOPE) &&
  283. (gh->gh_flags & LM_FLAG_NODE_SCOPE);
  284. case LM_ST_SHARED:
  285. case LM_ST_DEFERRED:
  286. return gh->gh_state == current_gh->gh_state;
  287. default:
  288. return false;
  289. }
  290. }
  291. if (gl->gl_state == gh->gh_state)
  292. return true;
  293. if (gh->gh_flags & GL_EXACT)
  294. return false;
  295. if (gl->gl_state == LM_ST_EXCLUSIVE) {
  296. return gh->gh_state == LM_ST_SHARED ||
  297. gh->gh_state == LM_ST_DEFERRED;
  298. }
  299. if (gh->gh_flags & LM_FLAG_ANY)
  300. return gl->gl_state != LM_ST_UNLOCKED;
  301. return false;
  302. }
  303. static void gfs2_holder_wake(struct gfs2_holder *gh)
  304. {
  305. clear_bit(HIF_WAIT, &gh->gh_iflags);
  306. smp_mb__after_atomic();
  307. wake_up_bit(&gh->gh_iflags, HIF_WAIT);
  308. if (gh->gh_flags & GL_ASYNC) {
  309. struct gfs2_sbd *sdp = glock_sbd(gh->gh_gl);
  310. wake_up(&sdp->sd_async_glock_wait);
  311. }
  312. }
  313. /**
  314. * do_error - Something unexpected has happened during a lock request
  315. * @gl: The glock
  316. * @ret: The status from the DLM
  317. */
  318. static void do_error(struct gfs2_glock *gl, const int ret)
  319. {
  320. struct gfs2_holder *gh, *tmp;
  321. list_for_each_entry_safe(gh, tmp, &gl->gl_holders, gh_list) {
  322. if (test_bit(HIF_HOLDER, &gh->gh_iflags))
  323. continue;
  324. if (ret & LM_OUT_ERROR)
  325. gh->gh_error = -EIO;
  326. else if (gh->gh_flags & (LM_FLAG_TRY | LM_FLAG_TRY_1CB))
  327. gh->gh_error = GLR_TRYFAILED;
  328. else
  329. continue;
  330. list_del_init(&gh->gh_list);
  331. trace_gfs2_glock_queue(gh, 0);
  332. gfs2_holder_wake(gh);
  333. }
  334. }
  335. /**
  336. * find_first_holder - find the first "holder" gh
  337. * @gl: the glock
  338. */
  339. static inline struct gfs2_holder *find_first_holder(const struct gfs2_glock *gl)
  340. {
  341. struct gfs2_holder *gh;
  342. if (!list_empty(&gl->gl_holders)) {
  343. gh = list_first_entry(&gl->gl_holders, struct gfs2_holder,
  344. gh_list);
  345. if (test_bit(HIF_HOLDER, &gh->gh_iflags))
  346. return gh;
  347. }
  348. return NULL;
  349. }
  350. /*
  351. * gfs2_instantiate - Call the glops instantiate function
  352. * @gh: The glock holder
  353. *
  354. * Returns: 0 if instantiate was successful, or error.
  355. */
  356. int gfs2_instantiate(struct gfs2_holder *gh)
  357. {
  358. struct gfs2_glock *gl = gh->gh_gl;
  359. const struct gfs2_glock_operations *glops = gl->gl_ops;
  360. int ret;
  361. again:
  362. if (!test_bit(GLF_INSTANTIATE_NEEDED, &gl->gl_flags))
  363. goto done;
  364. /*
  365. * Since we unlock the lockref lock, we set a flag to indicate
  366. * instantiate is in progress.
  367. */
  368. if (test_and_set_bit(GLF_INSTANTIATE_IN_PROG, &gl->gl_flags)) {
  369. wait_on_bit(&gl->gl_flags, GLF_INSTANTIATE_IN_PROG,
  370. TASK_UNINTERRUPTIBLE);
  371. /*
  372. * Here we just waited for a different instantiate to finish.
  373. * But that may not have been successful, as when a process
  374. * locks an inode glock _before_ it has an actual inode to
  375. * instantiate into. So we check again. This process might
  376. * have an inode to instantiate, so might be successful.
  377. */
  378. goto again;
  379. }
  380. ret = glops->go_instantiate(gl);
  381. if (!ret)
  382. clear_bit(GLF_INSTANTIATE_NEEDED, &gl->gl_flags);
  383. clear_and_wake_up_bit(GLF_INSTANTIATE_IN_PROG, &gl->gl_flags);
  384. if (ret)
  385. return ret;
  386. done:
  387. if (glops->go_held)
  388. return glops->go_held(gh);
  389. return 0;
  390. }
  391. /**
  392. * do_promote - promote as many requests as possible on the current queue
  393. * @gl: The glock
  394. */
  395. static void do_promote(struct gfs2_glock *gl)
  396. {
  397. struct gfs2_sbd *sdp = glock_sbd(gl);
  398. struct gfs2_holder *gh, *current_gh;
  399. if (gfs2_withdrawn(sdp)) {
  400. do_error(gl, LM_OUT_ERROR);
  401. return;
  402. }
  403. current_gh = find_first_holder(gl);
  404. list_for_each_entry(gh, &gl->gl_holders, gh_list) {
  405. if (test_bit(HIF_HOLDER, &gh->gh_iflags))
  406. continue;
  407. if (!may_grant(gl, current_gh, gh)) {
  408. /*
  409. * If we get here, it means we may not grant this
  410. * holder for some reason.
  411. */
  412. if (current_gh)
  413. do_error(gl, 0); /* Fail queued try locks */
  414. break;
  415. }
  416. set_bit(HIF_HOLDER, &gh->gh_iflags);
  417. trace_gfs2_promote(gh);
  418. gfs2_holder_wake(gh);
  419. if (!current_gh)
  420. current_gh = gh;
  421. }
  422. }
  423. /**
  424. * find_first_waiter - find the first gh that's waiting for the glock
  425. * @gl: the glock
  426. */
  427. static inline struct gfs2_holder *find_first_waiter(const struct gfs2_glock *gl)
  428. {
  429. struct gfs2_holder *gh;
  430. list_for_each_entry(gh, &gl->gl_holders, gh_list) {
  431. if (!test_bit(HIF_HOLDER, &gh->gh_iflags))
  432. return gh;
  433. }
  434. return NULL;
  435. }
  436. /**
  437. * find_last_waiter - find the last gh that's waiting for the glock
  438. * @gl: the glock
  439. *
  440. * This also is a fast way of finding out if there are any waiters.
  441. */
  442. static inline struct gfs2_holder *find_last_waiter(const struct gfs2_glock *gl)
  443. {
  444. struct gfs2_holder *gh;
  445. if (list_empty(&gl->gl_holders))
  446. return NULL;
  447. gh = list_last_entry(&gl->gl_holders, struct gfs2_holder, gh_list);
  448. return test_bit(HIF_HOLDER, &gh->gh_iflags) ? NULL : gh;
  449. }
  450. /**
  451. * state_change - record that the glock is now in a different state
  452. * @gl: the glock
  453. * @new_state: the new state
  454. */
  455. static void state_change(struct gfs2_glock *gl, unsigned int new_state)
  456. {
  457. if (new_state != gl->gl_target)
  458. /* shorten our minimum hold time */
  459. gl->gl_hold_time = max(gl->gl_hold_time - GL_GLOCK_HOLD_DECR,
  460. GL_GLOCK_MIN_HOLD);
  461. gl->gl_state = new_state;
  462. gl->gl_tchange = jiffies;
  463. }
  464. static void gfs2_set_demote(int nr, struct gfs2_glock *gl)
  465. {
  466. struct gfs2_sbd *sdp = glock_sbd(gl);
  467. set_bit(nr, &gl->gl_flags);
  468. smp_mb();
  469. wake_up(&sdp->sd_async_glock_wait);
  470. }
  471. static void gfs2_demote_wake(struct gfs2_glock *gl)
  472. {
  473. gl->gl_demote_state = LM_ST_EXCLUSIVE;
  474. clear_bit(GLF_DEMOTE, &gl->gl_flags);
  475. smp_mb__after_atomic();
  476. wake_up_bit(&gl->gl_flags, GLF_DEMOTE);
  477. }
  478. /**
  479. * finish_xmote - The DLM has replied to one of our lock requests
  480. * @gl: The glock
  481. * @ret: The status from the DLM
  482. *
  483. */
  484. static void finish_xmote(struct gfs2_glock *gl, unsigned int ret)
  485. {
  486. const struct gfs2_glock_operations *glops = gl->gl_ops;
  487. if (!(ret & ~LM_OUT_ST_MASK)) {
  488. unsigned state = ret & LM_OUT_ST_MASK;
  489. trace_gfs2_glock_state_change(gl, state);
  490. state_change(gl, state);
  491. }
  492. /* Demote to UN request arrived during demote to SH or DF */
  493. if (test_bit(GLF_DEMOTE_IN_PROGRESS, &gl->gl_flags) &&
  494. gl->gl_state != LM_ST_UNLOCKED &&
  495. gl->gl_demote_state == LM_ST_UNLOCKED)
  496. gl->gl_target = LM_ST_UNLOCKED;
  497. /* Check for state != intended state */
  498. if (unlikely(gl->gl_state != gl->gl_target)) {
  499. struct gfs2_holder *gh = find_first_waiter(gl);
  500. if (gh && !test_bit(GLF_DEMOTE_IN_PROGRESS, &gl->gl_flags)) {
  501. if (ret & LM_OUT_CANCELED) {
  502. list_del_init(&gh->gh_list);
  503. trace_gfs2_glock_queue(gh, 0);
  504. gfs2_holder_wake(gh);
  505. gl->gl_target = gl->gl_state;
  506. goto out;
  507. }
  508. /* Some error or failed "try lock" - report it */
  509. if ((ret & LM_OUT_ERROR) ||
  510. (gh->gh_flags & (LM_FLAG_TRY | LM_FLAG_TRY_1CB))) {
  511. gl->gl_target = gl->gl_state;
  512. do_error(gl, ret);
  513. goto out;
  514. }
  515. }
  516. switch(gl->gl_state) {
  517. /* Unlocked due to conversion deadlock, try again */
  518. case LM_ST_UNLOCKED:
  519. do_xmote(gl, gh, gl->gl_target,
  520. !test_bit(GLF_DEMOTE_IN_PROGRESS,
  521. &gl->gl_flags));
  522. break;
  523. /* Conversion fails, unlock and try again */
  524. case LM_ST_SHARED:
  525. case LM_ST_DEFERRED:
  526. do_xmote(gl, gh, LM_ST_UNLOCKED, false);
  527. break;
  528. default: /* Everything else */
  529. fs_err(glock_sbd(gl),
  530. "glock %u:%llu requested=%u ret=%u\n",
  531. glock_type(gl), glock_number(gl),
  532. gl->gl_req, ret);
  533. GLOCK_BUG_ON(gl, 1);
  534. }
  535. return;
  536. }
  537. /* Fast path - we got what we asked for */
  538. if (test_bit(GLF_DEMOTE_IN_PROGRESS, &gl->gl_flags)) {
  539. clear_bit(GLF_DEMOTE_IN_PROGRESS, &gl->gl_flags);
  540. gfs2_demote_wake(gl);
  541. }
  542. if (gl->gl_state != LM_ST_UNLOCKED) {
  543. if (glops->go_xmote_bh) {
  544. int rv;
  545. spin_unlock(&gl->gl_lockref.lock);
  546. rv = glops->go_xmote_bh(gl);
  547. spin_lock(&gl->gl_lockref.lock);
  548. if (rv) {
  549. do_error(gl, rv);
  550. goto out;
  551. }
  552. }
  553. do_promote(gl);
  554. }
  555. out:
  556. if (!test_bit(GLF_CANCELING, &gl->gl_flags))
  557. clear_and_wake_up_bit(GLF_LOCK, &gl->gl_flags);
  558. }
  559. /**
  560. * do_xmote - Calls the DLM to change the state of a lock
  561. * @gl: The lock state
  562. * @gh: The holder (only for promotes)
  563. * @target: The target lock state
  564. * @may_cancel: Operation may be canceled
  565. *
  566. */
  567. static void do_xmote(struct gfs2_glock *gl, struct gfs2_holder *gh,
  568. unsigned int target, bool may_cancel)
  569. __releases(&gl->gl_lockref.lock)
  570. __acquires(&gl->gl_lockref.lock)
  571. {
  572. const struct gfs2_glock_operations *glops = gl->gl_ops;
  573. struct gfs2_sbd *sdp = glock_sbd(gl);
  574. struct lm_lockstruct *ls = &sdp->sd_lockstruct;
  575. int ret;
  576. /*
  577. * When a filesystem is withdrawing, the remaining cluster nodes will
  578. * take care of recovering the withdrawing node's journal. We only
  579. * need to make sure that once we trigger remote recovery, we won't
  580. * write to the shared block device anymore. This means that here,
  581. *
  582. * - no new writes to the filesystem must be triggered (->go_sync()).
  583. *
  584. * - any cached data should be discarded by calling ->go_inval(), dirty
  585. * or not and journaled or unjournaled.
  586. *
  587. * - no more dlm locking operations should be issued (->lm_lock()).
  588. */
  589. GLOCK_BUG_ON(gl, gl->gl_state == target);
  590. GLOCK_BUG_ON(gl, gl->gl_state == gl->gl_target);
  591. if (!glops->go_inval || !glops->go_sync)
  592. goto skip_inval;
  593. spin_unlock(&gl->gl_lockref.lock);
  594. if (!gfs2_withdrawn(sdp)) {
  595. ret = glops->go_sync(gl);
  596. if (ret) {
  597. if (cmpxchg(&sdp->sd_log_error, 0, ret)) {
  598. fs_err(sdp, "Error %d syncing glock\n", ret);
  599. gfs2_dump_glock(NULL, gl, true);
  600. gfs2_withdraw(sdp);
  601. }
  602. }
  603. }
  604. if (target == LM_ST_UNLOCKED || target == LM_ST_DEFERRED)
  605. glops->go_inval(gl, target == LM_ST_DEFERRED ? 0 : DIO_METADATA);
  606. spin_lock(&gl->gl_lockref.lock);
  607. skip_inval:
  608. if (gfs2_withdrawn(sdp)) {
  609. if (target != LM_ST_UNLOCKED)
  610. target = LM_OUT_ERROR;
  611. goto out;
  612. }
  613. if (ls->ls_ops->lm_lock) {
  614. spin_unlock(&gl->gl_lockref.lock);
  615. ret = ls->ls_ops->lm_lock(gl, target, gh ? gh->gh_flags : 0);
  616. spin_lock(&gl->gl_lockref.lock);
  617. if (!ret) {
  618. if (may_cancel) {
  619. set_bit(GLF_MAY_CANCEL, &gl->gl_flags);
  620. smp_mb__after_atomic();
  621. wake_up_bit(&gl->gl_flags, GLF_LOCK);
  622. }
  623. /* The operation will be completed asynchronously. */
  624. gl->gl_lockref.count++;
  625. return;
  626. }
  627. if (ret == -ENODEV) {
  628. /*
  629. * The lockspace has been released and the lock has
  630. * been unlocked implicitly.
  631. */
  632. if (target != LM_ST_UNLOCKED) {
  633. target = LM_OUT_ERROR;
  634. goto out;
  635. }
  636. } else {
  637. fs_err(sdp, "lm_lock ret %d\n", ret);
  638. GLOCK_BUG_ON(gl, !gfs2_withdrawn(sdp));
  639. return;
  640. }
  641. }
  642. out:
  643. /* Complete the operation now. */
  644. finish_xmote(gl, target);
  645. gl->gl_lockref.count++;
  646. gfs2_glock_queue_work(gl, 0);
  647. }
  648. /**
  649. * run_queue - do all outstanding tasks related to a glock
  650. * @gl: The glock in question
  651. * @nonblock: True if we must not block in run_queue
  652. *
  653. */
  654. static void run_queue(struct gfs2_glock *gl, const int nonblock)
  655. __releases(&gl->gl_lockref.lock)
  656. __acquires(&gl->gl_lockref.lock)
  657. {
  658. struct gfs2_holder *gh;
  659. if (test_bit(GLF_LOCK, &gl->gl_flags))
  660. return;
  661. /*
  662. * The GLF_DEMOTE_IN_PROGRESS flag must only be set when the GLF_LOCK
  663. * flag is set as well.
  664. */
  665. GLOCK_BUG_ON(gl, test_bit(GLF_DEMOTE_IN_PROGRESS, &gl->gl_flags));
  666. if (test_bit(GLF_DEMOTE, &gl->gl_flags)) {
  667. if (gl->gl_demote_state == gl->gl_state) {
  668. gfs2_demote_wake(gl);
  669. goto promote;
  670. }
  671. if (find_first_holder(gl))
  672. return;
  673. if (nonblock)
  674. goto out_sched;
  675. set_bit(GLF_DEMOTE_IN_PROGRESS, &gl->gl_flags);
  676. GLOCK_BUG_ON(gl, gl->gl_demote_state == LM_ST_EXCLUSIVE);
  677. gl->gl_target = gl->gl_demote_state;
  678. set_bit(GLF_LOCK, &gl->gl_flags);
  679. do_xmote(gl, NULL, gl->gl_target, false);
  680. return;
  681. }
  682. promote:
  683. do_promote(gl);
  684. if (find_first_holder(gl))
  685. return;
  686. gh = find_first_waiter(gl);
  687. if (!gh)
  688. return;
  689. if (nonblock)
  690. goto out_sched;
  691. gl->gl_target = gh->gh_state;
  692. if (!(gh->gh_flags & (LM_FLAG_TRY | LM_FLAG_TRY_1CB)))
  693. do_error(gl, 0); /* Fail queued try locks */
  694. set_bit(GLF_LOCK, &gl->gl_flags);
  695. do_xmote(gl, gh, gl->gl_target, true);
  696. return;
  697. out_sched:
  698. gl->gl_lockref.count++;
  699. gfs2_glock_queue_work(gl, 0);
  700. }
  701. /**
  702. * glock_set_object - set the gl_object field of a glock
  703. * @gl: the glock
  704. * @object: the object
  705. */
  706. void glock_set_object(struct gfs2_glock *gl, void *object)
  707. {
  708. void *prev_object;
  709. spin_lock(&gl->gl_lockref.lock);
  710. prev_object = gl->gl_object;
  711. gl->gl_object = object;
  712. spin_unlock(&gl->gl_lockref.lock);
  713. if (gfs2_assert_warn(glock_sbd(gl), prev_object == NULL))
  714. gfs2_dump_glock(NULL, gl, true);
  715. }
  716. /**
  717. * glock_clear_object - clear the gl_object field of a glock
  718. * @gl: the glock
  719. * @object: object the glock currently points at
  720. */
  721. void glock_clear_object(struct gfs2_glock *gl, void *object)
  722. {
  723. void *prev_object;
  724. spin_lock(&gl->gl_lockref.lock);
  725. prev_object = gl->gl_object;
  726. gl->gl_object = NULL;
  727. spin_unlock(&gl->gl_lockref.lock);
  728. if (gfs2_assert_warn(glock_sbd(gl), prev_object == object))
  729. gfs2_dump_glock(NULL, gl, true);
  730. }
  731. void gfs2_inode_remember_delete(struct gfs2_glock *gl, u64 generation)
  732. {
  733. struct gfs2_inode_lvb *ri = (void *)gl->gl_lksb.sb_lvbptr;
  734. if (ri->ri_magic == 0)
  735. ri->ri_magic = cpu_to_be32(GFS2_MAGIC);
  736. if (ri->ri_magic == cpu_to_be32(GFS2_MAGIC))
  737. ri->ri_generation_deleted = cpu_to_be64(generation);
  738. }
  739. bool gfs2_inode_already_deleted(struct gfs2_glock *gl, u64 generation)
  740. {
  741. struct gfs2_inode_lvb *ri = (void *)gl->gl_lksb.sb_lvbptr;
  742. if (ri->ri_magic != cpu_to_be32(GFS2_MAGIC))
  743. return false;
  744. return generation <= be64_to_cpu(ri->ri_generation_deleted);
  745. }
  746. static void gfs2_glock_poke(struct gfs2_glock *gl)
  747. {
  748. int flags = LM_FLAG_TRY_1CB | LM_FLAG_ANY | GL_SKIP;
  749. struct gfs2_holder gh;
  750. int error;
  751. __gfs2_holder_init(gl, LM_ST_SHARED, flags, &gh, _RET_IP_);
  752. error = gfs2_glock_nq(&gh);
  753. if (!error)
  754. gfs2_glock_dq(&gh);
  755. gfs2_holder_uninit(&gh);
  756. }
  757. static struct gfs2_inode *gfs2_grab_existing_inode(struct gfs2_glock *gl)
  758. {
  759. struct gfs2_inode *ip;
  760. spin_lock(&gl->gl_lockref.lock);
  761. ip = gl->gl_object;
  762. if (ip && !igrab(&ip->i_inode))
  763. ip = NULL;
  764. spin_unlock(&gl->gl_lockref.lock);
  765. if (ip) {
  766. wait_on_new_inode(&ip->i_inode);
  767. if (is_bad_inode(&ip->i_inode)) {
  768. iput(&ip->i_inode);
  769. ip = NULL;
  770. }
  771. }
  772. return ip;
  773. }
  774. static void gfs2_try_to_evict(struct gfs2_glock *gl)
  775. {
  776. struct gfs2_inode *ip;
  777. /*
  778. * If there is contention on the iopen glock and we have an inode, try
  779. * to grab and release the inode so that it can be evicted. The
  780. * GLF_DEFER_DELETE flag indicates to gfs2_evict_inode() that the inode
  781. * should not be deleted locally. This will allow the remote node to
  782. * go ahead and delete the inode without us having to do it, which will
  783. * avoid rgrp glock thrashing.
  784. *
  785. * The remote node is likely still holding the corresponding inode
  786. * glock, so it will run before we get to verify that the delete has
  787. * happened below. (Verification is triggered by the call to
  788. * gfs2_queue_verify_delete() in gfs2_evict_inode().)
  789. */
  790. ip = gfs2_grab_existing_inode(gl);
  791. if (ip) {
  792. set_bit(GLF_DEFER_DELETE, &gl->gl_flags);
  793. d_prune_aliases(&ip->i_inode);
  794. iput(&ip->i_inode);
  795. clear_bit(GLF_DEFER_DELETE, &gl->gl_flags);
  796. /* If the inode was evicted, gl->gl_object will now be NULL. */
  797. ip = gfs2_grab_existing_inode(gl);
  798. if (ip) {
  799. gfs2_glock_poke(ip->i_gl);
  800. iput(&ip->i_inode);
  801. }
  802. }
  803. }
  804. bool gfs2_queue_try_to_evict(struct gfs2_glock *gl)
  805. {
  806. struct gfs2_sbd *sdp = glock_sbd(gl);
  807. if (test_and_set_bit(GLF_TRY_TO_EVICT, &gl->gl_flags))
  808. return false;
  809. return !mod_delayed_work(sdp->sd_delete_wq, &gl->gl_delete, 0);
  810. }
  811. bool gfs2_queue_verify_delete(struct gfs2_glock *gl, bool later)
  812. {
  813. struct gfs2_sbd *sdp = glock_sbd(gl);
  814. unsigned long delay;
  815. if (test_and_set_bit(GLF_VERIFY_DELETE, &gl->gl_flags))
  816. return false;
  817. delay = later ? HZ + get_random_long() % (HZ * 9) : 0;
  818. return queue_delayed_work(sdp->sd_delete_wq, &gl->gl_delete, delay);
  819. }
  820. static void delete_work_func(struct work_struct *work)
  821. {
  822. struct delayed_work *dwork = to_delayed_work(work);
  823. struct gfs2_glock *gl = container_of(dwork, struct gfs2_glock, gl_delete);
  824. struct gfs2_sbd *sdp = glock_sbd(gl);
  825. bool verify_delete = test_and_clear_bit(GLF_VERIFY_DELETE, &gl->gl_flags);
  826. /*
  827. * Check for the GLF_VERIFY_DELETE above: this ensures that we won't
  828. * immediately process GLF_VERIFY_DELETE work that the below call to
  829. * gfs2_try_to_evict() queues.
  830. */
  831. if (test_and_clear_bit(GLF_TRY_TO_EVICT, &gl->gl_flags))
  832. gfs2_try_to_evict(gl);
  833. if (verify_delete) {
  834. u64 no_addr = glock_number(gl);
  835. struct inode *inode;
  836. inode = gfs2_lookup_by_inum(sdp, no_addr, gl->gl_no_formal_ino,
  837. GFS2_BLKST_UNLINKED);
  838. if (IS_ERR(inode)) {
  839. if (PTR_ERR(inode) == -EAGAIN &&
  840. !test_bit(SDF_KILL, &sdp->sd_flags) &&
  841. gfs2_queue_verify_delete(gl, true))
  842. return;
  843. } else {
  844. d_prune_aliases(inode);
  845. iput(inode);
  846. }
  847. }
  848. gfs2_glock_put(gl);
  849. }
  850. static void glock_work_func(struct work_struct *work)
  851. {
  852. unsigned long delay = 0;
  853. struct gfs2_glock *gl = container_of(work, struct gfs2_glock, gl_work.work);
  854. unsigned int drop_refs = 1;
  855. spin_lock(&gl->gl_lockref.lock);
  856. if (test_bit(GLF_HAVE_REPLY, &gl->gl_flags)) {
  857. clear_bit(GLF_HAVE_REPLY, &gl->gl_flags);
  858. finish_xmote(gl, gl->gl_reply);
  859. drop_refs++;
  860. }
  861. if (test_bit(GLF_PENDING_DEMOTE, &gl->gl_flags) &&
  862. gl->gl_state != LM_ST_UNLOCKED &&
  863. gl->gl_demote_state != LM_ST_EXCLUSIVE) {
  864. if (glock_type(gl) == LM_TYPE_INODE) {
  865. unsigned long holdtime, now = jiffies;
  866. holdtime = gl->gl_tchange + gl->gl_hold_time;
  867. if (time_before(now, holdtime))
  868. delay = holdtime - now;
  869. }
  870. if (!delay) {
  871. clear_bit(GLF_PENDING_DEMOTE, &gl->gl_flags);
  872. gfs2_set_demote(GLF_DEMOTE, gl);
  873. }
  874. }
  875. run_queue(gl, 0);
  876. if (delay) {
  877. /* Keep one glock reference for the work we requeue. */
  878. drop_refs--;
  879. gfs2_glock_queue_work(gl, delay);
  880. }
  881. /* Drop the remaining glock references manually. */
  882. GLOCK_BUG_ON(gl, gl->gl_lockref.count < drop_refs);
  883. gl->gl_lockref.count -= drop_refs;
  884. if (!gl->gl_lockref.count) {
  885. if (gl->gl_state == LM_ST_UNLOCKED) {
  886. __gfs2_glock_put(gl);
  887. return;
  888. }
  889. gfs2_glock_add_to_lru(gl);
  890. }
  891. spin_unlock(&gl->gl_lockref.lock);
  892. }
  893. static struct gfs2_glock *find_insert_glock(struct lm_lockname *name,
  894. struct gfs2_glock *new)
  895. {
  896. struct wait_glock_queue wait;
  897. wait_queue_head_t *wq = glock_waitqueue(name);
  898. struct gfs2_glock *gl;
  899. wait.name = name;
  900. init_wait(&wait.wait);
  901. wait.wait.func = glock_wake_function;
  902. again:
  903. prepare_to_wait(wq, &wait.wait, TASK_UNINTERRUPTIBLE);
  904. rcu_read_lock();
  905. if (new) {
  906. gl = rhashtable_lookup_get_insert_fast(&gl_hash_table,
  907. &new->gl_node, ht_parms);
  908. if (IS_ERR(gl))
  909. goto out;
  910. } else {
  911. gl = rhashtable_lookup_fast(&gl_hash_table,
  912. name, ht_parms);
  913. }
  914. if (gl && !lockref_get_not_dead(&gl->gl_lockref)) {
  915. rcu_read_unlock();
  916. schedule();
  917. goto again;
  918. }
  919. out:
  920. rcu_read_unlock();
  921. finish_wait(wq, &wait.wait);
  922. if (gl)
  923. gfs2_glock_remove_from_lru(gl);
  924. return gl;
  925. }
  926. /**
  927. * gfs2_glock_get() - Get a glock, or create one if one doesn't exist
  928. * @sdp: The GFS2 superblock
  929. * @number: the lock number
  930. * @glops: The glock_operations to use
  931. * @create: If 0, don't create the glock if it doesn't exist
  932. * @glp: the glock is returned here
  933. *
  934. * This does not lock a glock, just finds/creates structures for one.
  935. *
  936. * Returns: errno
  937. */
  938. int gfs2_glock_get(struct gfs2_sbd *sdp, u64 number,
  939. const struct gfs2_glock_operations *glops, int create,
  940. struct gfs2_glock **glp)
  941. {
  942. struct lm_lockname name = { .ln_number = number,
  943. .ln_type = glops->go_type,
  944. .ln_sbd = sdp };
  945. struct gfs2_glock *gl, *tmp;
  946. struct address_space *mapping;
  947. gl = find_insert_glock(&name, NULL);
  948. if (gl)
  949. goto found;
  950. if (!create)
  951. return -ENOENT;
  952. if (glops->go_flags & GLOF_ASPACE) {
  953. struct gfs2_glock_aspace *gla =
  954. kmem_cache_alloc(gfs2_glock_aspace_cachep, GFP_NOFS);
  955. if (!gla)
  956. return -ENOMEM;
  957. gl = &gla->glock;
  958. } else {
  959. gl = kmem_cache_alloc(gfs2_glock_cachep, GFP_NOFS);
  960. if (!gl)
  961. return -ENOMEM;
  962. }
  963. memset(&gl->gl_lksb, 0, sizeof(struct dlm_lksb));
  964. gl->gl_ops = glops;
  965. if (glops->go_flags & GLOF_LVB) {
  966. gl->gl_lksb.sb_lvbptr = kzalloc(GDLM_LVB_SIZE, GFP_NOFS);
  967. if (!gl->gl_lksb.sb_lvbptr) {
  968. gfs2_glock_dealloc(&gl->gl_rcu);
  969. return -ENOMEM;
  970. }
  971. }
  972. atomic_inc(&sdp->sd_glock_disposal);
  973. gl->gl_node.next = NULL;
  974. gl->gl_flags = BIT(GLF_INITIAL);
  975. if (glops->go_instantiate)
  976. gl->gl_flags |= BIT(GLF_INSTANTIATE_NEEDED);
  977. gl->gl_name = name;
  978. lockref_init(&gl->gl_lockref);
  979. lockdep_set_subclass(&gl->gl_lockref.lock, glops->go_subclass);
  980. gl->gl_state = LM_ST_UNLOCKED;
  981. gl->gl_target = LM_ST_UNLOCKED;
  982. gl->gl_demote_state = LM_ST_EXCLUSIVE;
  983. gl->gl_dstamp = 0;
  984. preempt_disable();
  985. /* We use the global stats to estimate the initial per-glock stats */
  986. gl->gl_stats = this_cpu_ptr(sdp->sd_lkstats)->lkstats[glops->go_type];
  987. preempt_enable();
  988. gl->gl_stats.stats[GFS2_LKS_DCOUNT] = 0;
  989. gl->gl_stats.stats[GFS2_LKS_QCOUNT] = 0;
  990. gl->gl_tchange = jiffies;
  991. gl->gl_object = NULL;
  992. gl->gl_hold_time = GL_GLOCK_DFT_HOLD;
  993. INIT_DELAYED_WORK(&gl->gl_work, glock_work_func);
  994. if (glock_type(gl) == LM_TYPE_IOPEN)
  995. INIT_DELAYED_WORK(&gl->gl_delete, delete_work_func);
  996. mapping = gfs2_glock2aspace(gl);
  997. if (mapping) {
  998. gfp_t gfp_mask;
  999. mapping->a_ops = &gfs2_meta_aops;
  1000. mapping->host = sdp->sd_inode;
  1001. mapping->flags = 0;
  1002. gfp_mask = mapping_gfp_mask(sdp->sd_inode->i_mapping);
  1003. mapping_set_gfp_mask(mapping, gfp_mask);
  1004. mapping->i_private_data = NULL;
  1005. mapping->writeback_index = 0;
  1006. }
  1007. tmp = find_insert_glock(&name, gl);
  1008. if (tmp) {
  1009. gfs2_glock_dealloc(&gl->gl_rcu);
  1010. if (atomic_dec_and_test(&sdp->sd_glock_disposal))
  1011. wake_up(&sdp->sd_kill_wait);
  1012. if (IS_ERR(tmp))
  1013. return PTR_ERR(tmp);
  1014. gl = tmp;
  1015. }
  1016. found:
  1017. *glp = gl;
  1018. return 0;
  1019. }
  1020. /**
  1021. * __gfs2_holder_init - initialize a struct gfs2_holder in the default way
  1022. * @gl: the glock
  1023. * @state: the state we're requesting
  1024. * @flags: the modifier flags
  1025. * @gh: the holder structure
  1026. * @ip: caller's return address for debugging
  1027. */
  1028. void __gfs2_holder_init(struct gfs2_glock *gl, unsigned int state, u16 flags,
  1029. struct gfs2_holder *gh, unsigned long ip)
  1030. {
  1031. INIT_LIST_HEAD(&gh->gh_list);
  1032. gh->gh_gl = gfs2_glock_hold(gl);
  1033. gh->gh_ip = ip;
  1034. gh->gh_owner_pid = get_pid(task_pid(current));
  1035. gh->gh_state = state;
  1036. gh->gh_flags = flags;
  1037. gh->gh_iflags = 0;
  1038. }
  1039. /**
  1040. * gfs2_holder_reinit - reinitialize a struct gfs2_holder so we can requeue it
  1041. * @state: the state we're requesting
  1042. * @flags: the modifier flags
  1043. * @gh: the holder structure
  1044. *
  1045. * Don't mess with the glock.
  1046. *
  1047. */
  1048. void gfs2_holder_reinit(unsigned int state, u16 flags, struct gfs2_holder *gh)
  1049. {
  1050. gh->gh_state = state;
  1051. gh->gh_flags = flags;
  1052. gh->gh_iflags = 0;
  1053. gh->gh_ip = _RET_IP_;
  1054. put_pid(gh->gh_owner_pid);
  1055. gh->gh_owner_pid = get_pid(task_pid(current));
  1056. }
  1057. /**
  1058. * gfs2_holder_uninit - uninitialize a holder structure (drop glock reference)
  1059. * @gh: the holder structure
  1060. *
  1061. */
  1062. void gfs2_holder_uninit(struct gfs2_holder *gh)
  1063. {
  1064. put_pid(gh->gh_owner_pid);
  1065. gfs2_glock_put(gh->gh_gl);
  1066. gfs2_holder_mark_uninitialized(gh);
  1067. gh->gh_ip = 0;
  1068. }
  1069. static void gfs2_glock_update_hold_time(struct gfs2_glock *gl,
  1070. unsigned long start_time)
  1071. {
  1072. /* Have we waited longer that a second? */
  1073. if (time_after(jiffies, start_time + HZ)) {
  1074. /* Lengthen the minimum hold time. */
  1075. gl->gl_hold_time = min(gl->gl_hold_time + GL_GLOCK_HOLD_INCR,
  1076. GL_GLOCK_MAX_HOLD);
  1077. }
  1078. }
  1079. /**
  1080. * gfs2_glock_holder_ready - holder is ready and its error code can be collected
  1081. * @gh: the glock holder
  1082. *
  1083. * Called when a glock holder no longer needs to be waited for because it is
  1084. * now either held (HIF_HOLDER set; gh_error == 0), or acquiring the lock has
  1085. * failed (gh_error != 0).
  1086. */
  1087. int gfs2_glock_holder_ready(struct gfs2_holder *gh)
  1088. {
  1089. if (gh->gh_error || (gh->gh_flags & GL_SKIP))
  1090. return gh->gh_error;
  1091. gh->gh_error = gfs2_instantiate(gh);
  1092. if (gh->gh_error)
  1093. gfs2_glock_dq(gh);
  1094. return gh->gh_error;
  1095. }
  1096. /**
  1097. * gfs2_glock_wait - wait on a glock acquisition
  1098. * @gh: the glock holder
  1099. *
  1100. * Returns: 0 on success
  1101. */
  1102. int gfs2_glock_wait(struct gfs2_holder *gh)
  1103. {
  1104. unsigned long start_time = jiffies;
  1105. might_sleep();
  1106. wait_on_bit(&gh->gh_iflags, HIF_WAIT, TASK_UNINTERRUPTIBLE);
  1107. gfs2_glock_update_hold_time(gh->gh_gl, start_time);
  1108. return gfs2_glock_holder_ready(gh);
  1109. }
  1110. static int glocks_pending(unsigned int num_gh, struct gfs2_holder *ghs)
  1111. {
  1112. int i;
  1113. for (i = 0; i < num_gh; i++)
  1114. if (test_bit(HIF_WAIT, &ghs[i].gh_iflags))
  1115. return 1;
  1116. return 0;
  1117. }
  1118. /**
  1119. * gfs2_glock_async_wait - wait on multiple asynchronous glock acquisitions
  1120. * @num_gh: the number of holders in the array
  1121. * @ghs: the glock holder array
  1122. * @retries: number of retries attempted so far
  1123. *
  1124. * Returns: 0 on success, meaning all glocks have been granted and are held.
  1125. * -ESTALE if the request timed out, meaning all glocks were released,
  1126. * and the caller should retry the operation.
  1127. */
  1128. int gfs2_glock_async_wait(unsigned int num_gh, struct gfs2_holder *ghs,
  1129. unsigned int retries)
  1130. {
  1131. struct gfs2_sbd *sdp = glock_sbd(ghs[0].gh_gl);
  1132. unsigned long start_time = jiffies;
  1133. int i, ret = 0;
  1134. long timeout;
  1135. might_sleep();
  1136. timeout = GL_GLOCK_MIN_HOLD;
  1137. if (retries) {
  1138. unsigned int max_shift;
  1139. long incr;
  1140. /* Add a random delay and increase the timeout exponentially. */
  1141. max_shift = BITS_PER_LONG - 2 - __fls(GL_GLOCK_HOLD_INCR);
  1142. incr = min(GL_GLOCK_HOLD_INCR << min(retries - 1, max_shift),
  1143. 10 * HZ - GL_GLOCK_MIN_HOLD);
  1144. schedule_timeout_interruptible(get_random_long() % (incr / 3));
  1145. if (signal_pending(current))
  1146. goto interrupted;
  1147. timeout += (incr / 3) + get_random_long() % (incr / 3);
  1148. }
  1149. if (!wait_event_interruptible_timeout(sdp->sd_async_glock_wait,
  1150. !glocks_pending(num_gh, ghs), timeout)) {
  1151. ret = -ESTALE; /* request timed out. */
  1152. goto out;
  1153. }
  1154. if (signal_pending(current))
  1155. goto interrupted;
  1156. for (i = 0; i < num_gh; i++) {
  1157. struct gfs2_holder *gh = &ghs[i];
  1158. int ret2;
  1159. if (test_bit(HIF_HOLDER, &gh->gh_iflags)) {
  1160. gfs2_glock_update_hold_time(gh->gh_gl,
  1161. start_time);
  1162. }
  1163. ret2 = gfs2_glock_holder_ready(gh);
  1164. if (!ret)
  1165. ret = ret2;
  1166. }
  1167. out:
  1168. if (ret) {
  1169. for (i = 0; i < num_gh; i++) {
  1170. struct gfs2_holder *gh = &ghs[i];
  1171. gfs2_glock_dq(gh);
  1172. }
  1173. }
  1174. return ret;
  1175. interrupted:
  1176. ret = -EINTR;
  1177. goto out;
  1178. }
  1179. /**
  1180. * request_demote - process a demote request
  1181. * @gl: the glock
  1182. * @state: the state the caller wants us to change to
  1183. * @delay: zero to demote immediately; otherwise pending demote
  1184. * @remote: true if this came from a different cluster node
  1185. *
  1186. * There are only two requests that we are going to see in actual
  1187. * practise: LM_ST_SHARED and LM_ST_UNLOCKED
  1188. */
  1189. static void request_demote(struct gfs2_glock *gl, unsigned int state,
  1190. unsigned long delay, bool remote)
  1191. {
  1192. gfs2_set_demote(delay ? GLF_PENDING_DEMOTE : GLF_DEMOTE, gl);
  1193. if (gl->gl_demote_state == LM_ST_EXCLUSIVE) {
  1194. gl->gl_demote_state = state;
  1195. gl->gl_demote_time = jiffies;
  1196. } else if (gl->gl_demote_state != LM_ST_UNLOCKED &&
  1197. gl->gl_demote_state != state) {
  1198. gl->gl_demote_state = LM_ST_UNLOCKED;
  1199. }
  1200. if (gl->gl_ops->go_callback)
  1201. gl->gl_ops->go_callback(gl, remote);
  1202. trace_gfs2_demote_rq(gl, remote);
  1203. }
  1204. void gfs2_print_dbg(struct seq_file *seq, const char *fmt, ...)
  1205. {
  1206. struct va_format vaf;
  1207. va_list args;
  1208. va_start(args, fmt);
  1209. if (seq) {
  1210. seq_vprintf(seq, fmt, args);
  1211. } else {
  1212. vaf.fmt = fmt;
  1213. vaf.va = &args;
  1214. pr_err("%pV", &vaf);
  1215. }
  1216. va_end(args);
  1217. }
  1218. static bool gfs2_should_queue_trylock(struct gfs2_glock *gl,
  1219. struct gfs2_holder *gh)
  1220. {
  1221. struct gfs2_holder *current_gh, *gh2;
  1222. current_gh = find_first_holder(gl);
  1223. if (current_gh && !may_grant(gl, current_gh, gh))
  1224. return false;
  1225. list_for_each_entry(gh2, &gl->gl_holders, gh_list) {
  1226. if (test_bit(HIF_HOLDER, &gh2->gh_iflags))
  1227. continue;
  1228. if (!(gh2->gh_flags & (LM_FLAG_TRY | LM_FLAG_TRY_1CB)))
  1229. return false;
  1230. }
  1231. return true;
  1232. }
  1233. static inline bool pid_is_meaningful(const struct gfs2_holder *gh)
  1234. {
  1235. if (!(gh->gh_flags & GL_NOPID))
  1236. return true;
  1237. return !test_bit(HIF_HOLDER, &gh->gh_iflags);
  1238. }
  1239. /**
  1240. * add_to_queue - Add a holder to the wait queue (but look for recursion)
  1241. * @gh: the holder structure to add
  1242. *
  1243. * Eventually we should move the recursive locking trap to a
  1244. * debugging option or something like that. This is the fast
  1245. * path and needs to have the minimum number of distractions.
  1246. *
  1247. */
  1248. static inline void add_to_queue(struct gfs2_holder *gh)
  1249. {
  1250. struct gfs2_glock *gl = gh->gh_gl;
  1251. struct gfs2_sbd *sdp = glock_sbd(gl);
  1252. struct gfs2_holder *gh2;
  1253. GLOCK_BUG_ON(gl, gh->gh_owner_pid == NULL);
  1254. if (test_and_set_bit(HIF_WAIT, &gh->gh_iflags))
  1255. GLOCK_BUG_ON(gl, true);
  1256. if ((gh->gh_flags & (LM_FLAG_TRY | LM_FLAG_TRY_1CB)) &&
  1257. !gfs2_should_queue_trylock(gl, gh)) {
  1258. gh->gh_error = GLR_TRYFAILED;
  1259. gfs2_holder_wake(gh);
  1260. return;
  1261. }
  1262. list_for_each_entry(gh2, &gl->gl_holders, gh_list) {
  1263. if (likely(gh2->gh_owner_pid != gh->gh_owner_pid))
  1264. continue;
  1265. if (gh->gh_gl->gl_ops->go_type == LM_TYPE_FLOCK)
  1266. continue;
  1267. if (!pid_is_meaningful(gh2))
  1268. continue;
  1269. goto trap_recursive;
  1270. }
  1271. trace_gfs2_glock_queue(gh, 1);
  1272. gfs2_glstats_inc(gl, GFS2_LKS_QCOUNT);
  1273. gfs2_sbstats_inc(gl, GFS2_LKS_QCOUNT);
  1274. list_add_tail(&gh->gh_list, &gl->gl_holders);
  1275. return;
  1276. trap_recursive:
  1277. fs_err(sdp, "original: %pSR\n", (void *)gh2->gh_ip);
  1278. fs_err(sdp, "pid: %d\n", pid_nr(gh2->gh_owner_pid));
  1279. fs_err(sdp, "lock type: %d req lock state : %d\n",
  1280. glock_type(gh2->gh_gl), gh2->gh_state);
  1281. fs_err(sdp, "new: %pSR\n", (void *)gh->gh_ip);
  1282. fs_err(sdp, "pid: %d\n", pid_nr(gh->gh_owner_pid));
  1283. fs_err(sdp, "lock type: %d req lock state : %d\n",
  1284. glock_type(gh->gh_gl), gh->gh_state);
  1285. gfs2_dump_glock(NULL, gl, true);
  1286. BUG();
  1287. }
  1288. /**
  1289. * gfs2_glock_nq - enqueue a struct gfs2_holder onto a glock (acquire a glock)
  1290. * @gh: the holder structure
  1291. *
  1292. * if (gh->gh_flags & GL_ASYNC), this never returns an error
  1293. *
  1294. * Returns: 0, GLR_TRYFAILED, or errno on failure
  1295. */
  1296. int gfs2_glock_nq(struct gfs2_holder *gh)
  1297. {
  1298. struct gfs2_glock *gl = gh->gh_gl;
  1299. struct gfs2_sbd *sdp = glock_sbd(gl);
  1300. int error;
  1301. if (gfs2_withdrawn(sdp))
  1302. return -EIO;
  1303. if (gh->gh_flags & GL_NOBLOCK) {
  1304. struct gfs2_holder *current_gh;
  1305. error = -ECHILD;
  1306. spin_lock(&gl->gl_lockref.lock);
  1307. if (find_last_waiter(gl))
  1308. goto unlock;
  1309. current_gh = find_first_holder(gl);
  1310. if (!may_grant(gl, current_gh, gh))
  1311. goto unlock;
  1312. set_bit(HIF_HOLDER, &gh->gh_iflags);
  1313. list_add_tail(&gh->gh_list, &gl->gl_holders);
  1314. trace_gfs2_promote(gh);
  1315. error = 0;
  1316. unlock:
  1317. spin_unlock(&gl->gl_lockref.lock);
  1318. return error;
  1319. }
  1320. gh->gh_error = 0;
  1321. spin_lock(&gl->gl_lockref.lock);
  1322. add_to_queue(gh);
  1323. if (unlikely((LM_FLAG_RECOVER & gh->gh_flags) &&
  1324. test_and_clear_bit(GLF_HAVE_FROZEN_REPLY, &gl->gl_flags))) {
  1325. set_bit(GLF_HAVE_REPLY, &gl->gl_flags);
  1326. gl->gl_lockref.count++;
  1327. gfs2_glock_queue_work(gl, 0);
  1328. }
  1329. run_queue(gl, 1);
  1330. spin_unlock(&gl->gl_lockref.lock);
  1331. error = 0;
  1332. if (!(gh->gh_flags & GL_ASYNC))
  1333. error = gfs2_glock_wait(gh);
  1334. return error;
  1335. }
  1336. /**
  1337. * gfs2_glock_poll - poll to see if an async request has been completed
  1338. * @gh: the holder
  1339. *
  1340. * Returns: 1 if the request is ready to be gfs2_glock_wait()ed on
  1341. */
  1342. int gfs2_glock_poll(struct gfs2_holder *gh)
  1343. {
  1344. return test_bit(HIF_WAIT, &gh->gh_iflags) ? 0 : 1;
  1345. }
  1346. static void __gfs2_glock_dq(struct gfs2_holder *gh)
  1347. {
  1348. struct gfs2_glock *gl = gh->gh_gl;
  1349. unsigned delay = 0;
  1350. int fast_path = 0;
  1351. /*
  1352. * This holder should not be cached, so mark it for demote.
  1353. * Note: this should be done before the glock_needs_demote
  1354. * check below.
  1355. */
  1356. if (gh->gh_flags & GL_NOCACHE)
  1357. request_demote(gl, LM_ST_UNLOCKED, 0, false);
  1358. list_del_init(&gh->gh_list);
  1359. clear_bit(HIF_HOLDER, &gh->gh_iflags);
  1360. trace_gfs2_glock_queue(gh, 0);
  1361. if (test_bit(HIF_WAIT, &gh->gh_iflags))
  1362. gfs2_holder_wake(gh);
  1363. /*
  1364. * If there hasn't been a demote request we are done.
  1365. * (Let the remaining holders, if any, keep holding it.)
  1366. */
  1367. if (!glock_needs_demote(gl)) {
  1368. if (list_empty(&gl->gl_holders))
  1369. fast_path = 1;
  1370. }
  1371. if (unlikely(!fast_path)) {
  1372. gl->gl_lockref.count++;
  1373. if (test_bit(GLF_PENDING_DEMOTE, &gl->gl_flags) &&
  1374. !test_bit(GLF_DEMOTE, &gl->gl_flags) &&
  1375. glock_type(gl) == LM_TYPE_INODE)
  1376. delay = gl->gl_hold_time;
  1377. gfs2_glock_queue_work(gl, delay);
  1378. }
  1379. }
  1380. /**
  1381. * gfs2_glock_dq - dequeue a struct gfs2_holder from a glock (release a glock)
  1382. * @gh: the glock holder
  1383. *
  1384. */
  1385. void gfs2_glock_dq(struct gfs2_holder *gh)
  1386. {
  1387. struct gfs2_glock *gl = gh->gh_gl;
  1388. again:
  1389. spin_lock(&gl->gl_lockref.lock);
  1390. if (!gfs2_holder_queued(gh)) {
  1391. /*
  1392. * May have already been dequeued because the locking request
  1393. * was GL_ASYNC and it has failed in the meantime.
  1394. */
  1395. goto out;
  1396. }
  1397. if (list_is_first(&gh->gh_list, &gl->gl_holders) &&
  1398. !test_bit(HIF_HOLDER, &gh->gh_iflags) &&
  1399. test_bit(GLF_LOCK, &gl->gl_flags) &&
  1400. !test_bit(GLF_DEMOTE_IN_PROGRESS, &gl->gl_flags) &&
  1401. !test_bit(GLF_CANCELING, &gl->gl_flags)) {
  1402. if (!test_bit(GLF_MAY_CANCEL, &gl->gl_flags)) {
  1403. struct wait_queue_head *wq;
  1404. DEFINE_WAIT(wait);
  1405. wq = bit_waitqueue(&gl->gl_flags, GLF_LOCK);
  1406. prepare_to_wait(wq, &wait, TASK_UNINTERRUPTIBLE);
  1407. spin_unlock(&gl->gl_lockref.lock);
  1408. schedule();
  1409. finish_wait(wq, &wait);
  1410. goto again;
  1411. }
  1412. set_bit(GLF_CANCELING, &gl->gl_flags);
  1413. spin_unlock(&gl->gl_lockref.lock);
  1414. glock_sbd(gl)->sd_lockstruct.ls_ops->lm_cancel(gl);
  1415. wait_on_bit(&gh->gh_iflags, HIF_WAIT, TASK_UNINTERRUPTIBLE);
  1416. spin_lock(&gl->gl_lockref.lock);
  1417. clear_bit(GLF_CANCELING, &gl->gl_flags);
  1418. clear_and_wake_up_bit(GLF_LOCK, &gl->gl_flags);
  1419. if (!gfs2_holder_queued(gh))
  1420. goto out;
  1421. }
  1422. __gfs2_glock_dq(gh);
  1423. out:
  1424. spin_unlock(&gl->gl_lockref.lock);
  1425. }
  1426. void gfs2_glock_dq_wait(struct gfs2_holder *gh)
  1427. {
  1428. struct gfs2_glock *gl = gh->gh_gl;
  1429. gfs2_glock_dq(gh);
  1430. might_sleep();
  1431. wait_on_bit(&gl->gl_flags, GLF_DEMOTE, TASK_UNINTERRUPTIBLE);
  1432. }
  1433. /**
  1434. * gfs2_glock_dq_uninit - dequeue a holder from a glock and initialize it
  1435. * @gh: the holder structure
  1436. *
  1437. */
  1438. void gfs2_glock_dq_uninit(struct gfs2_holder *gh)
  1439. {
  1440. gfs2_glock_dq(gh);
  1441. gfs2_holder_uninit(gh);
  1442. }
  1443. /**
  1444. * gfs2_glock_nq_num - acquire a glock based on lock number
  1445. * @sdp: the filesystem
  1446. * @number: the lock number
  1447. * @glops: the glock operations for the type of glock
  1448. * @state: the state to acquire the glock in
  1449. * @flags: modifier flags for the acquisition
  1450. * @gh: the struct gfs2_holder
  1451. *
  1452. * Returns: errno
  1453. */
  1454. int gfs2_glock_nq_num(struct gfs2_sbd *sdp, u64 number,
  1455. const struct gfs2_glock_operations *glops,
  1456. unsigned int state, u16 flags, struct gfs2_holder *gh)
  1457. {
  1458. struct gfs2_glock *gl;
  1459. int error;
  1460. error = gfs2_glock_get(sdp, number, glops, CREATE, &gl);
  1461. if (!error) {
  1462. error = gfs2_glock_nq_init(gl, state, flags, gh);
  1463. gfs2_glock_put(gl);
  1464. }
  1465. return error;
  1466. }
  1467. /**
  1468. * glock_compare - Compare two struct gfs2_glock structures for sorting
  1469. * @arg_a: the first structure
  1470. * @arg_b: the second structure
  1471. *
  1472. */
  1473. static int glock_compare(const void *arg_a, const void *arg_b)
  1474. {
  1475. const struct gfs2_holder *gh_a = *(const struct gfs2_holder **)arg_a;
  1476. const struct gfs2_holder *gh_b = *(const struct gfs2_holder **)arg_b;
  1477. const struct lm_lockname *a = &gh_a->gh_gl->gl_name;
  1478. const struct lm_lockname *b = &gh_b->gh_gl->gl_name;
  1479. if (a->ln_number > b->ln_number)
  1480. return 1;
  1481. if (a->ln_number < b->ln_number)
  1482. return -1;
  1483. BUG_ON(gh_a->gh_gl->gl_ops->go_type == gh_b->gh_gl->gl_ops->go_type);
  1484. return 0;
  1485. }
  1486. /**
  1487. * nq_m_sync - synchronously acquire more than one glock in deadlock free order
  1488. * @num_gh: the number of structures
  1489. * @ghs: an array of struct gfs2_holder structures
  1490. * @p: placeholder for the holder structure to pass back
  1491. *
  1492. * Returns: 0 on success (all glocks acquired),
  1493. * errno on failure (no glocks acquired)
  1494. */
  1495. static int nq_m_sync(unsigned int num_gh, struct gfs2_holder *ghs,
  1496. struct gfs2_holder **p)
  1497. {
  1498. unsigned int x;
  1499. int error = 0;
  1500. for (x = 0; x < num_gh; x++)
  1501. p[x] = &ghs[x];
  1502. sort(p, num_gh, sizeof(struct gfs2_holder *), glock_compare, NULL);
  1503. for (x = 0; x < num_gh; x++) {
  1504. error = gfs2_glock_nq(p[x]);
  1505. if (error) {
  1506. while (x--)
  1507. gfs2_glock_dq(p[x]);
  1508. break;
  1509. }
  1510. }
  1511. return error;
  1512. }
  1513. /**
  1514. * gfs2_glock_nq_m - acquire multiple glocks
  1515. * @num_gh: the number of structures
  1516. * @ghs: an array of struct gfs2_holder structures
  1517. *
  1518. * Returns: 0 on success (all glocks acquired),
  1519. * errno on failure (no glocks acquired)
  1520. */
  1521. int gfs2_glock_nq_m(unsigned int num_gh, struct gfs2_holder *ghs)
  1522. {
  1523. struct gfs2_holder *tmp[4];
  1524. struct gfs2_holder **pph = tmp;
  1525. int error = 0;
  1526. switch(num_gh) {
  1527. case 0:
  1528. return 0;
  1529. case 1:
  1530. return gfs2_glock_nq(ghs);
  1531. default:
  1532. if (num_gh <= 4)
  1533. break;
  1534. pph = kmalloc_objs(struct gfs2_holder *, num_gh, GFP_NOFS);
  1535. if (!pph)
  1536. return -ENOMEM;
  1537. }
  1538. error = nq_m_sync(num_gh, ghs, pph);
  1539. if (pph != tmp)
  1540. kfree(pph);
  1541. return error;
  1542. }
  1543. /**
  1544. * gfs2_glock_dq_m - release multiple glocks
  1545. * @num_gh: the number of structures
  1546. * @ghs: an array of struct gfs2_holder structures
  1547. *
  1548. */
  1549. void gfs2_glock_dq_m(unsigned int num_gh, struct gfs2_holder *ghs)
  1550. {
  1551. while (num_gh--)
  1552. gfs2_glock_dq(&ghs[num_gh]);
  1553. }
  1554. void gfs2_glock_cb(struct gfs2_glock *gl, unsigned int state)
  1555. {
  1556. unsigned long delay = 0;
  1557. gfs2_glock_hold(gl);
  1558. spin_lock(&gl->gl_lockref.lock);
  1559. if (!list_empty(&gl->gl_holders) &&
  1560. glock_type(gl) == LM_TYPE_INODE) {
  1561. unsigned long now = jiffies;
  1562. unsigned long holdtime;
  1563. holdtime = gl->gl_tchange + gl->gl_hold_time;
  1564. if (time_before(now, holdtime))
  1565. delay = holdtime - now;
  1566. if (test_bit(GLF_HAVE_REPLY, &gl->gl_flags))
  1567. delay = gl->gl_hold_time;
  1568. }
  1569. request_demote(gl, state, delay, true);
  1570. gfs2_glock_queue_work(gl, delay);
  1571. spin_unlock(&gl->gl_lockref.lock);
  1572. }
  1573. /**
  1574. * gfs2_should_freeze - Figure out if glock should be frozen
  1575. * @gl: The glock in question
  1576. *
  1577. * Glocks are not frozen if (a) the result of the dlm operation is
  1578. * an error, (b) the locking operation was an unlock operation or
  1579. * (c) if there is a "recover" flagged request anywhere in the queue
  1580. *
  1581. * Returns: 1 if freezing should occur, 0 otherwise
  1582. */
  1583. static int gfs2_should_freeze(const struct gfs2_glock *gl)
  1584. {
  1585. const struct gfs2_holder *gh;
  1586. if (gl->gl_reply & ~LM_OUT_ST_MASK)
  1587. return 0;
  1588. if (gl->gl_target == LM_ST_UNLOCKED)
  1589. return 0;
  1590. list_for_each_entry(gh, &gl->gl_holders, gh_list) {
  1591. if (test_bit(HIF_HOLDER, &gh->gh_iflags))
  1592. continue;
  1593. if (LM_FLAG_RECOVER & gh->gh_flags)
  1594. return 0;
  1595. }
  1596. return 1;
  1597. }
  1598. /**
  1599. * gfs2_glock_complete - Callback used by locking
  1600. * @gl: Pointer to the glock
  1601. * @ret: The return value from the dlm
  1602. *
  1603. * The gl_reply field is under the gl_lockref.lock lock so that it is ok
  1604. * to use a bitfield shared with other glock state fields.
  1605. */
  1606. void gfs2_glock_complete(struct gfs2_glock *gl, int ret)
  1607. {
  1608. struct lm_lockstruct *ls = &glock_sbd(gl)->sd_lockstruct;
  1609. spin_lock(&gl->gl_lockref.lock);
  1610. clear_bit(GLF_MAY_CANCEL, &gl->gl_flags);
  1611. gl->gl_reply = ret;
  1612. if (unlikely(test_bit(DFL_BLOCK_LOCKS, &ls->ls_recover_flags))) {
  1613. if (gfs2_should_freeze(gl)) {
  1614. set_bit(GLF_HAVE_FROZEN_REPLY, &gl->gl_flags);
  1615. spin_unlock(&gl->gl_lockref.lock);
  1616. return;
  1617. }
  1618. }
  1619. gl->gl_lockref.count++;
  1620. set_bit(GLF_HAVE_REPLY, &gl->gl_flags);
  1621. gfs2_glock_queue_work(gl, 0);
  1622. spin_unlock(&gl->gl_lockref.lock);
  1623. }
  1624. static int glock_cmp(void *priv, const struct list_head *a,
  1625. const struct list_head *b)
  1626. {
  1627. struct gfs2_glock *gla, *glb;
  1628. gla = list_entry(a, struct gfs2_glock, gl_lru);
  1629. glb = list_entry(b, struct gfs2_glock, gl_lru);
  1630. if (glock_number(gla) > glock_number(glb))
  1631. return 1;
  1632. if (glock_number(gla) < glock_number(glb))
  1633. return -1;
  1634. return 0;
  1635. }
  1636. static bool can_free_glock(struct gfs2_glock *gl)
  1637. {
  1638. struct gfs2_sbd *sdp = glock_sbd(gl);
  1639. return !test_bit(GLF_LOCK, &gl->gl_flags) &&
  1640. !gl->gl_lockref.count &&
  1641. (!test_bit(GLF_LFLUSH, &gl->gl_flags) ||
  1642. test_bit(SDF_KILL, &sdp->sd_flags));
  1643. }
  1644. /**
  1645. * gfs2_dispose_glock_lru - Demote a list of glocks
  1646. * @list: The list to dispose of
  1647. *
  1648. * Disposing of glocks may involve disk accesses, so that here we sort
  1649. * the glocks by number (i.e. disk location of the inodes) so that if
  1650. * there are any such accesses, they'll be sent in order (mostly).
  1651. *
  1652. * Must be called under the lru_lock, but may drop and retake this
  1653. * lock. While the lru_lock is dropped, entries may vanish from the
  1654. * list, but no new entries will appear on the list (since it is
  1655. * private)
  1656. */
  1657. static unsigned long gfs2_dispose_glock_lru(struct list_head *list)
  1658. __releases(&lru_lock)
  1659. __acquires(&lru_lock)
  1660. {
  1661. struct gfs2_glock *gl;
  1662. unsigned long freed = 0;
  1663. list_sort(NULL, list, glock_cmp);
  1664. while(!list_empty(list)) {
  1665. gl = list_first_entry(list, struct gfs2_glock, gl_lru);
  1666. if (!spin_trylock(&gl->gl_lockref.lock)) {
  1667. add_back_to_lru:
  1668. list_move(&gl->gl_lru, &lru_list);
  1669. continue;
  1670. }
  1671. if (!can_free_glock(gl)) {
  1672. spin_unlock(&gl->gl_lockref.lock);
  1673. goto add_back_to_lru;
  1674. }
  1675. list_del_init(&gl->gl_lru);
  1676. atomic_dec(&lru_count);
  1677. clear_bit(GLF_LRU, &gl->gl_flags);
  1678. freed++;
  1679. gl->gl_lockref.count++;
  1680. if (gl->gl_state != LM_ST_UNLOCKED)
  1681. request_demote(gl, LM_ST_UNLOCKED, 0, false);
  1682. gfs2_glock_queue_work(gl, 0);
  1683. spin_unlock(&gl->gl_lockref.lock);
  1684. cond_resched_lock(&lru_lock);
  1685. }
  1686. return freed;
  1687. }
  1688. /**
  1689. * gfs2_scan_glock_lru - Scan the LRU looking for locks to demote
  1690. * @nr: The number of entries to scan
  1691. *
  1692. * This function selects the entries on the LRU which are able to
  1693. * be demoted, and then kicks off the process by calling
  1694. * gfs2_dispose_glock_lru() above.
  1695. */
  1696. static unsigned long gfs2_scan_glock_lru(unsigned long nr)
  1697. {
  1698. struct gfs2_glock *gl, *next;
  1699. LIST_HEAD(dispose);
  1700. unsigned long freed = 0;
  1701. spin_lock(&lru_lock);
  1702. list_for_each_entry_safe(gl, next, &lru_list, gl_lru) {
  1703. if (!nr--)
  1704. break;
  1705. if (can_free_glock(gl))
  1706. list_move(&gl->gl_lru, &dispose);
  1707. }
  1708. if (!list_empty(&dispose))
  1709. freed = gfs2_dispose_glock_lru(&dispose);
  1710. spin_unlock(&lru_lock);
  1711. return freed;
  1712. }
  1713. static unsigned long gfs2_glock_shrink_scan(struct shrinker *shrink,
  1714. struct shrink_control *sc)
  1715. {
  1716. if (!(sc->gfp_mask & __GFP_FS))
  1717. return SHRINK_STOP;
  1718. return gfs2_scan_glock_lru(sc->nr_to_scan);
  1719. }
  1720. static unsigned long gfs2_glock_shrink_count(struct shrinker *shrink,
  1721. struct shrink_control *sc)
  1722. {
  1723. return vfs_pressure_ratio(atomic_read(&lru_count));
  1724. }
  1725. static struct shrinker *glock_shrinker;
  1726. /**
  1727. * glock_hash_walk - Call a function for glock in a hash bucket
  1728. * @examiner: the function
  1729. * @sdp: the filesystem
  1730. *
  1731. * Note that the function can be called multiple times on the same
  1732. * object. So the user must ensure that the function can cope with
  1733. * that.
  1734. */
  1735. static void glock_hash_walk(glock_examiner examiner, const struct gfs2_sbd *sdp)
  1736. {
  1737. struct gfs2_glock *gl;
  1738. struct rhashtable_iter iter;
  1739. rhashtable_walk_enter(&gl_hash_table, &iter);
  1740. do {
  1741. rhashtable_walk_start(&iter);
  1742. while ((gl = rhashtable_walk_next(&iter)) && !IS_ERR(gl)) {
  1743. if (glock_sbd(gl) == sdp)
  1744. examiner(gl);
  1745. }
  1746. rhashtable_walk_stop(&iter);
  1747. } while (cond_resched(), gl == ERR_PTR(-EAGAIN));
  1748. rhashtable_walk_exit(&iter);
  1749. }
  1750. void gfs2_cancel_delete_work(struct gfs2_glock *gl)
  1751. {
  1752. clear_bit(GLF_TRY_TO_EVICT, &gl->gl_flags);
  1753. clear_bit(GLF_VERIFY_DELETE, &gl->gl_flags);
  1754. if (cancel_delayed_work(&gl->gl_delete))
  1755. gfs2_glock_put(gl);
  1756. }
  1757. static void flush_delete_work(struct gfs2_glock *gl)
  1758. {
  1759. if (glock_type(gl) == LM_TYPE_IOPEN) {
  1760. struct gfs2_sbd *sdp = glock_sbd(gl);
  1761. if (cancel_delayed_work(&gl->gl_delete)) {
  1762. queue_delayed_work(sdp->sd_delete_wq,
  1763. &gl->gl_delete, 0);
  1764. }
  1765. }
  1766. }
  1767. void gfs2_flush_delete_work(struct gfs2_sbd *sdp)
  1768. {
  1769. glock_hash_walk(flush_delete_work, sdp);
  1770. flush_workqueue(sdp->sd_delete_wq);
  1771. }
  1772. /**
  1773. * thaw_glock - thaw out a glock which has an unprocessed reply waiting
  1774. * @gl: The glock to thaw
  1775. *
  1776. */
  1777. static void thaw_glock(struct gfs2_glock *gl)
  1778. {
  1779. if (!test_and_clear_bit(GLF_HAVE_FROZEN_REPLY, &gl->gl_flags))
  1780. return;
  1781. if (!lockref_get_not_dead(&gl->gl_lockref))
  1782. return;
  1783. gfs2_glock_remove_from_lru(gl);
  1784. spin_lock(&gl->gl_lockref.lock);
  1785. set_bit(GLF_HAVE_REPLY, &gl->gl_flags);
  1786. gfs2_glock_queue_work(gl, 0);
  1787. spin_unlock(&gl->gl_lockref.lock);
  1788. }
  1789. /**
  1790. * clear_glock - look at a glock and see if we can free it from glock cache
  1791. * @gl: the glock to look at
  1792. *
  1793. */
  1794. static void clear_glock(struct gfs2_glock *gl)
  1795. {
  1796. gfs2_glock_remove_from_lru(gl);
  1797. spin_lock(&gl->gl_lockref.lock);
  1798. if (!__lockref_is_dead(&gl->gl_lockref)) {
  1799. gl->gl_lockref.count++;
  1800. if (gl->gl_state != LM_ST_UNLOCKED)
  1801. request_demote(gl, LM_ST_UNLOCKED, 0, false);
  1802. gfs2_glock_queue_work(gl, 0);
  1803. }
  1804. spin_unlock(&gl->gl_lockref.lock);
  1805. }
  1806. /**
  1807. * gfs2_glock_thaw - Thaw any frozen glocks
  1808. * @sdp: The super block
  1809. *
  1810. */
  1811. void gfs2_glock_thaw(struct gfs2_sbd *sdp)
  1812. {
  1813. glock_hash_walk(thaw_glock, sdp);
  1814. }
  1815. static void dump_glock(struct seq_file *seq, struct gfs2_glock *gl, bool fsid)
  1816. {
  1817. spin_lock(&gl->gl_lockref.lock);
  1818. gfs2_dump_glock(seq, gl, fsid);
  1819. spin_unlock(&gl->gl_lockref.lock);
  1820. }
  1821. static void dump_glock_func(struct gfs2_glock *gl)
  1822. {
  1823. dump_glock(NULL, gl, true);
  1824. }
  1825. static void withdraw_glock(struct gfs2_glock *gl)
  1826. {
  1827. spin_lock(&gl->gl_lockref.lock);
  1828. if (!__lockref_is_dead(&gl->gl_lockref)) {
  1829. /*
  1830. * We don't want to write back any more dirty data. Unlock the
  1831. * remaining inode and resource group glocks; this will cause
  1832. * their ->go_inval() hooks to toss out all the remaining
  1833. * cached data, dirty or not.
  1834. */
  1835. if (gl->gl_ops->go_inval && gl->gl_state != LM_ST_UNLOCKED)
  1836. request_demote(gl, LM_ST_UNLOCKED, 0, false);
  1837. do_error(gl, LM_OUT_ERROR); /* remove pending waiters */
  1838. }
  1839. spin_unlock(&gl->gl_lockref.lock);
  1840. }
  1841. void gfs2_withdraw_glocks(struct gfs2_sbd *sdp)
  1842. {
  1843. glock_hash_walk(withdraw_glock, sdp);
  1844. }
  1845. /**
  1846. * gfs2_gl_hash_clear - Empty out the glock hash table
  1847. * @sdp: the filesystem
  1848. *
  1849. * Called when unmounting the filesystem.
  1850. */
  1851. void gfs2_gl_hash_clear(struct gfs2_sbd *sdp)
  1852. {
  1853. unsigned long start = jiffies;
  1854. bool timed_out = false;
  1855. set_bit(SDF_SKIP_DLM_UNLOCK, &sdp->sd_flags);
  1856. flush_workqueue(sdp->sd_glock_wq);
  1857. glock_hash_walk(clear_glock, sdp);
  1858. flush_workqueue(sdp->sd_glock_wq);
  1859. while (!timed_out) {
  1860. wait_event_timeout(sdp->sd_kill_wait,
  1861. !atomic_read(&sdp->sd_glock_disposal),
  1862. HZ * 60);
  1863. if (!atomic_read(&sdp->sd_glock_disposal))
  1864. break;
  1865. timed_out = time_after(jiffies, start + (HZ * 600));
  1866. fs_warn(sdp, "%u glocks left after %u seconds%s\n",
  1867. atomic_read(&sdp->sd_glock_disposal),
  1868. jiffies_to_msecs(jiffies - start) / 1000,
  1869. timed_out ? ":" : "; still waiting");
  1870. }
  1871. gfs2_lm_unmount(sdp);
  1872. gfs2_free_dead_glocks(sdp);
  1873. glock_hash_walk(dump_glock_func, sdp);
  1874. destroy_workqueue(sdp->sd_glock_wq);
  1875. sdp->sd_glock_wq = NULL;
  1876. }
  1877. static const char *state2str(unsigned state)
  1878. {
  1879. switch(state) {
  1880. case LM_ST_UNLOCKED:
  1881. return "UN";
  1882. case LM_ST_SHARED:
  1883. return "SH";
  1884. case LM_ST_DEFERRED:
  1885. return "DF";
  1886. case LM_ST_EXCLUSIVE:
  1887. return "EX";
  1888. }
  1889. return "??";
  1890. }
  1891. static const char *hflags2str(char *buf, u16 flags, unsigned long iflags)
  1892. {
  1893. char *p = buf;
  1894. if (flags & LM_FLAG_TRY)
  1895. *p++ = 't';
  1896. if (flags & LM_FLAG_TRY_1CB)
  1897. *p++ = 'T';
  1898. if (flags & LM_FLAG_RECOVER)
  1899. *p++ = 'e';
  1900. if (flags & LM_FLAG_ANY)
  1901. *p++ = 'A';
  1902. if (flags & LM_FLAG_NODE_SCOPE)
  1903. *p++ = 'n';
  1904. if (flags & GL_ASYNC)
  1905. *p++ = 'a';
  1906. if (flags & GL_EXACT)
  1907. *p++ = 'E';
  1908. if (flags & GL_NOCACHE)
  1909. *p++ = 'c';
  1910. if (test_bit(HIF_HOLDER, &iflags))
  1911. *p++ = 'H';
  1912. if (test_bit(HIF_WAIT, &iflags))
  1913. *p++ = 'W';
  1914. if (flags & GL_SKIP)
  1915. *p++ = 's';
  1916. *p = 0;
  1917. return buf;
  1918. }
  1919. /**
  1920. * dump_holder - print information about a glock holder
  1921. * @seq: the seq_file struct
  1922. * @gh: the glock holder
  1923. * @fs_id_buf: pointer to file system id (if requested)
  1924. *
  1925. */
  1926. static void dump_holder(struct seq_file *seq, const struct gfs2_holder *gh,
  1927. const char *fs_id_buf)
  1928. {
  1929. const char *comm = "(none)";
  1930. pid_t owner_pid = 0;
  1931. char flags_buf[32];
  1932. rcu_read_lock();
  1933. if (pid_is_meaningful(gh)) {
  1934. struct task_struct *gh_owner;
  1935. comm = "(ended)";
  1936. owner_pid = pid_nr(gh->gh_owner_pid);
  1937. gh_owner = pid_task(gh->gh_owner_pid, PIDTYPE_PID);
  1938. if (gh_owner)
  1939. comm = gh_owner->comm;
  1940. }
  1941. gfs2_print_dbg(seq, "%s H: s:%s f:%s e:%d p:%ld [%s] %pS\n",
  1942. fs_id_buf, state2str(gh->gh_state),
  1943. hflags2str(flags_buf, gh->gh_flags, gh->gh_iflags),
  1944. gh->gh_error, (long)owner_pid, comm, (void *)gh->gh_ip);
  1945. rcu_read_unlock();
  1946. }
  1947. static const char *gflags2str(char *buf, const struct gfs2_glock *gl)
  1948. {
  1949. const unsigned long *gflags = &gl->gl_flags;
  1950. char *p = buf;
  1951. if (test_bit(GLF_LOCK, gflags))
  1952. *p++ = 'l';
  1953. if (test_bit(GLF_DEMOTE, gflags))
  1954. *p++ = 'D';
  1955. if (test_bit(GLF_PENDING_DEMOTE, gflags))
  1956. *p++ = 'd';
  1957. if (test_bit(GLF_DEMOTE_IN_PROGRESS, gflags))
  1958. *p++ = 'p';
  1959. if (test_bit(GLF_DIRTY, gflags))
  1960. *p++ = 'y';
  1961. if (test_bit(GLF_LFLUSH, gflags))
  1962. *p++ = 'f';
  1963. if (test_bit(GLF_MAY_CANCEL, gflags))
  1964. *p++ = 'c';
  1965. if (test_bit(GLF_HAVE_REPLY, gflags))
  1966. *p++ = 'r';
  1967. if (test_bit(GLF_INITIAL, gflags))
  1968. *p++ = 'a';
  1969. if (test_bit(GLF_HAVE_FROZEN_REPLY, gflags))
  1970. *p++ = 'F';
  1971. if (!list_empty(&gl->gl_holders))
  1972. *p++ = 'q';
  1973. if (test_bit(GLF_LRU, gflags))
  1974. *p++ = 'L';
  1975. if (gl->gl_object)
  1976. *p++ = 'o';
  1977. if (test_bit(GLF_BLOCKING, gflags))
  1978. *p++ = 'b';
  1979. if (test_bit(GLF_INSTANTIATE_NEEDED, gflags))
  1980. *p++ = 'n';
  1981. if (test_bit(GLF_INSTANTIATE_IN_PROG, gflags))
  1982. *p++ = 'N';
  1983. if (test_bit(GLF_TRY_TO_EVICT, gflags))
  1984. *p++ = 'e';
  1985. if (test_bit(GLF_VERIFY_DELETE, gflags))
  1986. *p++ = 'E';
  1987. if (test_bit(GLF_DEFER_DELETE, gflags))
  1988. *p++ = 's';
  1989. if (test_bit(GLF_CANCELING, gflags))
  1990. *p++ = 'C';
  1991. *p = 0;
  1992. return buf;
  1993. }
  1994. /**
  1995. * gfs2_dump_glock - print information about a glock
  1996. * @seq: The seq_file struct
  1997. * @gl: the glock
  1998. * @fsid: If true, also dump the file system id
  1999. *
  2000. * The file format is as follows:
  2001. * One line per object, capital letters are used to indicate objects
  2002. * G = glock, I = Inode, R = rgrp, H = holder. Glocks are not indented,
  2003. * other objects are indented by a single space and follow the glock to
  2004. * which they are related. Fields are indicated by lower case letters
  2005. * followed by a colon and the field value, except for strings which are in
  2006. * [] so that its possible to see if they are composed of spaces for
  2007. * example. The field's are n = number (id of the object), f = flags,
  2008. * t = type, s = state, r = refcount, e = error, p = pid.
  2009. *
  2010. */
  2011. void gfs2_dump_glock(struct seq_file *seq, struct gfs2_glock *gl, bool fsid)
  2012. {
  2013. const struct gfs2_glock_operations *glops = gl->gl_ops;
  2014. unsigned long long dtime;
  2015. const struct gfs2_holder *gh;
  2016. char gflags_buf[32];
  2017. struct gfs2_sbd *sdp = glock_sbd(gl);
  2018. char fs_id_buf[sizeof(sdp->sd_fsname) + 7];
  2019. unsigned long nrpages = 0;
  2020. if (gl->gl_ops->go_flags & GLOF_ASPACE) {
  2021. struct address_space *mapping = gfs2_glock2aspace(gl);
  2022. nrpages = mapping->nrpages;
  2023. }
  2024. memset(fs_id_buf, 0, sizeof(fs_id_buf));
  2025. if (fsid && sdp) /* safety precaution */
  2026. sprintf(fs_id_buf, "fsid=%s: ", sdp->sd_fsname);
  2027. dtime = jiffies - gl->gl_demote_time;
  2028. dtime *= 1000000/HZ; /* demote time in uSec */
  2029. if (!test_bit(GLF_DEMOTE, &gl->gl_flags))
  2030. dtime = 0;
  2031. gfs2_print_dbg(seq, "%sG: s:%s n:%u/%llx f:%s t:%s d:%s/%llu a:%d "
  2032. "v:%d r:%d m:%ld p:%lu\n",
  2033. fs_id_buf, state2str(gl->gl_state),
  2034. glock_type(gl),
  2035. (unsigned long long) glock_number(gl),
  2036. gflags2str(gflags_buf, gl),
  2037. state2str(gl->gl_target),
  2038. state2str(gl->gl_demote_state), dtime,
  2039. atomic_read(&gl->gl_ail_count),
  2040. atomic_read(&gl->gl_revokes),
  2041. (int)gl->gl_lockref.count, gl->gl_hold_time, nrpages);
  2042. list_for_each_entry(gh, &gl->gl_holders, gh_list)
  2043. dump_holder(seq, gh, fs_id_buf);
  2044. if (gl->gl_state != LM_ST_UNLOCKED && glops->go_dump)
  2045. glops->go_dump(seq, gl, fs_id_buf);
  2046. }
  2047. static int gfs2_glstats_seq_show(struct seq_file *seq, void *iter_ptr)
  2048. {
  2049. struct gfs2_glock *gl = iter_ptr;
  2050. seq_printf(seq, "G: n:%u/%llx rtt:%llu/%llu rttb:%llu/%llu irt:%llu/%llu dcnt: %llu qcnt: %llu\n",
  2051. glock_type(gl),
  2052. (unsigned long long) glock_number(gl),
  2053. (unsigned long long)gl->gl_stats.stats[GFS2_LKS_SRTT],
  2054. (unsigned long long)gl->gl_stats.stats[GFS2_LKS_SRTTVAR],
  2055. (unsigned long long)gl->gl_stats.stats[GFS2_LKS_SRTTB],
  2056. (unsigned long long)gl->gl_stats.stats[GFS2_LKS_SRTTVARB],
  2057. (unsigned long long)gl->gl_stats.stats[GFS2_LKS_SIRT],
  2058. (unsigned long long)gl->gl_stats.stats[GFS2_LKS_SIRTVAR],
  2059. (unsigned long long)gl->gl_stats.stats[GFS2_LKS_DCOUNT],
  2060. (unsigned long long)gl->gl_stats.stats[GFS2_LKS_QCOUNT]);
  2061. return 0;
  2062. }
  2063. static const char *gfs2_gltype[] = {
  2064. "type",
  2065. "reserved",
  2066. "nondisk",
  2067. "inode",
  2068. "rgrp",
  2069. "meta",
  2070. "iopen",
  2071. "flock",
  2072. "plock",
  2073. "quota",
  2074. "journal",
  2075. };
  2076. static const char *gfs2_stype[] = {
  2077. [GFS2_LKS_SRTT] = "srtt",
  2078. [GFS2_LKS_SRTTVAR] = "srttvar",
  2079. [GFS2_LKS_SRTTB] = "srttb",
  2080. [GFS2_LKS_SRTTVARB] = "srttvarb",
  2081. [GFS2_LKS_SIRT] = "sirt",
  2082. [GFS2_LKS_SIRTVAR] = "sirtvar",
  2083. [GFS2_LKS_DCOUNT] = "dlm",
  2084. [GFS2_LKS_QCOUNT] = "queue",
  2085. };
  2086. #define GFS2_NR_SBSTATS (ARRAY_SIZE(gfs2_gltype) * ARRAY_SIZE(gfs2_stype))
  2087. static int gfs2_sbstats_seq_show(struct seq_file *seq, void *iter_ptr)
  2088. {
  2089. struct gfs2_sbd *sdp = seq->private;
  2090. loff_t pos = *(loff_t *)iter_ptr;
  2091. unsigned index = pos >> 3;
  2092. unsigned subindex = pos & 0x07;
  2093. int i;
  2094. if (index == 0 && subindex != 0)
  2095. return 0;
  2096. seq_printf(seq, "%-10s %8s:", gfs2_gltype[index],
  2097. (index == 0) ? "cpu": gfs2_stype[subindex]);
  2098. for_each_possible_cpu(i) {
  2099. const struct gfs2_pcpu_lkstats *lkstats = per_cpu_ptr(sdp->sd_lkstats, i);
  2100. if (index == 0)
  2101. seq_printf(seq, " %15u", i);
  2102. else
  2103. seq_printf(seq, " %15llu", (unsigned long long)lkstats->
  2104. lkstats[index - 1].stats[subindex]);
  2105. }
  2106. seq_putc(seq, '\n');
  2107. return 0;
  2108. }
  2109. int __init gfs2_glock_init(void)
  2110. {
  2111. int i, ret;
  2112. ret = rhashtable_init(&gl_hash_table, &ht_parms);
  2113. if (ret < 0)
  2114. return ret;
  2115. glock_shrinker = shrinker_alloc(0, "gfs2-glock");
  2116. if (!glock_shrinker) {
  2117. rhashtable_destroy(&gl_hash_table);
  2118. return -ENOMEM;
  2119. }
  2120. glock_shrinker->count_objects = gfs2_glock_shrink_count;
  2121. glock_shrinker->scan_objects = gfs2_glock_shrink_scan;
  2122. shrinker_register(glock_shrinker);
  2123. for (i = 0; i < GLOCK_WAIT_TABLE_SIZE; i++)
  2124. init_waitqueue_head(glock_wait_table + i);
  2125. return 0;
  2126. }
  2127. void gfs2_glock_exit(void)
  2128. {
  2129. shrinker_free(glock_shrinker);
  2130. rhashtable_destroy(&gl_hash_table);
  2131. }
  2132. static void gfs2_glock_iter_next(struct gfs2_glock_iter *gi, loff_t n)
  2133. {
  2134. struct gfs2_glock *gl = gi->gl;
  2135. if (gl) {
  2136. if (n == 0)
  2137. return;
  2138. gfs2_glock_put_async(gl);
  2139. }
  2140. for (;;) {
  2141. gl = rhashtable_walk_next(&gi->hti);
  2142. if (IS_ERR_OR_NULL(gl)) {
  2143. if (gl == ERR_PTR(-EAGAIN)) {
  2144. n = 1;
  2145. continue;
  2146. }
  2147. gl = NULL;
  2148. break;
  2149. }
  2150. if (glock_sbd(gl) != gi->sdp)
  2151. continue;
  2152. if (n <= 1) {
  2153. if (!lockref_get_not_dead(&gl->gl_lockref))
  2154. continue;
  2155. break;
  2156. } else {
  2157. if (__lockref_is_dead(&gl->gl_lockref))
  2158. continue;
  2159. n--;
  2160. }
  2161. }
  2162. gi->gl = gl;
  2163. }
  2164. static void *gfs2_glock_seq_start(struct seq_file *seq, loff_t *pos)
  2165. __acquires(RCU)
  2166. {
  2167. struct gfs2_glock_iter *gi = seq->private;
  2168. loff_t n;
  2169. /*
  2170. * We can either stay where we are, skip to the next hash table
  2171. * entry, or start from the beginning.
  2172. */
  2173. if (*pos < gi->last_pos) {
  2174. rhashtable_walk_exit(&gi->hti);
  2175. rhashtable_walk_enter(&gl_hash_table, &gi->hti);
  2176. n = *pos + 1;
  2177. } else {
  2178. n = *pos - gi->last_pos;
  2179. }
  2180. rhashtable_walk_start(&gi->hti);
  2181. gfs2_glock_iter_next(gi, n);
  2182. gi->last_pos = *pos;
  2183. return gi->gl;
  2184. }
  2185. static void *gfs2_glock_seq_next(struct seq_file *seq, void *iter_ptr,
  2186. loff_t *pos)
  2187. {
  2188. struct gfs2_glock_iter *gi = seq->private;
  2189. (*pos)++;
  2190. gi->last_pos = *pos;
  2191. gfs2_glock_iter_next(gi, 1);
  2192. return gi->gl;
  2193. }
  2194. static void gfs2_glock_seq_stop(struct seq_file *seq, void *iter_ptr)
  2195. __releases(RCU)
  2196. {
  2197. struct gfs2_glock_iter *gi = seq->private;
  2198. rhashtable_walk_stop(&gi->hti);
  2199. }
  2200. static int gfs2_glock_seq_show(struct seq_file *seq, void *iter_ptr)
  2201. {
  2202. dump_glock(seq, iter_ptr, false);
  2203. return 0;
  2204. }
  2205. static void *gfs2_sbstats_seq_start(struct seq_file *seq, loff_t *pos)
  2206. {
  2207. preempt_disable();
  2208. if (*pos >= GFS2_NR_SBSTATS)
  2209. return NULL;
  2210. return pos;
  2211. }
  2212. static void *gfs2_sbstats_seq_next(struct seq_file *seq, void *iter_ptr,
  2213. loff_t *pos)
  2214. {
  2215. (*pos)++;
  2216. if (*pos >= GFS2_NR_SBSTATS)
  2217. return NULL;
  2218. return pos;
  2219. }
  2220. static void gfs2_sbstats_seq_stop(struct seq_file *seq, void *iter_ptr)
  2221. {
  2222. preempt_enable();
  2223. }
  2224. static const struct seq_operations gfs2_glock_seq_ops = {
  2225. .start = gfs2_glock_seq_start,
  2226. .next = gfs2_glock_seq_next,
  2227. .stop = gfs2_glock_seq_stop,
  2228. .show = gfs2_glock_seq_show,
  2229. };
  2230. static const struct seq_operations gfs2_glstats_seq_ops = {
  2231. .start = gfs2_glock_seq_start,
  2232. .next = gfs2_glock_seq_next,
  2233. .stop = gfs2_glock_seq_stop,
  2234. .show = gfs2_glstats_seq_show,
  2235. };
  2236. static const struct seq_operations gfs2_sbstats_sops = {
  2237. .start = gfs2_sbstats_seq_start,
  2238. .next = gfs2_sbstats_seq_next,
  2239. .stop = gfs2_sbstats_seq_stop,
  2240. .show = gfs2_sbstats_seq_show,
  2241. };
  2242. #define GFS2_SEQ_GOODSIZE min(PAGE_SIZE << PAGE_ALLOC_COSTLY_ORDER, 65536UL)
  2243. static int __gfs2_glocks_open(struct inode *inode, struct file *file,
  2244. const struct seq_operations *ops)
  2245. {
  2246. int ret = seq_open_private(file, ops, sizeof(struct gfs2_glock_iter));
  2247. if (ret == 0) {
  2248. struct seq_file *seq = file->private_data;
  2249. struct gfs2_glock_iter *gi = seq->private;
  2250. gi->sdp = inode->i_private;
  2251. seq->buf = kmalloc(GFS2_SEQ_GOODSIZE, GFP_KERNEL | __GFP_NOWARN);
  2252. if (seq->buf)
  2253. seq->size = GFS2_SEQ_GOODSIZE;
  2254. /*
  2255. * Initially, we are "before" the first hash table entry; the
  2256. * first call to rhashtable_walk_next gets us the first entry.
  2257. */
  2258. gi->last_pos = -1;
  2259. gi->gl = NULL;
  2260. rhashtable_walk_enter(&gl_hash_table, &gi->hti);
  2261. }
  2262. return ret;
  2263. }
  2264. static int gfs2_glocks_open(struct inode *inode, struct file *file)
  2265. {
  2266. return __gfs2_glocks_open(inode, file, &gfs2_glock_seq_ops);
  2267. }
  2268. static int gfs2_glocks_release(struct inode *inode, struct file *file)
  2269. {
  2270. struct seq_file *seq = file->private_data;
  2271. struct gfs2_glock_iter *gi = seq->private;
  2272. if (gi->gl)
  2273. gfs2_glock_put(gi->gl);
  2274. rhashtable_walk_exit(&gi->hti);
  2275. return seq_release_private(inode, file);
  2276. }
  2277. static int gfs2_glstats_open(struct inode *inode, struct file *file)
  2278. {
  2279. return __gfs2_glocks_open(inode, file, &gfs2_glstats_seq_ops);
  2280. }
  2281. static const struct file_operations gfs2_glocks_fops = {
  2282. .owner = THIS_MODULE,
  2283. .open = gfs2_glocks_open,
  2284. .read = seq_read,
  2285. .llseek = seq_lseek,
  2286. .release = gfs2_glocks_release,
  2287. };
  2288. static const struct file_operations gfs2_glstats_fops = {
  2289. .owner = THIS_MODULE,
  2290. .open = gfs2_glstats_open,
  2291. .read = seq_read,
  2292. .llseek = seq_lseek,
  2293. .release = gfs2_glocks_release,
  2294. };
  2295. struct gfs2_glockfd_iter {
  2296. struct super_block *sb;
  2297. unsigned int tgid;
  2298. struct task_struct *task;
  2299. unsigned int fd;
  2300. struct file *file;
  2301. };
  2302. static struct task_struct *gfs2_glockfd_next_task(struct gfs2_glockfd_iter *i)
  2303. {
  2304. struct pid_namespace *ns = task_active_pid_ns(current);
  2305. struct pid *pid;
  2306. if (i->task)
  2307. put_task_struct(i->task);
  2308. rcu_read_lock();
  2309. retry:
  2310. i->task = NULL;
  2311. pid = find_ge_pid(i->tgid, ns);
  2312. if (pid) {
  2313. i->tgid = pid_nr_ns(pid, ns);
  2314. i->task = pid_task(pid, PIDTYPE_TGID);
  2315. if (!i->task) {
  2316. i->tgid++;
  2317. goto retry;
  2318. }
  2319. get_task_struct(i->task);
  2320. }
  2321. rcu_read_unlock();
  2322. return i->task;
  2323. }
  2324. static struct file *gfs2_glockfd_next_file(struct gfs2_glockfd_iter *i)
  2325. {
  2326. if (i->file) {
  2327. fput(i->file);
  2328. i->file = NULL;
  2329. }
  2330. for(;; i->fd++) {
  2331. i->file = fget_task_next(i->task, &i->fd);
  2332. if (!i->file) {
  2333. i->fd = 0;
  2334. break;
  2335. }
  2336. if (file_inode(i->file)->i_sb == i->sb)
  2337. break;
  2338. fput(i->file);
  2339. }
  2340. return i->file;
  2341. }
  2342. static void *gfs2_glockfd_seq_start(struct seq_file *seq, loff_t *pos)
  2343. {
  2344. struct gfs2_glockfd_iter *i = seq->private;
  2345. if (*pos)
  2346. return NULL;
  2347. while (gfs2_glockfd_next_task(i)) {
  2348. if (gfs2_glockfd_next_file(i))
  2349. return i;
  2350. i->tgid++;
  2351. }
  2352. return NULL;
  2353. }
  2354. static void *gfs2_glockfd_seq_next(struct seq_file *seq, void *iter_ptr,
  2355. loff_t *pos)
  2356. {
  2357. struct gfs2_glockfd_iter *i = seq->private;
  2358. (*pos)++;
  2359. i->fd++;
  2360. do {
  2361. if (gfs2_glockfd_next_file(i))
  2362. return i;
  2363. i->tgid++;
  2364. } while (gfs2_glockfd_next_task(i));
  2365. return NULL;
  2366. }
  2367. static void gfs2_glockfd_seq_stop(struct seq_file *seq, void *iter_ptr)
  2368. {
  2369. struct gfs2_glockfd_iter *i = seq->private;
  2370. if (i->file)
  2371. fput(i->file);
  2372. if (i->task)
  2373. put_task_struct(i->task);
  2374. }
  2375. static void gfs2_glockfd_seq_show_flock(struct seq_file *seq,
  2376. struct gfs2_glockfd_iter *i)
  2377. {
  2378. struct gfs2_file *fp = i->file->private_data;
  2379. struct gfs2_holder *fl_gh = &fp->f_fl_gh;
  2380. struct lm_lockname gl_name = { .ln_type = LM_TYPE_RESERVED };
  2381. if (!READ_ONCE(fl_gh->gh_gl))
  2382. return;
  2383. spin_lock(&i->file->f_lock);
  2384. if (gfs2_holder_initialized(fl_gh))
  2385. gl_name = fl_gh->gh_gl->gl_name;
  2386. spin_unlock(&i->file->f_lock);
  2387. if (gl_name.ln_type != LM_TYPE_RESERVED) {
  2388. seq_printf(seq, "%d %u %u/%llx\n",
  2389. i->tgid, i->fd, gl_name.ln_type,
  2390. (unsigned long long)gl_name.ln_number);
  2391. }
  2392. }
  2393. static int gfs2_glockfd_seq_show(struct seq_file *seq, void *iter_ptr)
  2394. {
  2395. struct gfs2_glockfd_iter *i = seq->private;
  2396. struct inode *inode = file_inode(i->file);
  2397. struct gfs2_glock *gl;
  2398. inode_lock_shared(inode);
  2399. gl = GFS2_I(inode)->i_iopen_gh.gh_gl;
  2400. if (gl) {
  2401. seq_printf(seq, "%d %u %u/%llx\n",
  2402. i->tgid, i->fd, glock_type(gl),
  2403. (unsigned long long) glock_number(gl));
  2404. }
  2405. gfs2_glockfd_seq_show_flock(seq, i);
  2406. inode_unlock_shared(inode);
  2407. return 0;
  2408. }
  2409. static const struct seq_operations gfs2_glockfd_seq_ops = {
  2410. .start = gfs2_glockfd_seq_start,
  2411. .next = gfs2_glockfd_seq_next,
  2412. .stop = gfs2_glockfd_seq_stop,
  2413. .show = gfs2_glockfd_seq_show,
  2414. };
  2415. static int gfs2_glockfd_open(struct inode *inode, struct file *file)
  2416. {
  2417. struct gfs2_glockfd_iter *i;
  2418. struct gfs2_sbd *sdp = inode->i_private;
  2419. i = __seq_open_private(file, &gfs2_glockfd_seq_ops,
  2420. sizeof(struct gfs2_glockfd_iter));
  2421. if (!i)
  2422. return -ENOMEM;
  2423. i->sb = sdp->sd_vfs;
  2424. return 0;
  2425. }
  2426. static const struct file_operations gfs2_glockfd_fops = {
  2427. .owner = THIS_MODULE,
  2428. .open = gfs2_glockfd_open,
  2429. .read = seq_read,
  2430. .llseek = seq_lseek,
  2431. .release = seq_release_private,
  2432. };
  2433. DEFINE_SEQ_ATTRIBUTE(gfs2_sbstats);
  2434. void gfs2_create_debugfs_file(struct gfs2_sbd *sdp)
  2435. {
  2436. sdp->debugfs_dir = debugfs_create_dir(sdp->sd_table_name, gfs2_root);
  2437. debugfs_create_file("glocks", S_IFREG | S_IRUGO, sdp->debugfs_dir, sdp,
  2438. &gfs2_glocks_fops);
  2439. debugfs_create_file("glockfd", S_IFREG | S_IRUGO, sdp->debugfs_dir, sdp,
  2440. &gfs2_glockfd_fops);
  2441. debugfs_create_file("glstats", S_IFREG | S_IRUGO, sdp->debugfs_dir, sdp,
  2442. &gfs2_glstats_fops);
  2443. debugfs_create_file("sbstats", S_IFREG | S_IRUGO, sdp->debugfs_dir, sdp,
  2444. &gfs2_sbstats_fops);
  2445. }
  2446. void gfs2_delete_debugfs_file(struct gfs2_sbd *sdp)
  2447. {
  2448. debugfs_remove_recursive(sdp->debugfs_dir);
  2449. sdp->debugfs_dir = NULL;
  2450. }
  2451. void gfs2_register_debugfs(void)
  2452. {
  2453. gfs2_root = debugfs_create_dir("gfs2", NULL);
  2454. }
  2455. void gfs2_unregister_debugfs(void)
  2456. {
  2457. debugfs_remove(gfs2_root);
  2458. gfs2_root = NULL;
  2459. }