memset.S 4.4 KB

123456789101112131415161718192021222324252627282930313233343536373839404142434445464748495051525354555657585960616263646566676869707172737475767778798081828384858687888990919293949596979899100101102103104105106107108109110111112113114115116117118119120121122123124125126127128129130131132133134135136137138139140141142143144145146147148149150151152153154155156157158159
  1. /* memset.S: optimised assembly memset
  2. *
  3. * Copyright (C) 2003, 2004 Red Hat, Inc. All Rights Reserved.
  4. * Written by David Howells (dhowells@redhat.com)
  5. *
  6. * This library is free software; you can redistribute it and/or
  7. * modify it under the terms of the GNU Library General Public
  8. * License as published by the Free Software Foundation; either
  9. * version 2 of the License, or (at your option) any later version.
  10. *
  11. * This library is distributed in the hope that it will be useful,
  12. * but WITHOUT ANY WARRANTY; without even the implied warranty of
  13. * MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the GNU
  14. * Library General Public License for more details.
  15. *
  16. * You should have received a copy of the GNU Library General Public
  17. * License along with this library; if not, write to the Free
  18. * Software Foundation, Inc., 675 Mass Ave, Cambridge, MA 02139, USA.
  19. */
  20. #include <features.h>
  21. .text
  22. .p2align 4
  23. ###############################################################################
  24. #
  25. # void *memset(void *p, char ch, size_t count)
  26. #
  27. # - NOTE: must not use any stack. exception detection performs function return
  28. # to caller's fixup routine, aborting the remainder of the set
  29. # GR4, GR7, GR8, and GR11 must be managed
  30. #
  31. ###############################################################################
  32. .globl __memset
  33. .hidden __memset
  34. .type __memset,@function
  35. __memset:
  36. orcc.p gr10,gr0,gr5,icc3 ; GR5 = count
  37. andi gr9,#0xff,gr9
  38. or.p gr8,gr0,gr4 ; GR4 = address
  39. beqlr icc3,#0
  40. # conditionally write a byte to 2b-align the address
  41. setlos.p #1,gr6
  42. andicc gr4,#1,gr0,icc0
  43. ckne icc0,cc7
  44. cstb.p gr9,@(gr4,gr0) ,cc7,#1
  45. csubcc gr5,gr6,gr5 ,cc7,#1 ; also set ICC3
  46. cadd.p gr4,gr6,gr4 ,cc7,#1
  47. beqlr icc3,#0
  48. # conditionally write a word to 4b-align the address
  49. andicc.p gr4,#2,gr0,icc0
  50. subicc gr5,#2,gr0,icc1
  51. setlos.p #2,gr6
  52. ckne icc0,cc7
  53. slli.p gr9,#8,gr12 ; need to double up the pattern
  54. cknc icc1,cc5
  55. or.p gr9,gr12,gr12
  56. andcr cc7,cc5,cc7
  57. csth.p gr12,@(gr4,gr0) ,cc7,#1
  58. csubcc gr5,gr6,gr5 ,cc7,#1 ; also set ICC3
  59. cadd.p gr4,gr6,gr4 ,cc7,#1
  60. beqlr icc3,#0
  61. # conditionally write a dword to 8b-align the address
  62. andicc.p gr4,#4,gr0,icc0
  63. subicc gr5,#4,gr0,icc1
  64. setlos.p #4,gr6
  65. ckne icc0,cc7
  66. slli.p gr12,#16,gr13 ; need to quadruple-up the pattern
  67. cknc icc1,cc5
  68. or.p gr13,gr12,gr12
  69. andcr cc7,cc5,cc7
  70. cst.p gr12,@(gr4,gr0) ,cc7,#1
  71. csubcc gr5,gr6,gr5 ,cc7,#1 ; also set ICC3
  72. cadd.p gr4,gr6,gr4 ,cc7,#1
  73. beqlr icc3,#0
  74. or.p gr12,gr12,gr13 ; need to octuple-up the pattern
  75. # the address is now 8b-aligned - loop around writing 64b chunks
  76. setlos #8,gr7
  77. subi.p gr4,#8,gr4 ; store with update index does weird stuff
  78. setlos #64,gr6
  79. subicc gr5,#64,gr0,icc0
  80. 0: cknc icc0,cc7
  81. cstdu gr12,@(gr4,gr7) ,cc7,#1
  82. cstdu gr12,@(gr4,gr7) ,cc7,#1
  83. cstdu gr12,@(gr4,gr7) ,cc7,#1
  84. cstdu gr12,@(gr4,gr7) ,cc7,#1
  85. cstdu gr12,@(gr4,gr7) ,cc7,#1
  86. cstdu.p gr12,@(gr4,gr7) ,cc7,#1
  87. csubcc gr5,gr6,gr5 ,cc7,#1 ; also set ICC3
  88. cstdu.p gr12,@(gr4,gr7) ,cc7,#1
  89. subicc gr5,#64,gr0,icc0
  90. cstdu.p gr12,@(gr4,gr7) ,cc7,#1
  91. beqlr icc3,#0
  92. bnc icc0,#2,0b
  93. # now do 32-byte remnant
  94. subicc.p gr5,#32,gr0,icc0
  95. setlos #32,gr6
  96. cknc icc0,cc7
  97. cstdu.p gr12,@(gr4,gr7) ,cc7,#1
  98. csubcc gr5,gr6,gr5 ,cc7,#1 ; also set ICC3
  99. cstdu.p gr12,@(gr4,gr7) ,cc7,#1
  100. setlos #16,gr6
  101. cstdu.p gr12,@(gr4,gr7) ,cc7,#1
  102. subicc gr5,#16,gr0,icc0
  103. cstdu.p gr12,@(gr4,gr7) ,cc7,#1
  104. beqlr icc3,#0
  105. # now do 16-byte remnant
  106. cknc icc0,cc7
  107. cstdu.p gr12,@(gr4,gr7) ,cc7,#1
  108. csubcc gr5,gr6,gr5 ,cc7,#1 ; also set ICC3
  109. cstdu.p gr12,@(gr4,gr7) ,cc7,#1
  110. beqlr icc3,#0
  111. # now do 8-byte remnant
  112. subicc gr5,#8,gr0,icc1
  113. cknc icc1,cc7
  114. cstdu.p gr12,@(gr4,gr7) ,cc7,#1
  115. csubcc gr5,gr7,gr5 ,cc7,#1 ; also set ICC3
  116. setlos.p #4,gr7
  117. beqlr icc3,#0
  118. # now do 4-byte remnant
  119. subicc gr5,#4,gr0,icc0
  120. addi.p gr4,#4,gr4
  121. cknc icc0,cc7
  122. cstu.p gr12,@(gr4,gr7) ,cc7,#1
  123. csubcc gr5,gr7,gr5 ,cc7,#1 ; also set ICC3
  124. subicc.p gr5,#2,gr0,icc1
  125. beqlr icc3,#0
  126. # now do 2-byte remnant
  127. setlos #2,gr7
  128. addi.p gr4,#2,gr4
  129. cknc icc1,cc7
  130. csthu.p gr12,@(gr4,gr7) ,cc7,#1
  131. csubcc gr5,gr7,gr5 ,cc7,#1 ; also set ICC3
  132. subicc.p gr5,#1,gr0,icc0
  133. beqlr icc3,#0
  134. # now do 1-byte remnant
  135. setlos #0,gr7
  136. addi.p gr4,#2,gr4
  137. cknc icc0,cc7
  138. cstb.p gr12,@(gr4,gr0) ,cc7,#1
  139. bralr
  140. .size __memset, .-__memset
  141. strong_alias(__memset,memset)