sourcelibs/Strings/src/Strings.xtl

1⍝# Strings: text functions -- case, trimming, words, split and join, 2⍝# search, replace, padding. 3⍝# Import it with an alias of your choice: "t:" u_se< "Strings". 4⍝# Put libs/Strings/src on XETAL_PATH ("just path"); the reference is libs/Strings/docs. 5⍝# Names with l: are exported; those under h: are private to this file. 6⍝# 7⍝# A string is a Char vector; a list of strings is a nested vector 8⍝# (Box Char), as "ab" "cde" is. Patterns, separators and widths go on 9⍝# the left, the text on the right: "," t:s_plit "a,b". 10 11blanks ← " \t\n" ⍝ what trimming removes 12 13⍝## Case and trimming 14 15⍝# Case by character code ([]U_CS): a letter's two cases are 32 apart. 16⍝# ASCII letters only: []U_CHAR takes codes 0 to 127 (ask X16). 17ˡu̲pper ← { t → c ← ⎕U̲CS t◆ ⎕U̲CHAR c − 32 × (c ≥ 97) ∧ c ≤ 122 } ⍝ "Hello" to "HELLO" 18ˡl̲ower ← { t → c ← ⎕U̲CS t◆ ⎕U̲CHAR c + 32 × (c ≥ 65) ∧ c ≤ 90 } ⍝ "Hello" to "hello" 19 20⍝# Trimming: blanks (space, tab, newline) off the start, the end, or both. 21ˡt̲rimStart ← { t → (n̲ot '∧ s̲\ t m̲ember? blanks) r̲eplicate t } 22ˡt̲rimEnd ← { t → r̲ev ˡt̲rimStart r̲ev t } 23ˡt̲rim ← { t → ˡt̲rimEnd ˡt̲rimStart t } 24 25⍝## Words and joining 26 27⍝# w_ords t: the words of t, split at runs of blanks (dfns words). 28ˡw̲ords ← { t → (n̲ot t m̲ember? blanks) p̲artition t } 29 30⍝# sep j_oin list: the strings of list with sep between them. 31ˡj̲oin ← { sep b → 32 0 = t̲ally b ? "" 33 d̲isclose '{ e̲nclose (d̲isclose ⍺) c̲at sep c̲at d̲isclose ⍵ } r̲/ b 34} 35 36⍝# s_queeze t: the words of t one space apart (J's deb). 37ˡs̲queeze ← { t → " " ˡj̲oin ˡw̲ords t } 38 39⍝## Finding and splitting 40 41⍝# p f_ind t: where the pattern p starts in t, overlaps included (APL's 42⍝# find, dfns ss's search). 43ˡf̲ind ← { p t → 44 m ← t̲ally p 45 n ← t̲ally t 46 (m = 0) ∨ m > n ? 0 r̲eshape 0 47 cells ← ((r̲ange 1 + n − m) '+ t̲able o̲ffsets m) s̲elect t 48 w̲here '∧ r̲/₂ cells = (s̲hape cells) r̲eshape p 49} 50 51⍝# Starts ps of matches of length m, keeping only those that do not 52⍝# overlap an earlier one (left to right). 53ʰa̲part ← { m ps → 54 0 = t̲ally ps ? ps 55 f ← f̲irst ps 56 f c̲at m ʰa̲part (ps ≥ f + m) r̲eplicate ps 57} 58 59⍝# p o_ccurrences t: how many times p occurs in t, without overlaps. 60ˡo̲ccurrences ← { p t → t̲ally (t̲ally p) ʰa̲part p ˡf̲ind t } 61 62⍝# sep s_plit t: the pieces of t between the separators (any length); 63⍝# empty pieces are kept, so "a,,b" has three ("a" "" "b"). 64ˡs̲plit ← { sep t → 65 m ← t̲ally sep 66 m = 0 ? 1 r̲eshape e̲nclose t 67 ps ← m ʰa̲part sep ˡf̲ind t 68 starts ← 1 c̲at ps + m 69 ends ← ps c̲at 1 + t̲ally t 70 '{ i → ((i s̲elect ends) − i s̲elect starts) t̲ake ((i s̲elect starts) − 1) d̲rop t } m̲ap r̲ange t̲ally starts 71} 72 73⍝# l_ines t: the lines of t (a final newline does not make an empty line). 74ˡl̲ines ← { t → 75 e ← (0 < t̲ally t) ∧ ("\n" m̲atch -1 t̲ake t) 76 "\n" ˡs̲plit (n̲eg e) d̲rop t 77} 78 79⍝# Old and new as a pair: "cat" "dog" t:r_eplace t, without overlaps 80⍝# (dfns ss). 81ˡr̲eplace ← { pair t → 82 (d̲isclose 2 s̲elect pair) ˡj̲oin (d̲isclose 1 s̲elect pair) ˡs̲plit t 83} 84 85⍝## Prefixes 86 87⍝# p p_refix? t, p s_uffix? t, p i_nfix? t: whether p begins, ends or 88⍝# occurs in t. 89ˡp̲refix? ← { p t → (t̲ally p) > t̲ally t ? 0 = 1◆ p m̲atch (t̲ally p) t̲ake t } 90ˡs̲uffix? ← { p t → (t̲ally p) > t̲ally t ? 0 = 1◆ p m̲atch (n̲eg t̲ally p) t̲ake t } 91ˡi̲nfix? ← { p t → 0 < t̲ally p ˡf̲ind t } 92 93⍝## Padding 94 95⍝# n p_adLeft t, n p_adRight t, n c_enter t: t in a field n wide, filled 96⍝# with spaces (right-aligned, left-aligned, centered); a longer t is 97⍝# kept whole. 98ˡp̲adLeft ← { n t → (n̲eg n m̲ax t̲ally t) t̲ake t } 99ˡp̲adRight ← { n t → (n m̲ax t̲ally t) t̲ake t } 100ˡc̲enter ← { n t → 101 w ← n m̲ax t̲ally t 102 w t̲ake (n̲eg (t̲ally t) + (w − t̲ally t) d̲iv 2) t̲ake t 103} 104 105⍝# n r_epeat t: t, n times over. 106ˡr̲epeat ← { n t → 0 = t̲ally t ? t◆ (n × t̲ally t) r̲eshape t } 107 108⍝# m_ix list: the texts of a list as a character matrix, one per row, 109⍝# padded with spaces to the longest (APL2's disclose of a list of 110⍝# strings, "mix"). 111ˡm̲ix ← { b → 112 w ← 'm̲ax r̲/ '{ t̲ally d̲isclose ⍵ } e̲ach b 113 all ← d̲isclose '{ e̲nclose (d̲isclose ⍺) c̲at d̲isclose ⍵ } r̲/ '{ s → w t̲ake d̲isclose s } m̲ap b 114 ((t̲ally b) c̲at w) r̲eshape all 115}