sourcelibs/Strings/src/Strings.xtl
1⍝# Strings: text functions -- case, trimming, words, split and join,
2⍝# search, replace, padding.
3⍝# Import it with an alias of your choice: "t:" u_se< "Strings".
4⍝# Put libs/Strings/src on XETAL_PATH ("just path"); the reference is libs/Strings/docs.
5⍝# Names with l: are exported; those under h: are private to this file.
6⍝#
7⍝# A string is a Char vector; a list of strings is a nested vector
8⍝# (Box Char), as "ab" "cde" is. Patterns, separators and widths go on
9⍝# the left, the text on the right: "," t:s_plit "a,b".
10
11blanks ← " \t\n" ⍝ what trimming removes
12
13⍝## Case and trimming
14
15⍝# Case by character code ([]U_CS): a letter's two cases are 32 apart.
16⍝# ASCII letters only: []U_CHAR takes codes 0 to 127 (ask X16).
17ˡu̲pper ← { t → c ← ⎕U̲CS t◆ ⎕U̲CHAR c − 32 × (c ≥ 97) ∧ c ≤ 122 } ⍝ "Hello" to "HELLO"
18ˡl̲ower ← { t → c ← ⎕U̲CS t◆ ⎕U̲CHAR c + 32 × (c ≥ 65) ∧ c ≤ 90 } ⍝ "Hello" to "hello"
19
20⍝# Trimming: blanks (space, tab, newline) off the start, the end, or both.
21ˡt̲rimStart ← { t → (n̲ot '∧ s̲\ t m̲ember? blanks) r̲eplicate t }
22ˡt̲rimEnd ← { t → r̲ev ˡt̲rimStart r̲ev t }
23ˡt̲rim ← { t → ˡt̲rimEnd ˡt̲rimStart t }
24
25⍝## Words and joining
26
27⍝# w_ords t: the words of t, split at runs of blanks (dfns words).
28ˡw̲ords ← { t → (n̲ot t m̲ember? blanks) p̲artition t }
29
30⍝# sep j_oin list: the strings of list with sep between them.
31ˡj̲oin ← { sep b →
32 0 = t̲ally b ? ""
33 d̲isclose '{ e̲nclose (d̲isclose ⍺) c̲at sep c̲at d̲isclose ⍵ } r̲/ b
34}
35
36⍝# s_queeze t: the words of t one space apart (J's deb).
37ˡs̲queeze ← { t → " " ˡj̲oin ˡw̲ords t }
38
39⍝## Finding and splitting
40
41⍝# p f_ind t: where the pattern p starts in t, overlaps included (APL's
42⍝# find, dfns ss's search).
43ˡf̲ind ← { p t →
44 m ← t̲ally p
45 n ← t̲ally t
46 (m = 0) ∨ m > n ? 0 r̲eshape 0
47 cells ← ((r̲ange 1 + n − m) '+ t̲able o̲ffsets m) s̲elect t
48 w̲here '∧ r̲/₂ cells = (s̲hape cells) r̲eshape p
49}
50
51⍝# Starts ps of matches of length m, keeping only those that do not
52⍝# overlap an earlier one (left to right).
53ʰa̲part ← { m ps →
54 0 = t̲ally ps ? ps
55 f ← f̲irst ps
56 f c̲at m ʰa̲part (ps ≥ f + m) r̲eplicate ps
57}
58
59⍝# p o_ccurrences t: how many times p occurs in t, without overlaps.
60ˡo̲ccurrences ← { p t → t̲ally (t̲ally p) ʰa̲part p ˡf̲ind t }
61
62⍝# sep s_plit t: the pieces of t between the separators (any length);
63⍝# empty pieces are kept, so "a,,b" has three ("a" "" "b").
64ˡs̲plit ← { sep t →
65 m ← t̲ally sep
66 m = 0 ? 1 r̲eshape e̲nclose t
67 ps ← m ʰa̲part sep ˡf̲ind t
68 starts ← 1 c̲at ps + m
69 ends ← ps c̲at 1 + t̲ally t
70 '{ i → ((i s̲elect ends) − i s̲elect starts) t̲ake ((i s̲elect starts) − 1) d̲rop t } m̲ap r̲ange t̲ally starts
71}
72
73⍝# l_ines t: the lines of t (a final newline does not make an empty line).
74ˡl̲ines ← { t →
75 e ← (0 < t̲ally t) ∧ ("\n" m̲atch -1 t̲ake t)
76 "\n" ˡs̲plit (n̲eg e) d̲rop t
77}
78
79⍝# Old and new as a pair: "cat" "dog" t:r_eplace t, without overlaps
80⍝# (dfns ss).
81ˡr̲eplace ← { pair t →
82 (d̲isclose 2 s̲elect pair) ˡj̲oin (d̲isclose 1 s̲elect pair) ˡs̲plit t
83}
84
85⍝## Prefixes
86
87⍝# p p_refix? t, p s_uffix? t, p i_nfix? t: whether p begins, ends or
88⍝# occurs in t.
89ˡp̲refix? ← { p t → (t̲ally p) > t̲ally t ? 0 = 1◆ p m̲atch (t̲ally p) t̲ake t }
90ˡs̲uffix? ← { p t → (t̲ally p) > t̲ally t ? 0 = 1◆ p m̲atch (n̲eg t̲ally p) t̲ake t }
91ˡi̲nfix? ← { p t → 0 < t̲ally p ˡf̲ind t }
92
93⍝## Padding
94
95⍝# n p_adLeft t, n p_adRight t, n c_enter t: t in a field n wide, filled
96⍝# with spaces (right-aligned, left-aligned, centered); a longer t is
97⍝# kept whole.
98ˡp̲adLeft ← { n t → (n̲eg n m̲ax t̲ally t) t̲ake t }
99ˡp̲adRight ← { n t → (n m̲ax t̲ally t) t̲ake t }
100ˡc̲enter ← { n t →
101 w ← n m̲ax t̲ally t
102 w t̲ake (n̲eg (t̲ally t) + (w − t̲ally t) d̲iv 2) t̲ake t
103}
104
105⍝# n r_epeat t: t, n times over.
106ˡr̲epeat ← { n t → 0 = t̲ally t ? t◆ (n × t̲ally t) r̲eshape t }
107
108⍝# m_ix list: the texts of a list as a character matrix, one per row,
109⍝# padded with spaces to the longest (APL2's disclose of a list of
110⍝# strings, "mix").
111ˡm̲ix ← { b →
112 w ← 'm̲ax r̲/ '{ t̲ally d̲isclose ⍵ } e̲ach b
113 all ← d̲isclose '{ e̲nclose (d̲isclose ⍺) c̲at d̲isclose ⍵ } r̲/ '{ s → w t̲ake d̲isclose s } m̲ap b
114 ((t̲ally b) c̲at w) r̲eshape all
115}