Strings / String Utilities
3.1 String Utilities
Common string functions, many of which have standard library counterparts. The implementations below are educational rather than heavily optimized, and often depend on std::string operations with unspecified complexity.
Implementation
#include <cctype>
#include <cstddef>
#include <istream>
#include <regex>
#include <sstream>
#include <stdexcept>
#include <string>
#include <vector>
using std::string;
Integer Conversion:
to_str(i)returns the string representation of integeri, much likestd::to_string().to_int(s)returns the integer representation of strings, much likestd::atoi(), except here we handle special cases of overflow by throwing an exception.itoa(value, str, base = 10)implements the non-standard C function which convertsvalueinto a C string, storing it into pointerstrin the givenbase. For more generalized base conversion, see the math utilities section.is_integer(s)returns whethersis a valid base-10 integer literal with an optional sign.is_number(s)returns whethersis a decimal number literal with optional sign, decimal point, and exponent.
Implementation
template<typename Int>
string to_str(Int i) {
std::ostringstream oss;
oss << i;
return oss.str();
}
int to_int(const string &s) {
std::istringstream iss(s);
int res;
if (!(iss >> res) || !(iss >> std::ws).eof()) {
throw std::runtime_error("to_int failed");
}
return res;
}
char *itoa(int value, char *str, int base = 10) {
if (base < 2 || base > 36) {
*str = '\0';
return str;
}
char *ptr = str, *ptr1 = str, tmp_c;
int tmp_v;
do {
tmp_v = value;
value /= base;
*ptr++ =
"zyxwvutsrqponmlkjihgfedcba9876543210123456789"
"abcdefghijklmnopqrstuvwxyz"[35 + (tmp_v - value * base)];
} while (value);
if (tmp_v < 0) {
*ptr++ = '-';
}
for (*ptr-- = '\0'; ptr1 < ptr; *ptr1++ = tmp_c) {
tmp_c = *ptr;
*ptr-- = *ptr1;
}
return str;
}
bool is_integer(const string &s) {
// return std::regex_match(s, std::regex(R"([+-]?\d+)"));
int i = (s.empty() || (s[0] != '-' && s[0] != '+')) ? 0 : 1;
if (i == static_cast<int>(s.size())) {
return false;
}
for (; i < static_cast<int>(s.size()); i++) {
if (!isdigit(static_cast<unsigned char>(s[i]))) {
return false;
}
}
return true;
}
bool is_number(const string &s) {
// return std::regex_match(s, std::regex(R"([+-]?(\d+(\.\d*)?|\.\d+)([eE][+-]?\d+)?)"));
int i = (s.empty() || (s[0] != '-' && s[0] != '+')) ? 0 : 1;
bool seen_digit = false, seen_dot = false;
for (; i < static_cast<int>(s.size()); i++) {
if (isdigit(static_cast<unsigned char>(s[i]))) {
seen_digit = true;
} else if (s[i] == '.' && !seen_dot) {
seen_dot = true;
} else {
break;
}
}
if (!seen_digit) {
return false;
}
if (i < static_cast<int>(s.size()) && (s[i] == 'e' || s[i] == 'E')) {
i++;
if (i < static_cast<int>(s.size()) && (s[i] == '-' || s[i] == '+')) {
i++;
}
bool seen_exp_digit = false;
for (; i < static_cast<int>(s.size()); i++) {
if (!isdigit(static_cast<unsigned char>(s[i]))) {
return false;
}
seen_exp_digit = true;
}
return seen_exp_digit;
}
return i == static_cast<int>(s.size());
}
Case Conversion:
to_upper(s)returnsswith all alphabetical characters converted to uppercase.to_lower(s)returnsswith all alphabetical characters converted to lowercase.to_title(s)returns the title case representation of strings, where the first letter of every word (consecutive alphabetical characters) is capitalized.
Implementation
string to_upper(const string &s) {
string res;
res.reserve(s.size());
for (unsigned char c : s) {
res.push_back(static_cast<char>(toupper(c)));
}
return res;
}
string to_lower(const string &s) {
string res;
res.reserve(s.size());
for (unsigned char c : s) {
res.push_back(static_cast<char>(tolower(c)));
}
return res;
}
string to_title(const string &s) {
string res;
unsigned char prev = '\0';
for (unsigned char c : s) {
res.push_back(isalpha(prev) ? static_cast<char>(tolower(c)) : static_cast<char>(toupper(c)));
prev = res.back();
}
return res;
}
Stripping:
lstrip(s, delim = WHITESPACE)strips the left side ofsin-place (that is, the input is modified) using the given delimiters and returns a reference to the stripped string.rstrip(s, delim = WHITESPACE)strips the right side ofsin-place using the given delimiters and returns a reference to the stripped string.strip(s, delim = WHITESPACE)strips both sides ofsin-place and returns a reference to the stripped string.ltrimmed(s, delim = WHITESPACE),rtrimmed(s, delim = WHITESPACE), andtrimmed(s, delim = WHITESPACE)do not modifysand return stripped copies.
Implementation
const string WHITESPACE = " \n\t\v\f\r";
string &lstrip(string &s, const string &delim = WHITESPACE) {
std::size_t pos = s.find_first_not_of(delim);
if (pos != string::npos) {
s.erase(0, pos);
} else {
s.clear();
}
return s;
}
string &rstrip(string &s, const string &delim = WHITESPACE) {
std::size_t pos = s.find_last_not_of(delim);
if (pos != string::npos) {
s.erase(pos + 1);
} else {
s.clear();
}
return s;
}
string &strip(string &s, const string &delim = WHITESPACE) {
return lstrip(rstrip(s, delim), delim);
}
string ltrimmed(string s, const string &delim = WHITESPACE) {
return lstrip(s, delim);
}
string rtrimmed(string s, const string &delim = WHITESPACE) {
return rstrip(s, delim);
}
string trimmed(string s, const string &delim = WHITESPACE) {
return strip(s, delim);
}
Find and Replace:
starts_with(s, prefix)returns whethersbegins withprefix.ends_with(s, suffix)returns whethersends withsuffix.contains(s, needle)returns whetherneedleoccurs ins.find_all(haystack, needle)returns a vector of all positions where the nonempty stringneedleappears in the stringhaystack. Matches may overlap; e.g.find_all("aaa", "aa")returns{0, 1}. An emptyneedleproduces no matches.count(haystack, needle)returns the number of non-overlapping occurrences ofneedleinhaystack, like Python'sstr.count. This differs fromfind_all().size(), which counts overlapping matches; e.g.count("aaa", "aa")returns $1$, butfind_all("aaa", "aa")finds $2$.replace(s, old, replacement)returns a copy ofswith all occurrences of the stringoldreplaced with the givenreplacement.regex_find_all(s, pattern)returns every substring ofsmatched bypattern.regex_groups(s, pattern)returns the capture groups of the first match ofpatternins, excluding the full match. If there is no match, returns an empty vector.regex_replace_all(s, pattern, replacement)returns a copy ofswith every regex match replaced byreplacement, which may refer to capture groups as$1,$2, and so on.
Implementation
bool starts_with(const string &s, const string &prefix) {
return s.size() >= prefix.size() && s.compare(0, prefix.size(), prefix) == 0;
}
bool ends_with(const string &s, const string &suffix) {
return s.size() >= suffix.size() &&
s.compare(s.size() - suffix.size(), suffix.size(), suffix) == 0;
}
bool contains(const string &s, const string &needle) {
return s.find(needle) != string::npos;
}
std::vector<int> find_all(const string &haystack, const string &needle) {
std::vector<int> res;
if (needle.empty()) {
return res;
}
std::size_t pos = haystack.find(needle, 0);
while (pos != string::npos) {
res.push_back(static_cast<int>(pos));
pos = haystack.find(needle, pos + 1);
}
return res;
}
int count(const string &haystack, const string &needle) {
if (needle.empty()) {
return 0;
}
int res = 0;
std::size_t pos = haystack.find(needle, 0);
while (pos != string::npos) {
res++;
pos = haystack.find(needle, pos + needle.size());
}
return res;
}
string replace(const string &s, const string &old, const string &replacement) {
if (old.empty()) {
return s;
}
string res(s);
std::size_t pos = 0;
while ((pos = res.find(old, pos)) != string::npos) {
res.replace(pos, old.length(), replacement);
pos += replacement.length();
}
return res;
}
std::vector<string> regex_find_all(const string &s, const string &pattern) {
std::regex re(pattern);
return {std::sregex_token_iterator(s.begin(), s.end(), re, 0), std::sregex_token_iterator()};
}
std::vector<string> regex_groups(const string &s, const string &pattern) {
std::smatch match;
if (!std::regex_search(s, match, std::regex(pattern))) {
return {};
}
std::vector<string> res;
for (int i = 1; i < static_cast<int>(match.size()); i++) {
res.push_back(match[i].str());
}
return res;
}
string regex_replace_all(const string &s, const string &pattern, const string &replacement) {
return std::regex_replace(s, std::regex(pattern), replacement);
}
Joining and Splitting:
join(v, delim = " ")returns the strings in vectorvconcatenated, separated by the given delimiter.repeat(s, n)returnssconcatenated with itselfntimes, like Python'ss * n, or an empty string ifnis nonpositive. For a single repeated character, preferstd::string(n, ch).ljust(s, width, ch = ' ')returnssleft-justified (padded on the right withch) to at leastwidthcharacters, like Python'sstr.ljust.rjust(s, width, ch = ' ')returnssright-justified (padded on the left withch) to at leastwidthcharacters, like Python'sstr.rjust.center(s, width, ch = ' ')returnsscentered in a field of at leastwidthcharacters by padding both sides withch, like Python'sstr.center. If the padding is uneven, the extra character goes on the right.split(s, char delim)returns a vector of tokens ofs, split on a single character delimiter. Empty tokens are skipped, sosplit("a::b", ':')returns{"a", "b"}, not{"a", "", "b"}.split(s, string delim = WHITESPACE)returns a vector of tokens ofs, split on a set of many possible single character delimiters. All characters ofdelimwill be removed froms, and the remaining token(s) ofswill be added sequentially to a vector and returned. As with the first version, empty tokens are skipped. For example,split("a::b", ":")returns{"a", "b"}, not{"a", "", "b"}.explode(s, delim)returns a vector of tokens ofs, split on the entire delimiter stringdelim. Unlike thesplit()functions above,delimis treated as a contiguous boundary string, not merely a set of possible boundary characters. This will not skip empty tokens. For example,explode("a::::b", "::")yields{"a", "", "b"}, not{"a", "b"}. An empty delimiter returns{s}.explode(s, char delim)is the single-character delimiter version that also preserves empty tokens.
Implementation
string join(const std::vector<string> &v, const string &delim = " ") {
if (v.empty()) {
return "";
}
string res;
std::size_t total = delim.size() * (v.size() - 1);
for (const string &s : v) {
total += s.size();
}
res.reserve(total);
for (int i = 0; i < static_cast<int>(v.size()); i++) {
res += (i > 0 ? delim : "") + v[i];
}
return res;
}
string repeat(const string &s, int n) {
if (n <= 0) {
return "";
}
string res;
res.reserve(s.size() * n);
for (int i = 0; i < n; i++) {
res += s;
}
return res;
}
string ljust(const string &s, int width, char ch = ' ') {
int pad = width - static_cast<int>(s.size());
return pad > 0 ? s + string(pad, ch) : s;
}
string rjust(const string &s, int width, char ch = ' ') {
int pad = width - static_cast<int>(s.size());
return pad > 0 ? string(pad, ch) + s : s;
}
string center(const string &s, int width, char ch = ' ') {
int pad = width - static_cast<int>(s.size());
if (pad <= 0) {
return s;
}
int left = pad / 2;
return string(left, ch) + s + string(pad - left, ch);
}
std::vector<string> split(const string &s, char delim) {
std::vector<string> res;
std::stringstream ss(s);
string curr;
while (std::getline(ss, curr, delim)) {
if (!curr.empty()) {
res.push_back(curr);
}
}
return res;
}
std::vector<string> split(const string &s, const string &delim = WHITESPACE) {
std::vector<string> res;
string curr;
for (char c : s) {
if (delim.find(c) == string::npos) {
curr += c;
} else if (!curr.empty()) {
res.push_back(curr);
curr = "";
}
}
if (!curr.empty()) {
res.push_back(curr);
}
return res;
}
std::vector<string> explode(const string &s, const string &delim) {
if (delim.empty()) {
return {s};
}
std::vector<string> res;
std::size_t last = 0, next = 0;
while ((next = s.find(delim, last)) != string::npos) {
res.push_back(s.substr(last, next - last));
last = next + delim.size();
}
res.push_back(s.substr(last));
return res;
}
std::vector<string> explode(const string &s, char delim) {
std::vector<string> res;
std::size_t last = 0, next = 0;
while ((next = s.find(delim, last)) != string::npos) {
res.push_back(s.substr(last, next - last));
last = next + 1;
}
res.push_back(s.substr(last));
return res;
}
Example Usage
#include <cassert>
using namespace std;
int main() {
assert(to_str(123) + "4" == "1234");
assert(to_int("1234") == 1234);
bool threw = false;
try {
to_int("123x");
} catch (const runtime_error &) {
threw = true;
}
assert(threw);
vector<char> buffer(50);
assert(string(itoa(1750, buffer.data(), 10)) == "1750");
assert(string(itoa(1750, buffer.data(), 16)) == "6d6");
assert(string(itoa(1750, buffer.data(), 2)) == "11011010110");
assert(is_integer("-123") && !is_integer("12.3"));
assert(is_number("-.5e+2") && is_number("123.") && !is_number("1e"));
assert(to_upper("Hello world") == "HELLO WORLD");
assert(to_lower("Hello World") == "hello world");
assert(to_title("hello world") == "Hello World");
string s(" abc \n");
string t = s;
assert(lstrip(s) == "abc \n");
assert(rstrip(s) == strip(t));
assert(trimmed(" \t abc \n") == "abc");
assert(trimmed("xxx", "x").empty());
vector<int> pos;
pos.push_back(0);
pos.push_back(7);
assert(starts_with("abracadabra", "abra"));
assert(ends_with("abracadabra", "dabra"));
assert(contains("abracadabra", "cad"));
assert(find_all("abracadabra", "ab") == pos);
assert(count("abracadabra", "ab") == 2);
assert(count("aaa", "aa") == 1);
assert((find_all("aaa", "aa") == vector<int>{0, 1}));
assert(find_all("abc", "").empty());
assert(replace("abcdabba", "ab", "00") == "00cd00ba");
assert(join(regex_find_all("a12b345", "[0-9]+"), "|") == "12|345");
assert(join(regex_groups("abc-123", "([a-z]+)-([0-9]+)"), "|") == "abc|123");
assert(regex_replace_all("a12b345", "[0-9]+", "#") == "a#b#");
assert(regex_replace_all("2026-08-04", R"((\d{4})-(\d{2})-(\d{2}))", "$2/$3/$1") == "08/04/2026");
assert(repeat("ab", 3) == "ababab");
assert(repeat("ab", -1).empty());
assert(rjust("42", 5, '0') == "00042");
assert(ljust("hi", 4, '.') == "hi..");
assert(center("ab", 5, '*') == "*ab**");
assert(join(split("a\nb\ncde\nf", '\n'), "|") == "a|b|cde|f"); // split v1
assert(join(split("a::b", ':'), "|") == "a|b"); // split v1 skips empty tokens
assert(join(split("a::b,cde:,f", ":,"), "|") == "a|b|cde|f"); // split v2
assert(join(explode("a..b.cde....f", ".."), "|") == "a|b.cde||f");
assert((explode("abc", "") == vector<string>{"abc"}));
assert(join(explode("a::b:", ':'), "|") == "a||b|");
return 0;
}
/*
Common string functions, many of which have standard library counterparts. The implementations below
are educational rather than heavily optimized, and often depend on `std::string` operations with
unspecified complexity.
Time Complexity:
- O(n) per call to most operations, where $n$ is the length of the input string or total length of
all input strings. Exceptions are noted with the individual functions below.
Space Complexity:
- O(n) auxiliary for operations that return a new string or vector of strings.
- O(1) auxiliary for in-place operations.
*/
#include <cctype>
#include <cstddef>
#include <istream>
#include <regex>
#include <sstream>
#include <stdexcept>
#include <string>
#include <vector>
using std::string;
/*
Integer Conversion:
- `to_str(i)` returns the string representation of integer `i`, much like `std::to_string()`.
- `to_int(s)` returns the integer representation of string `s`, much like `std::atoi()`, except here
we handle special cases of overflow by throwing an exception.
- `itoa(value, str, base = 10)` implements the non-standard C function which converts `value` into a
C string, storing it into pointer `str` in the given `base`. For more generalized base conversion,
see the math utilities section.
- `is_integer(s)` returns whether `s` is a valid base-10 integer literal with an optional sign.
- `is_number(s)` returns whether `s` is a decimal number literal with optional sign, decimal point,
and exponent.
*/
template<typename Int>
string to_str(Int i) {
std::ostringstream oss;
oss << i;
return oss.str();
}
int to_int(const string &s) {
std::istringstream iss(s);
int res;
if (!(iss >> res) || !(iss >> std::ws).eof()) {
throw std::runtime_error("to_int failed");
}
return res;
}
char *itoa(int value, char *str, int base = 10) {
if (base < 2 || base > 36) {
*str = '\0';
return str;
}
char *ptr = str, *ptr1 = str, tmp_c;
int tmp_v;
do {
tmp_v = value;
value /= base;
*ptr++ =
"zyxwvutsrqponmlkjihgfedcba9876543210123456789"
"abcdefghijklmnopqrstuvwxyz"[35 + (tmp_v - value * base)];
} while (value);
if (tmp_v < 0) {
*ptr++ = '-';
}
for (*ptr-- = '\0'; ptr1 < ptr; *ptr1++ = tmp_c) {
tmp_c = *ptr;
*ptr-- = *ptr1;
}
return str;
}
bool is_integer(const string &s) {
// return std::regex_match(s, std::regex(R"([+-]?\d+)"));
int i = (s.empty() || (s[0] != '-' && s[0] != '+')) ? 0 : 1;
if (i == static_cast<int>(s.size())) {
return false;
}
for (; i < static_cast<int>(s.size()); i++) {
if (!isdigit(static_cast<unsigned char>(s[i]))) {
return false;
}
}
return true;
}
bool is_number(const string &s) {
// return std::regex_match(s, std::regex(R"([+-]?(\d+(\.\d*)?|\.\d+)([eE][+-]?\d+)?)"));
int i = (s.empty() || (s[0] != '-' && s[0] != '+')) ? 0 : 1;
bool seen_digit = false, seen_dot = false;
for (; i < static_cast<int>(s.size()); i++) {
if (isdigit(static_cast<unsigned char>(s[i]))) {
seen_digit = true;
} else if (s[i] == '.' && !seen_dot) {
seen_dot = true;
} else {
break;
}
}
if (!seen_digit) {
return false;
}
if (i < static_cast<int>(s.size()) && (s[i] == 'e' || s[i] == 'E')) {
i++;
if (i < static_cast<int>(s.size()) && (s[i] == '-' || s[i] == '+')) {
i++;
}
bool seen_exp_digit = false;
for (; i < static_cast<int>(s.size()); i++) {
if (!isdigit(static_cast<unsigned char>(s[i]))) {
return false;
}
seen_exp_digit = true;
}
return seen_exp_digit;
}
return i == static_cast<int>(s.size());
}
/*
Case Conversion:
- `to_upper(s)` returns `s` with all alphabetical characters converted to uppercase.
- `to_lower(s)` returns `s` with all alphabetical characters converted to lowercase.
- `to_title(s)` returns the title case representation of string `s`, where the first letter of every
word (consecutive alphabetical characters) is capitalized.
*/
string to_upper(const string &s) {
string res;
res.reserve(s.size());
for (unsigned char c : s) {
res.push_back(static_cast<char>(toupper(c)));
}
return res;
}
string to_lower(const string &s) {
string res;
res.reserve(s.size());
for (unsigned char c : s) {
res.push_back(static_cast<char>(tolower(c)));
}
return res;
}
string to_title(const string &s) {
string res;
unsigned char prev = '\0';
for (unsigned char c : s) {
res.push_back(isalpha(prev) ? static_cast<char>(tolower(c)) : static_cast<char>(toupper(c)));
prev = res.back();
}
return res;
}
/*
Stripping:
- `lstrip(s, delim = WHITESPACE)` strips the left side of `s` in-place (that is, the input is
modified) using the given delimiters and returns a reference to the stripped string.
- `rstrip(s, delim = WHITESPACE)` strips the right side of `s` in-place using the given delimiters
and returns a reference to the stripped string.
- `strip(s, delim = WHITESPACE)` strips both sides of `s` in-place and returns a reference to the
stripped string.
- `ltrimmed(s, delim = WHITESPACE)`, `rtrimmed(s, delim = WHITESPACE)`, and
`trimmed(s, delim = WHITESPACE)` do not modify `s` and return stripped copies.
*/
const string WHITESPACE = " \n\t\v\f\r";
string &lstrip(string &s, const string &delim = WHITESPACE) {
std::size_t pos = s.find_first_not_of(delim);
if (pos != string::npos) {
s.erase(0, pos);
} else {
s.clear();
}
return s;
}
string &rstrip(string &s, const string &delim = WHITESPACE) {
std::size_t pos = s.find_last_not_of(delim);
if (pos != string::npos) {
s.erase(pos + 1);
} else {
s.clear();
}
return s;
}
string &strip(string &s, const string &delim = WHITESPACE) {
return lstrip(rstrip(s, delim), delim);
}
string ltrimmed(string s, const string &delim = WHITESPACE) {
return lstrip(s, delim);
}
string rtrimmed(string s, const string &delim = WHITESPACE) {
return rstrip(s, delim);
}
string trimmed(string s, const string &delim = WHITESPACE) {
return strip(s, delim);
}
/*
Find and Replace:
- `starts_with(s, prefix)` returns whether `s` begins with `prefix`.
- `ends_with(s, suffix)` returns whether `s` ends with `suffix`.
- `contains(s, needle)` returns whether `needle` occurs in `s`.
- `find_all(haystack, needle)` returns a vector of all positions where the nonempty string `needle`
appears in the string `haystack`. Matches may overlap; e.g. `find_all("aaa", "aa")` returns
`{0, 1}`. An empty `needle` produces no matches.
- `count(haystack, needle)` returns the number of non-overlapping occurrences of `needle` in
`haystack`, like Python's `str.count`. This differs from `find_all().size()`, which counts
overlapping matches; e.g. `count("aaa", "aa")` returns $1$, but `find_all("aaa", "aa")` finds $2$.
- `replace(s, old, replacement)` returns a copy of `s` with all occurrences of the string `old`
replaced with the given `replacement`.
- `regex_find_all(s, pattern)` returns every substring of `s` matched by `pattern`.
- `regex_groups(s, pattern)` returns the capture groups of the first match of `pattern` in `s`,
excluding the full match. If there is no match, returns an empty vector.
- `regex_replace_all(s, pattern, replacement)` returns a copy of `s` with every regex match replaced
by `replacement`, which may refer to capture groups as `$1`, `$2`, and so on.
*/
bool starts_with(const string &s, const string &prefix) {
return s.size() >= prefix.size() && s.compare(0, prefix.size(), prefix) == 0;
}
bool ends_with(const string &s, const string &suffix) {
return s.size() >= suffix.size() &&
s.compare(s.size() - suffix.size(), suffix.size(), suffix) == 0;
}
bool contains(const string &s, const string &needle) {
return s.find(needle) != string::npos;
}
std::vector<int> find_all(const string &haystack, const string &needle) {
std::vector<int> res;
if (needle.empty()) {
return res;
}
std::size_t pos = haystack.find(needle, 0);
while (pos != string::npos) {
res.push_back(static_cast<int>(pos));
pos = haystack.find(needle, pos + 1);
}
return res;
}
int count(const string &haystack, const string &needle) {
if (needle.empty()) {
return 0;
}
int res = 0;
std::size_t pos = haystack.find(needle, 0);
while (pos != string::npos) {
res++;
pos = haystack.find(needle, pos + needle.size());
}
return res;
}
string replace(const string &s, const string &old, const string &replacement) {
if (old.empty()) {
return s;
}
string res(s);
std::size_t pos = 0;
while ((pos = res.find(old, pos)) != string::npos) {
res.replace(pos, old.length(), replacement);
pos += replacement.length();
}
return res;
}
std::vector<string> regex_find_all(const string &s, const string &pattern) {
std::regex re(pattern);
return {std::sregex_token_iterator(s.begin(), s.end(), re, 0), std::sregex_token_iterator()};
}
std::vector<string> regex_groups(const string &s, const string &pattern) {
std::smatch match;
if (!std::regex_search(s, match, std::regex(pattern))) {
return {};
}
std::vector<string> res;
for (int i = 1; i < static_cast<int>(match.size()); i++) {
res.push_back(match[i].str());
}
return res;
}
string regex_replace_all(const string &s, const string &pattern, const string &replacement) {
return std::regex_replace(s, std::regex(pattern), replacement);
}
/*
Joining and Splitting:
- `join(v, delim = " ")` returns the strings in vector `v` concatenated, separated by the given
delimiter.
- `repeat(s, n)` returns `s` concatenated with itself `n` times, like Python's `s * n`, or an empty
string if `n` is nonpositive. For a single repeated character, prefer `std::string(n, ch)`.
- `ljust(s, width, ch = ' ')` returns `s` left-justified (padded on the right with `ch`) to at least
`width` characters, like Python's `str.ljust`.
- `rjust(s, width, ch = ' ')` returns `s` right-justified (padded on the left with `ch`) to at least
`width` characters, like Python's `str.rjust`.
- `center(s, width, ch = ' ')` returns `s` centered in a field of at least `width` characters by
padding both sides with `ch`, like Python's `str.center`. If the padding is uneven, the extra
character goes on the right.
- `split(s, char delim)` returns a vector of tokens of `s`, split on a single character delimiter.
Empty tokens are skipped, so `split("a::b", ':')` returns `{"a", "b"}`, not `{"a", "", "b"}`.
- `split(s, string delim = WHITESPACE)` returns a vector of tokens of `s`, split on a set of many
possible single character delimiters. All characters of `delim` will be removed from `s`, and the
remaining token(s) of `s` will be added sequentially to a vector and returned. As with the first
version, empty tokens are skipped. For example, `split("a::b", ":")` returns `{"a", "b"}`, not
`{"a", "", "b"}`.
- `explode(s, delim)` returns a vector of tokens of `s`, split on the entire delimiter string
`delim`. Unlike the `split()` functions above, `delim` is treated as a contiguous boundary string,
not merely a set of possible boundary characters. This will not skip empty tokens. For example,
`explode("a::::b", "::")` yields `{"a", "", "b"}`, not `{"a", "b"}`. An empty delimiter returns
`{s}`.
- `explode(s, char delim)` is the single-character delimiter version that also preserves empty
tokens.
*/
string join(const std::vector<string> &v, const string &delim = " ") {
if (v.empty()) {
return "";
}
string res;
std::size_t total = delim.size() * (v.size() - 1);
for (const string &s : v) {
total += s.size();
}
res.reserve(total);
for (int i = 0; i < static_cast<int>(v.size()); i++) {
res += (i > 0 ? delim : "") + v[i];
}
return res;
}
string repeat(const string &s, int n) {
if (n <= 0) {
return "";
}
string res;
res.reserve(s.size() * n);
for (int i = 0; i < n; i++) {
res += s;
}
return res;
}
string ljust(const string &s, int width, char ch = ' ') {
int pad = width - static_cast<int>(s.size());
return pad > 0 ? s + string(pad, ch) : s;
}
string rjust(const string &s, int width, char ch = ' ') {
int pad = width - static_cast<int>(s.size());
return pad > 0 ? string(pad, ch) + s : s;
}
string center(const string &s, int width, char ch = ' ') {
int pad = width - static_cast<int>(s.size());
if (pad <= 0) {
return s;
}
int left = pad / 2;
return string(left, ch) + s + string(pad - left, ch);
}
std::vector<string> split(const string &s, char delim) {
std::vector<string> res;
std::stringstream ss(s);
string curr;
while (std::getline(ss, curr, delim)) {
if (!curr.empty()) {
res.push_back(curr);
}
}
return res;
}
std::vector<string> split(const string &s, const string &delim = WHITESPACE) {
std::vector<string> res;
string curr;
for (char c : s) {
if (delim.find(c) == string::npos) {
curr += c;
} else if (!curr.empty()) {
res.push_back(curr);
curr = "";
}
}
if (!curr.empty()) {
res.push_back(curr);
}
return res;
}
std::vector<string> explode(const string &s, const string &delim) {
if (delim.empty()) {
return {s};
}
std::vector<string> res;
std::size_t last = 0, next = 0;
while ((next = s.find(delim, last)) != string::npos) {
res.push_back(s.substr(last, next - last));
last = next + delim.size();
}
res.push_back(s.substr(last));
return res;
}
std::vector<string> explode(const string &s, char delim) {
std::vector<string> res;
std::size_t last = 0, next = 0;
while ((next = s.find(delim, last)) != string::npos) {
res.push_back(s.substr(last, next - last));
last = next + 1;
}
res.push_back(s.substr(last));
return res;
}
/*** Example Usage ***/
#include <cassert>
using namespace std;
int main() {
assert(to_str(123) + "4" == "1234");
assert(to_int("1234") == 1234);
bool threw = false;
try {
to_int("123x");
} catch (const runtime_error &) {
threw = true;
}
assert(threw);
vector<char> buffer(50);
assert(string(itoa(1750, buffer.data(), 10)) == "1750");
assert(string(itoa(1750, buffer.data(), 16)) == "6d6");
assert(string(itoa(1750, buffer.data(), 2)) == "11011010110");
assert(is_integer("-123") && !is_integer("12.3"));
assert(is_number("-.5e+2") && is_number("123.") && !is_number("1e"));
assert(to_upper("Hello world") == "HELLO WORLD");
assert(to_lower("Hello World") == "hello world");
assert(to_title("hello world") == "Hello World");
string s(" abc \n");
string t = s;
assert(lstrip(s) == "abc \n");
assert(rstrip(s) == strip(t));
assert(trimmed(" \t abc \n") == "abc");
assert(trimmed("xxx", "x").empty());
vector<int> pos;
pos.push_back(0);
pos.push_back(7);
assert(starts_with("abracadabra", "abra"));
assert(ends_with("abracadabra", "dabra"));
assert(contains("abracadabra", "cad"));
assert(find_all("abracadabra", "ab") == pos);
assert(count("abracadabra", "ab") == 2);
assert(count("aaa", "aa") == 1);
assert((find_all("aaa", "aa") == vector<int>{0, 1}));
assert(find_all("abc", "").empty());
assert(replace("abcdabba", "ab", "00") == "00cd00ba");
assert(join(regex_find_all("a12b345", "[0-9]+"), "|") == "12|345");
assert(join(regex_groups("abc-123", "([a-z]+)-([0-9]+)"), "|") == "abc|123");
assert(regex_replace_all("a12b345", "[0-9]+", "#") == "a#b#");
assert(regex_replace_all("2026-08-04", R"((\d{4})-(\d{2})-(\d{2}))", "$2/$3/$1") == "08/04/2026");
assert(repeat("ab", 3) == "ababab");
assert(repeat("ab", -1).empty());
assert(rjust("42", 5, '0') == "00042");
assert(ljust("hi", 4, '.') == "hi..");
assert(center("ab", 5, '*') == "*ab**");
assert(join(split("a\nb\ncde\nf", '\n'), "|") == "a|b|cde|f"); // split v1
assert(join(split("a::b", ':'), "|") == "a|b"); // split v1 skips empty tokens
assert(join(split("a::b,cde:,f", ":,"), "|") == "a|b|cde|f"); // split v2
assert(join(explode("a..b.cde....f", ".."), "|") == "a|b.cde||f");
assert((explode("abc", "") == vector<string>{"abc"}));
assert(join(explode("a::b:", ':'), "|") == "a||b|");
return 0;
}