Strings / String Utilities
3.1 String Utilities
Common string functions, many of which already have standard STL equivalents. Most of the following implementations are presented for educational purposes, and are not heavily optimized. They often depend on certain std::string functions that have unspecified complexity.
Implementation
#include <cctype>
#include <cstddef>
#include <regex>
#include <sstream>
#include <stdexcept>
#include <string>
#include <vector>
using std::string;
Integer Conversion:
to_str(i)returns the string representation of integeri, much likestd::to_string().to_int(s)returns the integer representation of strings, much likestd::atoi(), except here we handle special cases of overflow by throwing an exception.itoa(value, str, base = 10)implements the non-standard C function which convertsvalueinto a C string, storing it into pointerstrin the givenbase. For more generalized base conversion, see the math utilities section.is_integer(s)returns whethersis a valid base-10 integer literal with an optional sign.is_number(s)returns whethersis a decimal number literal with optional sign, decimal point, and exponent.
Implementation
template<typename Int>
string to_str(Int i) {
std::ostringstream oss;
oss << i;
return oss.str();
}
int to_int(const string &s) {
std::istringstream iss(s);
int res;
if (!(iss >> res) || !(iss >> std::ws).eof()) {
throw std::runtime_error("to_int failed");
}
return res;
}
char *itoa(int value, char *str, int base = 10) {
if (base < 2 || base > 36) {
*str = '\0';
return str;
}
char *ptr = str, *ptr1 = str, tmp_c;
int tmp_v;
do {
tmp_v = value;
value /= base;
*ptr++ =
"zyxwvutsrqponmlkjihgfedcba9876543210123456789"
"abcdefghijklmnopqrstuvwxyz"[35 + (tmp_v - value * base)];
} while (value);
if (tmp_v < 0) {
*ptr++ = '-';
}
for (*ptr-- = '\0'; ptr1 < ptr; *ptr1++ = tmp_c) {
tmp_c = *ptr;
*ptr-- = *ptr1;
}
return str;
}
bool is_integer(const string &s) {
// return std::regex_match(s, std::regex(R"([+-]?\d+)"));
int i = (s.empty() || (s[0] != '-' && s[0] != '+')) ? 0 : 1;
if (i == static_cast<int>(s.size())) {
return false;
}
for (; i < static_cast<int>(s.size()); i++) {
if (!isdigit(static_cast<unsigned char>(s[i]))) {
return false;
}
}
return true;
}
bool is_number(const string &s) {
// return std::regex_match(s, std::regex(R"([+-]?(\d+(\.\d*)?|\.\d+)([eE][+-]?\d+)?)"));
int i = (s.empty() || (s[0] != '-' && s[0] != '+')) ? 0 : 1;
bool seen_digit = false, seen_dot = false;
for (; i < static_cast<int>(s.size()); i++) {
if (isdigit(static_cast<unsigned char>(s[i]))) {
seen_digit = true;
} else if (s[i] == '.' && !seen_dot) {
seen_dot = true;
} else {
break;
}
}
if (!seen_digit) {
return false;
}
if (i < static_cast<int>(s.size()) && (s[i] == 'e' || s[i] == 'E')) {
i++;
if (i < static_cast<int>(s.size()) && (s[i] == '-' || s[i] == '+')) {
i++;
}
bool seen_exp_digit = false;
for (; i < static_cast<int>(s.size()); i++) {
if (!isdigit(static_cast<unsigned char>(s[i]))) {
return false;
}
seen_exp_digit = true;
}
return seen_exp_digit;
}
return i == static_cast<int>(s.size());
}
Case Conversion:
to_upper(s)returnsswith all alphabetical characters converted to uppercase.to_lower(s)returnsswith all alphabetical characters converted to lowercase.to_title(s)returns the title case representation of strings, where the first letter of every word (consecutive alphabetical characters) is capitalized.
Implementation
string to_upper(const string &s) {
string res;
res.reserve(s.size());
for (unsigned char c : s) {
res.push_back(static_cast<char>(toupper(c)));
}
return res;
}
string to_lower(const string &s) {
string res;
res.reserve(s.size());
for (unsigned char c : s) {
res.push_back(static_cast<char>(tolower(c)));
}
return res;
}
string to_title(const string &s) {
string res;
unsigned char prev = '\0';
for (unsigned char c : s) {
res.push_back(isalpha(prev) ? static_cast<char>(tolower(c)) : static_cast<char>(toupper(c)));
prev = res.back();
}
return res;
}
Stripping:
lstrip(s, delim = WHITESPACE)strips the left side ofsin-place (that is, the input is modified) using the given delimiters and returns a reference to the stripped string.rstrip(s, delim = WHITESPACE)strips the right side ofsin-place using the given delimiters and returns a reference to the stripped string.strip(s, delim = WHITESPACE)strips both sides ofsin-place and returns a reference to the stripped string.ltrimmed(s, delim = WHITESPACE),rtrimmed(s, delim = WHITESPACE), andtrimmed(s, delim = WHITESPACE)do not modifysand return stripped copies.
Implementation
const string WHITESPACE = " \n\t\v\f\r";
string &lstrip(string &s, const string &delim = WHITESPACE) {
std::size_t pos = s.find_first_not_of(delim);
if (pos != string::npos) {
s.erase(0, pos);
} else {
s.clear();
}
return s;
}
string &rstrip(string &s, const string &delim = WHITESPACE) {
std::size_t pos = s.find_last_not_of(delim);
if (pos != string::npos) {
s.erase(pos + 1);
} else {
s.clear();
}
return s;
}
string &strip(string &s, const string &delim = WHITESPACE) {
return lstrip(rstrip(s, delim), delim);
}
string ltrimmed(string s, const string &delim = WHITESPACE) {
return lstrip(s, delim);
}
string rtrimmed(string s, const string &delim = WHITESPACE) {
return rstrip(s, delim);
}
string trimmed(string s, const string &delim = WHITESPACE) {
return strip(s, delim);
}
Find and Replace:
starts_with(s, prefix)returns whethersbegins withprefix.ends_with(s, suffix)returns whethersends withsuffix.contains(s, needle)returns whetherneedleoccurs ins.find_all(haystack, needle)returns a vector of all positions where the nonempty stringneedleappears in the stringhaystack. Matches may overlap; e.g.find_all("aaa", "aa")returns{0, 1}. An emptyneedleproduces no matches.count(haystack, needle)returns the number of non-overlapping occurrences ofneedleinhaystack, like Python'sstr.count. This differs fromfind_all().size(), which counts overlapping matches; e.g.count("aaa", "aa")returns $1$, butfind_all("aaa", "aa")finds $2$.replace(s, old, replacement)returns a copy ofswith all occurrences of the stringoldreplaced with the givenreplacement.regex_find_all(s, pattern)returns every substring ofsmatched bypattern.regex_groups(s, pattern)returns the capture groups of the first match ofpatternins, excluding the full match. If there is no match, returns an empty vector.regex_replace_all(s, pattern, replacement)returns a copy ofswith every regex match replaced byreplacement, which may refer to capture groups as$1,$2, and so on.
Implementation
bool starts_with(const string &s, const string &prefix) {
return s.size() >= prefix.size() && s.compare(0, prefix.size(), prefix) == 0;
}
bool ends_with(const string &s, const string &suffix) {
return s.size() >= suffix.size() &&
s.compare(s.size() - suffix.size(), suffix.size(), suffix) == 0;
}
bool contains(const string &s, const string &needle) {
return s.find(needle) != string::npos;
}
std::vector<int> find_all(const string &haystack, const string &needle) {
std::vector<int> res;
if (needle.empty()) {
return res;
}
std::size_t pos = haystack.find(needle, 0);
while (pos != string::npos) {
res.push_back(static_cast<int>(pos));
pos = haystack.find(needle, pos + 1);
}
return res;
}
int count(const string &haystack, const string &needle) {
if (needle.empty()) {
return 0;
}
int res = 0;
std::size_t pos = haystack.find(needle, 0);
while (pos != string::npos) {
res++;
pos = haystack.find(needle, pos + needle.size());
}
return res;
}
string replace(const string &s, const string &old, const string &replacement) {
if (old.empty()) {
return s;
}
string res(s);
std::size_t pos = 0;
while ((pos = res.find(old, pos)) != string::npos) {
res.replace(pos, old.length(), replacement);
pos += replacement.length();
}
return res;
}
std::vector<string> regex_find_all(const string &s, const string &pattern) {
std::regex re(pattern);
return {std::sregex_token_iterator(s.begin(), s.end(), re, 0), std::sregex_token_iterator()};
}
std::vector<string> regex_groups(const string &s, const string &pattern) {
std::smatch match;
if (!std::regex_search(s, match, std::regex(pattern))) {
return {};
}
std::vector<string> res;
for (int i = 1; i < static_cast<int>(match.size()); i++) {
res.push_back(match[i].str());
}
return res;
}
string regex_replace_all(const string &s, const string &pattern, const string &replacement) {
return std::regex_replace(s, std::regex(pattern), replacement);
}
Joining and Splitting:
join(v, delim = " ")returns the strings in vectorvconcatenated, separated by the given delimiter.repeat(s, n)returnssconcatenated with itselfntimes, like Python'ss * n, or an empty string ifnis nonpositive. For a single repeated character, preferstd::string(n, ch).ljust(s, width, ch = ' ')returnssleft-justified (padded on the right withch) to at leastwidthcharacters, like Python'sstr.ljust.rjust(s, width, ch = ' ')returnssright-justified (padded on the left withch) to at leastwidthcharacters, like Python'sstr.rjust.center(s, width, ch = ' ')returnsscentered in a field of at leastwidthcharacters by padding both sides withch, like Python'sstr.center. If the padding is uneven, the extra character goes on the right.split(s, char delim)returns a vector of tokens ofs, split on a single character delimiter. Empty tokens are skipped, sosplit("a::b", ':')returns{"a", "b"}, not{"a", "", "b"}.split(s, string delim = WHITESPACE)returns a vector of tokens ofs, split on a set of many possible single character delimiters. All characters ofdelimwill be removed froms, and the remaining token(s) ofswill be added sequentially to a vector and returned. As with the first version, empty tokens are skipped. For example,split("a::b", ":")returns{"a", "b"}, not{"a", "", "b"}.explode(s, delim)returns a vector of tokens ofs, split on the entire delimiter stringdelim. Unlike thesplit()functions above,delimis treated as a contiguous boundary string, not merely a set of possible boundary characters. This will not skip empty tokens. For example,explode("a::::b", "::")yields{"a", "", "b"}, not{"a", "b"}. An empty delimiter returns{s}.explode(s, char delim)is the single-character delimiter version that also preserves empty tokens.
Implementation
string join(const std::vector<string> &v, const string &delim = " ") {
if (v.empty()) {
return "";
}
string res;
std::size_t total = delim.size() * (v.size() - 1);
for (const string &s : v) {
total += s.size();
}
res.reserve(total);
for (int i = 0; i < static_cast<int>(v.size()); i++) {
res += (i > 0 ? delim : "") + v[i];
}
return res;
}
string repeat(const string &s, int n) {
if (n <= 0) {
return "";
}
string res;
res.reserve(s.size() * n);
for (int i = 0; i < n; i++) {
res += s;
}
return res;
}
string ljust(const string &s, int width, char ch = ' ') {
int pad = width - static_cast<int>(s.size());
return pad > 0 ? s + string(pad, ch) : s;
}
string rjust(const string &s, int width, char ch = ' ') {
int pad = width - static_cast<int>(s.size());
return pad > 0 ? string(pad, ch) + s : s;
}
string center(const string &s, int width, char ch = ' ') {
int pad = width - static_cast<int>(s.size());
if (pad <= 0) {
return s;
}
int left = pad / 2;
return string(left, ch) + s + string(pad - left, ch);
}
std::vector<string> split(const string &s, char delim) {
std::vector<string> res;
std::stringstream ss(s);
string curr;
while (std::getline(ss, curr, delim)) {
if (!curr.empty()) {
res.push_back(curr);
}
}
return res;
}
std::vector<string> split(const string &s, const string &delim = WHITESPACE) {
std::vector<string> res;
string curr;
for (char c : s) {
if (delim.find(c) == string::npos) {
curr += c;
} else if (!curr.empty()) {
res.push_back(curr);
curr = "";
}
}
if (!curr.empty()) {
res.push_back(curr);
}
return res;
}
std::vector<string> explode(const string &s, const string &delim) {
if (delim.empty()) {
return {s};
}
std::vector<string> res;
std::size_t last = 0, next = 0;
while ((next = s.find(delim, last)) != string::npos) {
res.push_back(s.substr(last, next - last));
last = next + delim.size();
}
res.push_back(s.substr(last));
return res;
}
std::vector<string> explode(const string &s, char delim) {
std::vector<string> res;
std::size_t last = 0, next = 0;
while ((next = s.find(delim, last)) != string::npos) {
res.push_back(s.substr(last, next - last));
last = next + 1;
}
res.push_back(s.substr(last));
return res;
}
Example Usage
#include <cassert>
using namespace std;
int main() {
assert(to_str(123) + "4" == "1234");
assert(to_int("1234") == 1234);
bool threw = false;
try {
to_int("123x");
} catch (const runtime_error &) {
threw = true;
}
assert(threw);
vector<char> buffer(50);
assert(string(itoa(1750, buffer.data(), 10)) == "1750");
assert(string(itoa(1750, buffer.data(), 16)) == "6d6");
assert(string(itoa(1750, buffer.data(), 2)) == "11011010110");
assert(is_integer("-123") && !is_integer("12.3"));
assert(is_number("-.5e+2") && is_number("123.") && !is_number("1e"));
assert(to_upper("Hello world") == "HELLO WORLD");
assert(to_lower("Hello World") == "hello world");
assert(to_title("hello world") == "Hello World");
string s(" abc \n");
string t = s;
assert(lstrip(s) == "abc \n");
assert(rstrip(s) == strip(t));
assert(trimmed(" \t abc \n") == "abc");
assert(trimmed("xxx", "x").empty());
vector<int> pos;
pos.push_back(0);
pos.push_back(7);
assert(starts_with("abracadabra", "abra"));
assert(ends_with("abracadabra", "dabra"));
assert(contains("abracadabra", "cad"));
assert(find_all("abracadabra", "ab") == pos);
assert(count("abracadabra", "ab") == 2);
assert(count("aaa", "aa") == 1);
assert((find_all("aaa", "aa") == vector<int>{0, 1}));
assert(find_all("abc", "").empty());
assert(replace("abcdabba", "ab", "00") == "00cd00ba");
assert(join(regex_find_all("a12b345", "[0-9]+"), "|") == "12|345");
assert(join(regex_groups("abc-123", "([a-z]+)-([0-9]+)"), "|") == "abc|123");
assert(regex_replace_all("a12b345", "[0-9]+", "#") == "a#b#");
assert(regex_replace_all("2026-08-04", R"((\d{4})-(\d{2})-(\d{2}))", "$2/$3/$1") == "08/04/2026");
assert(repeat("ab", 3) == "ababab");
assert(repeat("ab", -1).empty());
assert(rjust("42", 5, '0') == "00042");
assert(ljust("hi", 4, '.') == "hi..");
assert(center("ab", 5, '*') == "*ab**");
assert(join(split("a\nb\ncde\nf", '\n'), "|") == "a|b|cde|f"); // split v1
assert(join(split("a::b", ':'), "|") == "a|b"); // split v1 skips empty tokens
assert(join(split("a::b,cde:,f", ":,"), "|") == "a|b|cde|f"); // split v2
assert(join(explode("a..b.cde....f", ".."), "|") == "a|b.cde||f");
assert((explode("abc", "") == vector<string>{"abc"}));
assert(join(explode("a::b:", ':'), "|") == "a||b|");
return 0;
}
/*
Common string functions, many of which already have standard STL equivalents. Most of the following
implementations are presented for educational purposes, and are not heavily optimized. They often
depend on certain `std::string` functions that have unspecified complexity.
Time Complexity:
- O(n) per call to most operations, where $n$ is the length of the input string or total length of
all input strings. Exceptions are noted with the individual functions below.
Space Complexity:
- O(n) auxiliary for operations that return a new string or vector of strings.
- O(1) auxiliary for in-place operations.
*/
#include <cctype>
#include <cstddef>
#include <regex>
#include <sstream>
#include <stdexcept>
#include <string>
#include <vector>
using std::string;
/*
Integer Conversion:
- `to_str(i)` returns the string representation of integer `i`, much like `std::to_string()`.
- `to_int(s)` returns the integer representation of string `s`, much like `std::atoi()`, except here
we handle special cases of overflow by throwing an exception.
- `itoa(value, str, base = 10)` implements the non-standard C function which converts `value` into a
C string, storing it into pointer `str` in the given `base`. For more generalized base conversion,
see the math utilities section.
- `is_integer(s)` returns whether `s` is a valid base-10 integer literal with an optional sign.
- `is_number(s)` returns whether `s` is a decimal number literal with optional sign, decimal point,
and exponent.
*/
template<typename Int>
string to_str(Int i) {
std::ostringstream oss;
oss << i;
return oss.str();
}
int to_int(const string &s) {
std::istringstream iss(s);
int res;
if (!(iss >> res) || !(iss >> std::ws).eof()) {
throw std::runtime_error("to_int failed");
}
return res;
}
char *itoa(int value, char *str, int base = 10) {
if (base < 2 || base > 36) {
*str = '\0';
return str;
}
char *ptr = str, *ptr1 = str, tmp_c;
int tmp_v;
do {
tmp_v = value;
value /= base;
*ptr++ =
"zyxwvutsrqponmlkjihgfedcba9876543210123456789"
"abcdefghijklmnopqrstuvwxyz"[35 + (tmp_v - value * base)];
} while (value);
if (tmp_v < 0) {
*ptr++ = '-';
}
for (*ptr-- = '\0'; ptr1 < ptr; *ptr1++ = tmp_c) {
tmp_c = *ptr;
*ptr-- = *ptr1;
}
return str;
}
bool is_integer(const string &s) {
// return std::regex_match(s, std::regex(R"([+-]?\d+)"));
int i = (s.empty() || (s[0] != '-' && s[0] != '+')) ? 0 : 1;
if (i == static_cast<int>(s.size())) {
return false;
}
for (; i < static_cast<int>(s.size()); i++) {
if (!isdigit(static_cast<unsigned char>(s[i]))) {
return false;
}
}
return true;
}
bool is_number(const string &s) {
// return std::regex_match(s, std::regex(R"([+-]?(\d+(\.\d*)?|\.\d+)([eE][+-]?\d+)?)"));
int i = (s.empty() || (s[0] != '-' && s[0] != '+')) ? 0 : 1;
bool seen_digit = false, seen_dot = false;
for (; i < static_cast<int>(s.size()); i++) {
if (isdigit(static_cast<unsigned char>(s[i]))) {
seen_digit = true;
} else if (s[i] == '.' && !seen_dot) {
seen_dot = true;
} else {
break;
}
}
if (!seen_digit) {
return false;
}
if (i < static_cast<int>(s.size()) && (s[i] == 'e' || s[i] == 'E')) {
i++;
if (i < static_cast<int>(s.size()) && (s[i] == '-' || s[i] == '+')) {
i++;
}
bool seen_exp_digit = false;
for (; i < static_cast<int>(s.size()); i++) {
if (!isdigit(static_cast<unsigned char>(s[i]))) {
return false;
}
seen_exp_digit = true;
}
return seen_exp_digit;
}
return i == static_cast<int>(s.size());
}
/*
Case Conversion:
- `to_upper(s)` returns `s` with all alphabetical characters converted to uppercase.
- `to_lower(s)` returns `s` with all alphabetical characters converted to lowercase.
- `to_title(s)` returns the title case representation of string `s`, where the first letter of every
word (consecutive alphabetical characters) is capitalized.
*/
string to_upper(const string &s) {
string res;
res.reserve(s.size());
for (unsigned char c : s) {
res.push_back(static_cast<char>(toupper(c)));
}
return res;
}
string to_lower(const string &s) {
string res;
res.reserve(s.size());
for (unsigned char c : s) {
res.push_back(static_cast<char>(tolower(c)));
}
return res;
}
string to_title(const string &s) {
string res;
unsigned char prev = '\0';
for (unsigned char c : s) {
res.push_back(isalpha(prev) ? static_cast<char>(tolower(c)) : static_cast<char>(toupper(c)));
prev = res.back();
}
return res;
}
/*
Stripping:
- `lstrip(s, delim = WHITESPACE)` strips the left side of `s` in-place (that is, the input is
modified) using the given delimiters and returns a reference to the stripped string.
- `rstrip(s, delim = WHITESPACE)` strips the right side of `s` in-place using the given delimiters
and returns a reference to the stripped string.
- `strip(s, delim = WHITESPACE)` strips both sides of `s` in-place and returns a reference to the
stripped string.
- `ltrimmed(s, delim = WHITESPACE)`, `rtrimmed(s, delim = WHITESPACE)`, and
`trimmed(s, delim = WHITESPACE)` do not modify `s` and return stripped copies.
*/
const string WHITESPACE = " \n\t\v\f\r";
string &lstrip(string &s, const string &delim = WHITESPACE) {
std::size_t pos = s.find_first_not_of(delim);
if (pos != string::npos) {
s.erase(0, pos);
} else {
s.clear();
}
return s;
}
string &rstrip(string &s, const string &delim = WHITESPACE) {
std::size_t pos = s.find_last_not_of(delim);
if (pos != string::npos) {
s.erase(pos + 1);
} else {
s.clear();
}
return s;
}
string &strip(string &s, const string &delim = WHITESPACE) {
return lstrip(rstrip(s, delim), delim);
}
string ltrimmed(string s, const string &delim = WHITESPACE) {
return lstrip(s, delim);
}
string rtrimmed(string s, const string &delim = WHITESPACE) {
return rstrip(s, delim);
}
string trimmed(string s, const string &delim = WHITESPACE) {
return strip(s, delim);
}
/*
Find and Replace:
- `starts_with(s, prefix)` returns whether `s` begins with `prefix`.
- `ends_with(s, suffix)` returns whether `s` ends with `suffix`.
- `contains(s, needle)` returns whether `needle` occurs in `s`.
- `find_all(haystack, needle)` returns a vector of all positions where the nonempty string `needle`
appears in the string `haystack`. Matches may overlap; e.g. `find_all("aaa", "aa")` returns
`{0, 1}`. An empty `needle` produces no matches.
- `count(haystack, needle)` returns the number of non-overlapping occurrences of `needle` in
`haystack`, like Python's `str.count`. This differs from `find_all().size()`, which counts
overlapping matches; e.g. `count("aaa", "aa")` returns $1$, but `find_all("aaa", "aa")` finds $2$.
- `replace(s, old, replacement)` returns a copy of `s` with all occurrences of the string `old`
replaced with the given `replacement`.
- `regex_find_all(s, pattern)` returns every substring of `s` matched by `pattern`.
- `regex_groups(s, pattern)` returns the capture groups of the first match of `pattern` in `s`,
excluding the full match. If there is no match, returns an empty vector.
- `regex_replace_all(s, pattern, replacement)` returns a copy of `s` with every regex match replaced
by `replacement`, which may refer to capture groups as `$1`, `$2`, and so on.
*/
bool starts_with(const string &s, const string &prefix) {
return s.size() >= prefix.size() && s.compare(0, prefix.size(), prefix) == 0;
}
bool ends_with(const string &s, const string &suffix) {
return s.size() >= suffix.size() &&
s.compare(s.size() - suffix.size(), suffix.size(), suffix) == 0;
}
bool contains(const string &s, const string &needle) {
return s.find(needle) != string::npos;
}
std::vector<int> find_all(const string &haystack, const string &needle) {
std::vector<int> res;
if (needle.empty()) {
return res;
}
std::size_t pos = haystack.find(needle, 0);
while (pos != string::npos) {
res.push_back(static_cast<int>(pos));
pos = haystack.find(needle, pos + 1);
}
return res;
}
int count(const string &haystack, const string &needle) {
if (needle.empty()) {
return 0;
}
int res = 0;
std::size_t pos = haystack.find(needle, 0);
while (pos != string::npos) {
res++;
pos = haystack.find(needle, pos + needle.size());
}
return res;
}
string replace(const string &s, const string &old, const string &replacement) {
if (old.empty()) {
return s;
}
string res(s);
std::size_t pos = 0;
while ((pos = res.find(old, pos)) != string::npos) {
res.replace(pos, old.length(), replacement);
pos += replacement.length();
}
return res;
}
std::vector<string> regex_find_all(const string &s, const string &pattern) {
std::regex re(pattern);
return {std::sregex_token_iterator(s.begin(), s.end(), re, 0), std::sregex_token_iterator()};
}
std::vector<string> regex_groups(const string &s, const string &pattern) {
std::smatch match;
if (!std::regex_search(s, match, std::regex(pattern))) {
return {};
}
std::vector<string> res;
for (int i = 1; i < static_cast<int>(match.size()); i++) {
res.push_back(match[i].str());
}
return res;
}
string regex_replace_all(const string &s, const string &pattern, const string &replacement) {
return std::regex_replace(s, std::regex(pattern), replacement);
}
/*
Joining and Splitting:
- `join(v, delim = " ")` returns the strings in vector `v` concatenated, separated by the given
delimiter.
- `repeat(s, n)` returns `s` concatenated with itself `n` times, like Python's `s * n`, or an empty
string if `n` is nonpositive. For a single repeated character, prefer `std::string(n, ch)`.
- `ljust(s, width, ch = ' ')` returns `s` left-justified (padded on the right with `ch`) to at least
`width` characters, like Python's `str.ljust`.
- `rjust(s, width, ch = ' ')` returns `s` right-justified (padded on the left with `ch`) to at least
`width` characters, like Python's `str.rjust`.
- `center(s, width, ch = ' ')` returns `s` centered in a field of at least `width` characters by
padding both sides with `ch`, like Python's `str.center`. If the padding is uneven, the extra
character goes on the right.
- `split(s, char delim)` returns a vector of tokens of `s`, split on a single character delimiter.
Empty tokens are skipped, so `split("a::b", ':')` returns `{"a", "b"}`, not `{"a", "", "b"}`.
- `split(s, string delim = WHITESPACE)` returns a vector of tokens of `s`, split on a set of many
possible single character delimiters. All characters of `delim` will be removed from `s`, and the
remaining token(s) of `s` will be added sequentially to a vector and returned. As with the first
version, empty tokens are skipped. For example, `split("a::b", ":")` returns `{"a", "b"}`, not
`{"a", "", "b"}`.
- `explode(s, delim)` returns a vector of tokens of `s`, split on the entire delimiter string
`delim`. Unlike the `split()` functions above, `delim` is treated as a contiguous boundary string,
not merely a set of possible boundary characters. This will not skip empty tokens. For example,
`explode("a::::b", "::")` yields `{"a", "", "b"}`, not `{"a", "b"}`. An empty delimiter returns
`{s}`.
- `explode(s, char delim)` is the single-character delimiter version that also preserves empty
tokens.
*/
string join(const std::vector<string> &v, const string &delim = " ") {
if (v.empty()) {
return "";
}
string res;
std::size_t total = delim.size() * (v.size() - 1);
for (const string &s : v) {
total += s.size();
}
res.reserve(total);
for (int i = 0; i < static_cast<int>(v.size()); i++) {
res += (i > 0 ? delim : "") + v[i];
}
return res;
}
string repeat(const string &s, int n) {
if (n <= 0) {
return "";
}
string res;
res.reserve(s.size() * n);
for (int i = 0; i < n; i++) {
res += s;
}
return res;
}
string ljust(const string &s, int width, char ch = ' ') {
int pad = width - static_cast<int>(s.size());
return pad > 0 ? s + string(pad, ch) : s;
}
string rjust(const string &s, int width, char ch = ' ') {
int pad = width - static_cast<int>(s.size());
return pad > 0 ? string(pad, ch) + s : s;
}
string center(const string &s, int width, char ch = ' ') {
int pad = width - static_cast<int>(s.size());
if (pad <= 0) {
return s;
}
int left = pad / 2;
return string(left, ch) + s + string(pad - left, ch);
}
std::vector<string> split(const string &s, char delim) {
std::vector<string> res;
std::stringstream ss(s);
string curr;
while (std::getline(ss, curr, delim)) {
if (!curr.empty()) {
res.push_back(curr);
}
}
return res;
}
std::vector<string> split(const string &s, const string &delim = WHITESPACE) {
std::vector<string> res;
string curr;
for (char c : s) {
if (delim.find(c) == string::npos) {
curr += c;
} else if (!curr.empty()) {
res.push_back(curr);
curr = "";
}
}
if (!curr.empty()) {
res.push_back(curr);
}
return res;
}
std::vector<string> explode(const string &s, const string &delim) {
if (delim.empty()) {
return {s};
}
std::vector<string> res;
std::size_t last = 0, next = 0;
while ((next = s.find(delim, last)) != string::npos) {
res.push_back(s.substr(last, next - last));
last = next + delim.size();
}
res.push_back(s.substr(last));
return res;
}
std::vector<string> explode(const string &s, char delim) {
std::vector<string> res;
std::size_t last = 0, next = 0;
while ((next = s.find(delim, last)) != string::npos) {
res.push_back(s.substr(last, next - last));
last = next + 1;
}
res.push_back(s.substr(last));
return res;
}
/*** Example Usage ***/
#include <cassert>
using namespace std;
int main() {
assert(to_str(123) + "4" == "1234");
assert(to_int("1234") == 1234);
bool threw = false;
try {
to_int("123x");
} catch (const runtime_error &) {
threw = true;
}
assert(threw);
vector<char> buffer(50);
assert(string(itoa(1750, buffer.data(), 10)) == "1750");
assert(string(itoa(1750, buffer.data(), 16)) == "6d6");
assert(string(itoa(1750, buffer.data(), 2)) == "11011010110");
assert(is_integer("-123") && !is_integer("12.3"));
assert(is_number("-.5e+2") && is_number("123.") && !is_number("1e"));
assert(to_upper("Hello world") == "HELLO WORLD");
assert(to_lower("Hello World") == "hello world");
assert(to_title("hello world") == "Hello World");
string s(" abc \n");
string t = s;
assert(lstrip(s) == "abc \n");
assert(rstrip(s) == strip(t));
assert(trimmed(" \t abc \n") == "abc");
assert(trimmed("xxx", "x").empty());
vector<int> pos;
pos.push_back(0);
pos.push_back(7);
assert(starts_with("abracadabra", "abra"));
assert(ends_with("abracadabra", "dabra"));
assert(contains("abracadabra", "cad"));
assert(find_all("abracadabra", "ab") == pos);
assert(count("abracadabra", "ab") == 2);
assert(count("aaa", "aa") == 1);
assert((find_all("aaa", "aa") == vector<int>{0, 1}));
assert(find_all("abc", "").empty());
assert(replace("abcdabba", "ab", "00") == "00cd00ba");
assert(join(regex_find_all("a12b345", "[0-9]+"), "|") == "12|345");
assert(join(regex_groups("abc-123", "([a-z]+)-([0-9]+)"), "|") == "abc|123");
assert(regex_replace_all("a12b345", "[0-9]+", "#") == "a#b#");
assert(regex_replace_all("2026-08-04", R"((\d{4})-(\d{2})-(\d{2}))", "$2/$3/$1") == "08/04/2026");
assert(repeat("ab", 3) == "ababab");
assert(repeat("ab", -1).empty());
assert(rjust("42", 5, '0') == "00042");
assert(ljust("hi", 4, '.') == "hi..");
assert(center("ab", 5, '*') == "*ab**");
assert(join(split("a\nb\ncde\nf", '\n'), "|") == "a|b|cde|f"); // split v1
assert(join(split("a::b", ':'), "|") == "a|b"); // split v1 skips empty tokens
assert(join(split("a::b,cde:,f", ":,"), "|") == "a|b|cde|f"); // split v2
assert(join(explode("a..b.cde....f", ".."), "|") == "a|b.cde||f");
assert((explode("abc", "") == vector<string>{"abc"}));
assert(join(explode("a::b:", ':'), "|") == "a||b|");
return 0;
}