atlus

Convert raw address and phone number strings into the OSM format.

atlus is a Python package to convert raw address, phone number, and opening hours strings into the OSM format. It's designed to be used with US and Canadian phone numbers and addresses.

>>> import atlus
>>> atlus.abbrs("St. Francis")
"Saint Francis"
>>> atlus.get_address("789 Oak Dr, Smallville California, 98765")[0]
{"addr:housenumber": "789", "addr:street": "Oak Drive", "addr:city": "Smallville",
    "addr:state": "CA", "addr:postcode": "98765"}
>>> atlus.get_phone("(202) 900-9019")
"+1-202-900-9019"
>>> atlus.get_hours("Monday to Friday 9am-5pm, Saturday 9am-12pm")
"Mo-Fr 09:00-17:00; Sa 09:00-12:00"
>>> atlus.get_times("Mo-Fr 15:00,18:00,19:00,23:00; Sa 15:00; Su 10:30,23:00")
"Mo-Fr 15:00,18:00,19:00,23:00; Sa 15:00; Su 10:30,23:00"
 1"""Convert raw address and phone number strings into the OSM format.
 2
 3`atlus` is a Python package to convert raw address, phone number, and opening
 4hours strings into the OSM format. It's designed to be used with US and Canadian
 5phone numbers and addresses.
 6
 7```python
 8>>> import atlus
 9>>> atlus.abbrs("St. Francis")
10"Saint Francis"
11>>> atlus.get_address("789 Oak Dr, Smallville California, 98765")[0]
12{"addr:housenumber": "789", "addr:street": "Oak Drive", "addr:city": "Smallville",
13    "addr:state": "CA", "addr:postcode": "98765"}
14>>> atlus.get_phone("(202) 900-9019")
15"+1-202-900-9019"
16>>> atlus.get_hours("Monday to Friday 9am-5pm, Saturday 9am-12pm")
17"Mo-Fr 09:00-17:00; Sa 09:00-12:00"
18>>> atlus.get_times("Mo-Fr 15:00,18:00,19:00,23:00; Sa 15:00; Su 10:30,23:00")
19"Mo-Fr 15:00,18:00,19:00,23:00; Sa 15:00; Su 10:30,23:00"
20```
21
22"""
23
24# SPDX-FileCopyrightText: 2024-present Will <wahubsch@gmail.com>
25#
26# SPDX-License-Identifier: MIT
27
28from . import atlus, hours, resources
29from .atlus import (
30    abbrs,
31    get_address,
32    get_phone,
33    get_title,
34    mc_replace,
35    ord_replace,
36    remove_br_unicode,
37    us_replace,
38)
39from .hours import get_hours, get_times
40
41__all__ = [
42    "get_address",
43    "get_phone",
44    "get_hours",
45    "get_times",
46    "abbrs",
47    "get_title",
48    "mc_replace",
49    "us_replace",
50    "ord_replace",
51    "remove_br_unicode",
52    "atlus",
53    "hours",
54    "resources",
55]
def get_address(address_string: str) -> tuple[dict[str, str], list[str | None]]:
688def get_address(address_string: str) -> tuple[dict[str, str], list[str | None]]:
689    """Process address strings.
690
691    ```python
692    >>> get_address("345 MAPLE RD, COUNTRYSIDE, PA 24680-0198")[0]
693    {"addr:housenumber": "345", "addr:street": "Maple Road",
694    "addr:city": "Countryside", "addr:state": "PA", "addr:postcode": "24680-0198"}
695    >>> get_address("777 Strawberry St.")[0]
696    {"addr:housenumber": "777", "addr:street": "Strawberry Street"}
697    >>> address = get_address("222 NW Pineapple Ave Suite A Unit B")
698    >>> address[0]
699    {"addr:housenumber": "222", "addr:street": "Northwest Pineapple Avenue"}
700    >>> address[1]
701    ["addr:unit"]
702    ```
703
704    Args:
705        address_string (str): The address string to process.
706
707    Returns:
708        tuple[dict[str, str], list[str | None]]:
709        The processed address string and the removed fields.
710    """
711    if not address_string.strip().replace("\n", ""):
712        raise ValueError("Address string cannot be empty")
713
714    # Segment the address string into fields
715    cleaned, removed = _parse_address(address_string)
716
717    # Apply field-specific processors
718    cleaned = _apply_field_processors(cleaned)
719
720    # Drop fields that were parsed but came out empty
721    cleaned = {key: value for key, value in cleaned.items() if value}
722
723    # Validate and return
724    return _validate_and_clean(cleaned, removed)

Process address strings.

>>> get_address("345 MAPLE RD, COUNTRYSIDE, PA 24680-0198")[0]
{"addr:housenumber": "345", "addr:street": "Maple Road",
"addr:city": "Countryside", "addr:state": "PA", "addr:postcode": "24680-0198"}
>>> get_address("777 Strawberry St.")[0]
{"addr:housenumber": "777", "addr:street": "Strawberry Street"}
>>> address = get_address("222 NW Pineapple Ave Suite A Unit B")
>>> address[0]
{"addr:housenumber": "222", "addr:street": "Northwest Pineapple Avenue"}
>>> address[1]
["addr:unit"]
Arguments:
  • address_string (str): The address string to process.
Returns:

tuple[dict[str, str], list[str | None]]: The processed address string and the removed fields.

def get_phone(phone: str) -> str:
727def get_phone(phone: str) -> str:
728    """Format phone numbers to the US and Canadian standard format of `+1-XXX-XXX-XXXX`.
729
730    ```python
731    >>> get_phone("2029009019")
732    "+1-202-900-9019"
733    >>> get_phone("(202) 900-9019")
734    "+1-202-900-9019"
735    >>> get_phone("202-900-901")
736    ValueError: Invalid phone number: 202-900-901
737    ```
738
739    Args:
740        phone (str): The phone number to format.
741
742    Returns:
743        str: The formatted phone number.
744
745    Raises:
746        ValueError: If the phone number is invalid.
747    """
748    phone_valid = phone_comp.search(phone)
749    if phone_valid:
750        return (
751            f"+1-{phone_valid.group(1)}-{phone_valid.group(2)}-{phone_valid.group(3)}"
752        )
753    raise ValueError(f"Invalid phone number: {phone}")

Format phone numbers to the US and Canadian standard format of +1-XXX-XXX-XXXX.

>>> get_phone("2029009019")
"+1-202-900-9019"
>>> get_phone("(202) 900-9019")
"+1-202-900-9019"
>>> get_phone("202-900-901")
ValueError: Invalid phone number: 202-900-901
Arguments:
  • phone (str): The phone number to format.
Returns:

str: The formatted phone number.

Raises:
  • ValueError: If the phone number is invalid.
def get_hours(value: str, no_wrap: bool = False) -> str:
1129def get_hours(value: str, no_wrap: bool = False) -> str:
1130    """Process opening hours strings into the OSM `opening_hours` format.
1131
1132    ```python
1133    >>> get_hours("Mo-Fr 08:00-12:00,13:00-17:30")
1134    "Mo-Fr 08:00-12:00,13:00-17:30"
1135    >>> get_hours("Monday to Friday 9am-5pm, Saturday 9am-12pm")
1136    "Mo-Fr 09:00-17:00; Sa 09:00-12:00"
1137    >>> get_hours("Closed")
1138    "off"
1139    >>> get_hours("Mo-Fr 09:00-17:00; PH off")
1140    "Mo-Fr 09:00-17:00; PH off"
1141    >>> get_hours("Mo-Fr sunrise-sunset")
1142    "Mo-Fr sunrise-sunset"
1143    >>> get_hours("Mo-Fr 9:00-5:00")
1144    "Mo-Fr 09:00-05:00"
1145    >>> get_hours("Mo-Fr 9:00-5:00", no_wrap=True)
1146    "Mo-Fr 09:00-17:00"
1147    >>> get_hours("Mo-Fr 13-2", no_wrap=True)
1148    "Mo-Fr 13:00-02:00"
1149    >>> get_hours("Mo-Fr 9-2", no_wrap=True)
1150    "Mo-Fr 09:00-14:00"
1151    >>> get_hours("Mo-Fr 16-14", no_wrap=True)
1152    Traceback (most recent call last):
1153        ...
1154    ValueError: Invalid time range: '16-14' ends before it starts, and isn't
1155    a plausible overnight closing time
1156    ```
1157
1158    The solar keywords `dawn`, `dusk`, `sunrise`, and `sunset` are accepted
1159    in place of a clock time (on either or both sides of a time span), and
1160    are rendered in lowercase exactly as OSM expects.
1161
1162    `PH` (public holiday) is supported as a special, non-weekday indicator:
1163    it's recognized only as the exact token `PH` (no other aliases or
1164    forms), can never be part of an actual day range (e.g. `PH-Mo` is
1165    rejected), and always sorts after every other day/rule in the output,
1166    regardless of where it appeared in the input.
1167
1168    Calendar/date-based rules -- month names or specific dates (e.g.
1169    `"Jan 1"`), named holidays (e.g. `"Easter"`, `"Thanksgiving"`), and
1170    OSM's "nth weekday of month" notation (e.g. `"Th[4]"` for the fourth
1171    Thursday) -- aren't supported. Rather than risk silently mangling them,
1172    any input containing one of these raises `ValueError` instead of
1173    returning a partial or incorrect result.
1174
1175    Note:
1176        This function has a few quirks to be aware of:
1177
1178        - Only strings with English day names (and abbreviations) are
1179          supported; day names in other languages will not be recognized.
1180        - If the same day is mentioned more than once anywhere in the
1181          string, the later mention wins and silently overrides the
1182          earlier one (e.g. "Mo 09:00-17:00, Mo 10:00-14:00" resolves to
1183          just "Mo 10:00-14:00").
1184        - If a string contains both a day range and a specific day that
1185          overlap (e.g. "Mo-Fr 09:00-17:00, We 10:00-14:00"), the explicit,
1186          more specific day definition takes precedence over the range for
1187          that day.
1188        - Days that are not mentioned anywhere in the input string are
1189          simply omitted from the output; they are not assumed to be
1190          `off`.
1191        - Bare, ambiguous times with no am/pm marker (e.g. "9-5") are, by
1192          default (`no_wrap=False`), assumed to be typical AM-to-PM
1193          business hours, so "9-5" resolves to "09:00-17:00" rather than
1194          being rejected or resolved another way.
1195        - A colon on its own does *not* make a time unambiguous -- it only
1196          fixes the minutes, not whether the hour means AM or PM. By
1197          default, a colon form with no am/pm marker (e.g. the "5:00" in
1198          "9:00-5:00") is instead assumed to already be correct 24-hour
1199          time, taken completely at face value. This is a common source of
1200          surprise: "Mo-Fr 9:00-5:00" resolves to "Mo-Fr 09:00-05:00" (open
1201          until 5 AM, not 5 PM) rather than the probably-intended
1202          "09:00-17:00". Use `no_wrap=True` to avoid this by resolving
1203          such times the same way bare digits are.
1204
1205    Args:
1206        value (str): The opening hours string to process.
1207        no_wrap (bool): If True, disables the "assume overnight" behavior
1208            for ambiguous times -- a bare digit (e.g. the "9" or "5" in
1209            "9-5") or a bare colon form (e.g. the "9:00" or "5:00" in
1210            "9:00-5:00") with no am/pm marker either way. Instead of ever
1211            wrapping such a span past midnight, each side is resolved so
1212            the times stay on the same day whenever that's possible:
1213
1214            - If the start hour is already > 12 (e.g. the "13" in "13-2"),
1215              there's no 12-hour reading of it, so neither side is
1216              adjusted and the span is taken at face value ("13:00-02:00"
1217              -- this is still an overnight span, but an intentional one,
1218              since the start hour couldn't have meant anything else).
1219            - Otherwise the start hour is assumed to be AM, and the end
1220              hour is only shifted to PM (by adding 12) when it's
1221              numerically less than or equal to the start hour -- just
1222              enough to keep the span from running backwards on the same
1223              day. So "9-5"/"9:00-5:00" becomes "09:00-17:00" (5 <= 9,
1224              shifted to PM), "9-14" stays "09:00-14:00" (14 is already
1225              later than 9, no shift needed), and "9-2" becomes
1226              "09:00-14:00" (2 <= 9, shifted to PM -- *not* "09:00-02:00",
1227              since nothing here signals an overnight span was intended).
1228
1229            This only affects ambiguous times. A real am/pm marker (e.g.
1230            "9am-5pm" or "10pm-2am") always resolves the same way whether
1231            or not `no_wrap` is set, and can still cross midnight when both
1232            sides are explicit, since that's an intentional signal rather
1233            than a guess.
1234
1235            `no_wrap` doesn't disable the separate, pre-existing check
1236            that rejects a span which still ends up looking backwards
1237            (end < start) once the end hour is too late in the day to be a
1238            plausible overnight close -- currently 6 AM or later (see
1239            `EARLY_MORNING_CUTOFF_HOUR`). So a bare "16-14" still raises
1240            `ValueError` with `no_wrap=True`, just as it does without it;
1241            `no_wrap` changes how an ambiguous hour is *interpreted*, not
1242            whether an implausible result is still caught. Defaults to
1243            False.
1244
1245    Returns:
1246        str: The formatted opening hours string.
1247
1248    Raises:
1249        ValueError: If the string cannot be parsed, or if it references a
1250            calendar/date-based rule that isn't supported.
1251    """
1252    normalized = _normalize(value)
1253    if not normalized:
1254        raise ValueError("Empty opening hours string.")
1255    _reject_unsupported_calendar_refs(normalized)
1256
1257    stripped = normalized.strip()
1258    if closed_comp.fullmatch(stripped):
1259        return "off"
1260    if day_24_comp.fullmatch(stripped):
1261        return "24/7"
1262
1263    top_segments = [s for s in rule_split_comp.split(normalized) if s.strip()]
1264    top_segments = _merge_day_time_lines(top_segments)
1265    segments = [sub for top in top_segments for sub in _split_space_days(top)]
1266    segments = [sub for seg in segments for sub in _split_comma_days(seg)]
1267    rules = [_parse_segment(segment, no_wrap=no_wrap) for segment in segments]
1268    rules = _merge_duplicate_day_rules(rules)
1269
1270    # only coalesce/reorder when every rule specifies explicit days -- if any
1271    # rule applies to the whole week (e.g. "daily"), leave the input order
1272    # alone since day semantics may be intentionally layered
1273    if rules and all(rule.days for rule in rules):
1274        rules = _coalesce_rules(rules)
1275
1276    output = OpeningHours(rules=rules).to_osm()
1277    _validate_opening_hours_output(output)
1278    return output

Process opening hours strings into the OSM opening_hours format.

>>> get_hours("Mo-Fr 08:00-12:00,13:00-17:30")
"Mo-Fr 08:00-12:00,13:00-17:30"
>>> get_hours("Monday to Friday 9am-5pm, Saturday 9am-12pm")
"Mo-Fr 09:00-17:00; Sa 09:00-12:00"
>>> get_hours("Closed")
"off"
>>> get_hours("Mo-Fr 09:00-17:00; PH off")
"Mo-Fr 09:00-17:00; PH off"
>>> get_hours("Mo-Fr sunrise-sunset")
"Mo-Fr sunrise-sunset"
>>> get_hours("Mo-Fr 9:00-5:00")
"Mo-Fr 09:00-05:00"
>>> get_hours("Mo-Fr 9:00-5:00", no_wrap=True)
"Mo-Fr 09:00-17:00"
>>> get_hours("Mo-Fr 13-2", no_wrap=True)
"Mo-Fr 13:00-02:00"
>>> get_hours("Mo-Fr 9-2", no_wrap=True)
"Mo-Fr 09:00-14:00"
>>> get_hours("Mo-Fr 16-14", no_wrap=True)
Traceback (most recent call last):
    ...
ValueError: Invalid time range: '16-14' ends before it starts, and isn't
a plausible overnight closing time

The solar keywords dawn, dusk, sunrise, and sunset are accepted in place of a clock time (on either or both sides of a time span), and are rendered in lowercase exactly as OSM expects.

PH (public holiday) is supported as a special, non-weekday indicator: it's recognized only as the exact token PH (no other aliases or forms), can never be part of an actual day range (e.g. PH-Mo is rejected), and always sorts after every other day/rule in the output, regardless of where it appeared in the input.

Calendar/date-based rules -- month names or specific dates (e.g. "Jan 1"), named holidays (e.g. "Easter", "Thanksgiving"), and OSM's "nth weekday of month" notation (e.g. "Th[4]" for the fourth Thursday) -- aren't supported. Rather than risk silently mangling them, any input containing one of these raises ValueError instead of returning a partial or incorrect result.

Note:

This function has a few quirks to be aware of:

  • Only strings with English day names (and abbreviations) are supported; day names in other languages will not be recognized.
  • If the same day is mentioned more than once anywhere in the string, the later mention wins and silently overrides the earlier one (e.g. "Mo 09:00-17:00, Mo 10:00-14:00" resolves to just "Mo 10:00-14:00").
  • If a string contains both a day range and a specific day that overlap (e.g. "Mo-Fr 09:00-17:00, We 10:00-14:00"), the explicit, more specific day definition takes precedence over the range for that day.
  • Days that are not mentioned anywhere in the input string are simply omitted from the output; they are not assumed to be off.
  • Bare, ambiguous times with no am/pm marker (e.g. "9-5") are, by default (no_wrap=False), assumed to be typical AM-to-PM business hours, so "9-5" resolves to "09:00-17:00" rather than being rejected or resolved another way.
  • A colon on its own does not make a time unambiguous -- it only fixes the minutes, not whether the hour means AM or PM. By default, a colon form with no am/pm marker (e.g. the "5:00" in "9:00-5:00") is instead assumed to already be correct 24-hour time, taken completely at face value. This is a common source of surprise: "Mo-Fr 9:00-5:00" resolves to "Mo-Fr 09:00-05:00" (open until 5 AM, not 5 PM) rather than the probably-intended "09:00-17:00". Use no_wrap=True to avoid this by resolving such times the same way bare digits are.
Arguments:
  • value (str): The opening hours string to process.
  • no_wrap (bool): If True, disables the "assume overnight" behavior for ambiguous times -- a bare digit (e.g. the "9" or "5" in "9-5") or a bare colon form (e.g. the "9:00" or "5:00" in "9:00-5:00") with no am/pm marker either way. Instead of ever wrapping such a span past midnight, each side is resolved so the times stay on the same day whenever that's possible:

    • If the start hour is already > 12 (e.g. the "13" in "13-2"), there's no 12-hour reading of it, so neither side is adjusted and the span is taken at face value ("13:00-02:00" -- this is still an overnight span, but an intentional one, since the start hour couldn't have meant anything else).
    • Otherwise the start hour is assumed to be AM, and the end hour is only shifted to PM (by adding 12) when it's numerically less than or equal to the start hour -- just enough to keep the span from running backwards on the same day. So "9-5"/"9:00-5:00" becomes "09:00-17:00" (5 <= 9, shifted to PM), "9-14" stays "09:00-14:00" (14 is already later than 9, no shift needed), and "9-2" becomes "09:00-14:00" (2 <= 9, shifted to PM -- not "09:00-02:00", since nothing here signals an overnight span was intended).

    This only affects ambiguous times. A real am/pm marker (e.g. "9am-5pm" or "10pm-2am") always resolves the same way whether or not no_wrap is set, and can still cross midnight when both sides are explicit, since that's an intentional signal rather than a guess.

    no_wrap doesn't disable the separate, pre-existing check that rejects a span which still ends up looking backwards (end < start) once the end hour is too late in the day to be a plausible overnight close -- currently 6 AM or later (see EARLY_MORNING_CUTOFF_HOUR). So a bare "16-14" still raises ValueError with no_wrap=True, just as it does without it; no_wrap changes how an ambiguous hour is interpreted, not whether an implausible result is still caught. Defaults to False.

Returns:

str: The formatted opening hours string.

Raises:
  • ValueError: If the string cannot be parsed, or if it references a calendar/date-based rule that isn't supported.
def get_times(value: str) -> str:
1066def get_times(value: str) -> str:
1067    """Process point-in-time strings into the OSM format.
1068
1069    ```python
1070    >>> get_times("Mo-Fr 15:00,18:00,19:00,23:00; Sa 15:00; Su 10:30,23:00")
1071    "Mo-Fr 15:00,18:00,19:00,23:00; Sa 15:00; Su 10:30,23:00"
1072    >>> get_times("Monday to Friday 3pm and 6pm")
1073    "Mo-Fr 15:00,18:00"
1074    >>> get_times("Mo-Fr sunrise,sunset")
1075    "Mo-Fr sunrise,sunset"
1076    >>> get_times("Monday-Friday: 4:15pm Saturday: 1:00pm Sunday: Closed")
1077    "Mo-Fr 16:15; Sa 13:00"
1078    ```
1079
1080    Point-in-time tags have no "closed" concept of their own -- a day with
1081    no scheduled times simply has no entry -- so a "closed"/"off" rule
1082    (e.g. `"Sunday: Closed"`) is dropped entirely rather than raising or
1083    fabricating a value.
1084
1085    The solar keywords `dawn`, `dusk`, `sunrise`, and `sunset` are accepted
1086    in place of a clock time, and are rendered in lowercase exactly as OSM
1087    expects.
1088
1089    Calendar/date-based rules -- month names or specific dates, named
1090    holidays, and OSM's "nth weekday of month" notation (e.g. `"Th[4]"`)
1091    -- aren't supported. Rather than risk silently mangling them, any input
1092    containing one of these raises `ValueError` instead of returning a
1093    partial or incorrect result.
1094
1095    Args:
1096        value (str): The point-in-time string to process.
1097
1098    Returns:
1099        str: The formatted point-in-time string.
1100
1101    Raises:
1102        ValueError: If the string cannot be parsed, or if it references a
1103            calendar/date-based rule that isn't supported.
1104    """
1105    normalized = _normalize(value)
1106    if not normalized:
1107        raise ValueError("Empty collection/service times string.")
1108    _reject_unsupported_calendar_refs(normalized)
1109
1110    top_segments = [s for s in rule_split_comp.split(normalized) if s.strip()]
1111    top_segments = _merge_day_time_lines(top_segments)
1112    segments = [sub for top in top_segments for sub in _split_space_days(top)]
1113    segments = [sub for seg in segments for sub in _split_comma_days(seg)]
1114    rules = [
1115        rule
1116        for rule in (_parse_point_segment(segment) for segment in segments)
1117        if rule is not None
1118    ]
1119    rules = _merge_duplicate_point_day_rules(rules)
1120
1121    if rules and all(rule.days for rule in rules):
1122        rules = _coalesce_point_rules(rules)
1123
1124    output = PointTimes(rules=rules).to_osm()
1125    _validate_point_times_output(output)
1126    return output

Process point-in-time strings into the OSM format.

>>> get_times("Mo-Fr 15:00,18:00,19:00,23:00; Sa 15:00; Su 10:30,23:00")
"Mo-Fr 15:00,18:00,19:00,23:00; Sa 15:00; Su 10:30,23:00"
>>> get_times("Monday to Friday 3pm and 6pm")
"Mo-Fr 15:00,18:00"
>>> get_times("Mo-Fr sunrise,sunset")
"Mo-Fr sunrise,sunset"
>>> get_times("Monday-Friday: 4:15pm Saturday: 1:00pm Sunday: Closed")
"Mo-Fr 16:15; Sa 13:00"

Point-in-time tags have no "closed" concept of their own -- a day with no scheduled times simply has no entry -- so a "closed"/"off" rule (e.g. "Sunday: Closed") is dropped entirely rather than raising or fabricating a value.

The solar keywords dawn, dusk, sunrise, and sunset are accepted in place of a clock time, and are rendered in lowercase exactly as OSM expects.

Calendar/date-based rules -- month names or specific dates, named holidays, and OSM's "nth weekday of month" notation (e.g. "Th[4]") -- aren't supported. Rather than risk silently mangling them, any input containing one of these raises ValueError instead of returning a partial or incorrect result.

Arguments:
  • value (str): The point-in-time string to process.
Returns:

str: The formatted point-in-time string.

Raises:
  • ValueError: If the string cannot be parsed, or if it references a calendar/date-based rule that isn't supported.
def abbrs(value: str) -> str:
210def abbrs(value: str) -> str:
211    """Bundle most common abbreviation expansion functions.
212
213    ```python
214    >>> abbrs("St. Francis")
215    "Saint Francis"
216    >>> abbrs("E Sewell Rd")
217    "East Sewell Road"
218    ```
219
220    Note that `St` is left alone here, since it is ambiguous between `Saint`
221    and `Street` outside a known saint name. `_process_street` resolves it
222    once the token's position in the address is known.
223
224    Args:
225        value (str): String to expand.
226
227    Returns:
228        str: Expanded string.
229    """
230    value = ord_replace(us_replace(mc_replace(get_title(value))))
231
232    # change likely 'St' to 'Saint'
233    value = saint_comp.sub("Saint", value)
234
235    # expand common street and word abbreviations
236    value = abbr_word_comp.sub(_expand_word, value)
237
238    # expand directionals
239    value = dir_fill_comp.sub(direct_expand, value)
240
241    # normalize 'US'
242    value = us_replace(value)
243
244    # uppercase shortened street descriptors
245    value = cap_comp.sub(cap_match, value)
246
247    # remove unremoved abbr periods
248    if "." in value:
249        value = period_comp.sub(r"\1", value)
250
251    # expand 'SR' if no other street types
252    value = sr_comp.sub("State Route", value)
253    return value.strip(" .")

Bundle most common abbreviation expansion functions.

>>> abbrs("St. Francis")
"Saint Francis"
>>> abbrs("E Sewell Rd")
"East Sewell Road"

Note that St is left alone here, since it is ambiguous between Saint and Street outside a known saint name. _process_street resolves it once the token's position in the address is known.

Arguments:
  • value (str): String to expand.
Returns:

str: Expanded string.

def get_title(value: str, single_word: bool = False) -> str:
57def get_title(value: str, single_word: bool = False) -> str:
58    """Fix ALL-CAPS string.
59
60    ```python
61    >>> get_title("PALM BEACH")
62    "Palm Beach"
63    >>> get_title("BOSTON")
64    "BOSTON"
65    >>> get_title("BOSTON", single_word=True)
66    "Boston"
67    >>> get_title("KING'S BEACH")
68    "King's Beach"
69    ```
70
71    Args:
72        value: String to fix.
73        single_word: Whether the string should be fixed even if it is a single word.
74
75    Returns:
76        str: Fixed string.
77    """
78    if (value.isupper() and " " in value) or (value.isupper() and single_word):
79        return mc_replace(" ".join(x.capitalize() for x in value.split()))
80    return value

Fix ALL-CAPS string.

>>> get_title("PALM BEACH")
"Palm Beach"
>>> get_title("BOSTON")
"BOSTON"
>>> get_title("BOSTON", single_word=True)
"Boston"
>>> get_title("KING'S BEACH")
"King's Beach"
Arguments:
  • value: String to fix.
  • single_word: Whether the string should be fixed even if it is a single word.
Returns:

str: Fixed string.

def mc_replace(value: str) -> str:
100def mc_replace(value: str) -> str:
101    """Fix string containing improperly formatted Mc- prefix.
102
103    ```python
104    >>> mc_replace("Fort Mchenry")
105    "Fort McHenry"
106    ```
107
108    Args:
109        value: String to fix.
110
111    Returns:
112        str: Fixed string.
113    """
114    words = []
115    for word in value.split():
116        mc_match = word.partition("Mc")
117        words.append(mc_match[0] + mc_match[1] + mc_match[2].capitalize())
118    return " ".join(words)

Fix string containing improperly formatted Mc- prefix.

>>> mc_replace("Fort Mchenry")
"Fort McHenry"
Arguments:
  • value: String to fix.
Returns:

str: Fixed string.

def us_replace(value: str) -> str:
83def us_replace(value: str) -> str:
84    """Fix string containing improperly formatted US.
85
86    ```python
87    >>> us_replace("U.S. Route 15")
88    "US Route 15"
89    ```
90
91    Args:
92        value: String to fix.
93
94    Returns:
95        str: Fixed string.
96    """
97    return value.replace("U.S.", "US").replace("U. S.", "US").replace("U S ", "US ")

Fix string containing improperly formatted US.

>>> us_replace("U.S. Route 15")
"US Route 15"
Arguments:
  • value: String to fix.
Returns:

str: Fixed string.

def ord_replace(value: str) -> str:
121def ord_replace(value: str) -> str:
122    """Fix string containing improperly capitalized ordinal.
123
124    ```python
125    >>> ord_replace("3Rd St. NW")
126    "3rd St. NW"
127    ```
128
129    Args:
130        value: String to fix.
131
132    Returns:
133        str: Fixed string.
134    """
135    return ord_comp.sub(lower_match, value)

Fix string containing improperly capitalized ordinal.

>>> ord_replace("3Rd St. NW")
"3rd St. NW"
Arguments:
  • value: String to fix.
Returns:

str: Fixed string.

def remove_br_unicode(old: str) -> str:
256def remove_br_unicode(old: str) -> str:
257    """Clean the input string before sending to parser by removing newlines and unicode.
258
259    Args:
260        old (str): String to clean.
261
262    Returns:
263        str: Cleaned string.
264    """
265    if "<br" in old:
266        old = br_comp.sub(",", old)
267    # the pattern only ever matches code points above 0x7F
268    if not old.isascii():
269        old = unicode_comp.sub("", old)
270    return old

Clean the input string before sending to parser by removing newlines and unicode.

Arguments:
  • old (str): String to clean.
Returns:

str: Cleaned string.