atlus
Convert raw address and phone number strings into the OSM format.
atlus is a Python package to convert raw address, phone number, and opening
hours strings into the OSM format. It's designed to be used with US and Canadian
phone numbers and addresses.
>>> import atlus
>>> atlus.abbrs("St. Francis")
"Saint Francis"
>>> atlus.get_address("789 Oak Dr, Smallville California, 98765")[0]
{"addr:housenumber": "789", "addr:street": "Oak Drive", "addr:city": "Smallville",
"addr:state": "CA", "addr:postcode": "98765"}
>>> atlus.get_phone("(202) 900-9019")
"+1-202-900-9019"
>>> atlus.get_hours("Monday to Friday 9am-5pm, Saturday 9am-12pm")
"Mo-Fr 09:00-17:00; Sa 09:00-12:00"
>>> atlus.get_times("Mo-Fr 15:00,18:00,19:00,23:00; Sa 15:00; Su 10:30,23:00")
"Mo-Fr 15:00,18:00,19:00,23:00; Sa 15:00; Su 10:30,23:00"
1"""Convert raw address and phone number strings into the OSM format. 2 3`atlus` is a Python package to convert raw address, phone number, and opening 4hours strings into the OSM format. It's designed to be used with US and Canadian 5phone numbers and addresses. 6 7```python 8>>> import atlus 9>>> atlus.abbrs("St. Francis") 10"Saint Francis" 11>>> atlus.get_address("789 Oak Dr, Smallville California, 98765")[0] 12{"addr:housenumber": "789", "addr:street": "Oak Drive", "addr:city": "Smallville", 13 "addr:state": "CA", "addr:postcode": "98765"} 14>>> atlus.get_phone("(202) 900-9019") 15"+1-202-900-9019" 16>>> atlus.get_hours("Monday to Friday 9am-5pm, Saturday 9am-12pm") 17"Mo-Fr 09:00-17:00; Sa 09:00-12:00" 18>>> atlus.get_times("Mo-Fr 15:00,18:00,19:00,23:00; Sa 15:00; Su 10:30,23:00") 19"Mo-Fr 15:00,18:00,19:00,23:00; Sa 15:00; Su 10:30,23:00" 20``` 21 22""" 23 24# SPDX-FileCopyrightText: 2024-present Will <wahubsch@gmail.com> 25# 26# SPDX-License-Identifier: MIT 27 28from . import atlus, hours, resources 29from .atlus import ( 30 abbrs, 31 get_address, 32 get_phone, 33 get_title, 34 mc_replace, 35 ord_replace, 36 remove_br_unicode, 37 us_replace, 38) 39from .hours import get_hours, get_times 40 41__all__ = [ 42 "get_address", 43 "get_phone", 44 "get_hours", 45 "get_times", 46 "abbrs", 47 "get_title", 48 "mc_replace", 49 "us_replace", 50 "ord_replace", 51 "remove_br_unicode", 52 "atlus", 53 "hours", 54 "resources", 55]
688def get_address(address_string: str) -> tuple[dict[str, str], list[str | None]]: 689 """Process address strings. 690 691 ```python 692 >>> get_address("345 MAPLE RD, COUNTRYSIDE, PA 24680-0198")[0] 693 {"addr:housenumber": "345", "addr:street": "Maple Road", 694 "addr:city": "Countryside", "addr:state": "PA", "addr:postcode": "24680-0198"} 695 >>> get_address("777 Strawberry St.")[0] 696 {"addr:housenumber": "777", "addr:street": "Strawberry Street"} 697 >>> address = get_address("222 NW Pineapple Ave Suite A Unit B") 698 >>> address[0] 699 {"addr:housenumber": "222", "addr:street": "Northwest Pineapple Avenue"} 700 >>> address[1] 701 ["addr:unit"] 702 ``` 703 704 Args: 705 address_string (str): The address string to process. 706 707 Returns: 708 tuple[dict[str, str], list[str | None]]: 709 The processed address string and the removed fields. 710 """ 711 if not address_string.strip().replace("\n", ""): 712 raise ValueError("Address string cannot be empty") 713 714 # Segment the address string into fields 715 cleaned, removed = _parse_address(address_string) 716 717 # Apply field-specific processors 718 cleaned = _apply_field_processors(cleaned) 719 720 # Drop fields that were parsed but came out empty 721 cleaned = {key: value for key, value in cleaned.items() if value} 722 723 # Validate and return 724 return _validate_and_clean(cleaned, removed)
Process address strings.
>>> get_address("345 MAPLE RD, COUNTRYSIDE, PA 24680-0198")[0]
{"addr:housenumber": "345", "addr:street": "Maple Road",
"addr:city": "Countryside", "addr:state": "PA", "addr:postcode": "24680-0198"}
>>> get_address("777 Strawberry St.")[0]
{"addr:housenumber": "777", "addr:street": "Strawberry Street"}
>>> address = get_address("222 NW Pineapple Ave Suite A Unit B")
>>> address[0]
{"addr:housenumber": "222", "addr:street": "Northwest Pineapple Avenue"}
>>> address[1]
["addr:unit"]
Arguments:
- address_string (str): The address string to process.
Returns:
tuple[dict[str, str], list[str | None]]: The processed address string and the removed fields.
727def get_phone(phone: str) -> str: 728 """Format phone numbers to the US and Canadian standard format of `+1-XXX-XXX-XXXX`. 729 730 ```python 731 >>> get_phone("2029009019") 732 "+1-202-900-9019" 733 >>> get_phone("(202) 900-9019") 734 "+1-202-900-9019" 735 >>> get_phone("202-900-901") 736 ValueError: Invalid phone number: 202-900-901 737 ``` 738 739 Args: 740 phone (str): The phone number to format. 741 742 Returns: 743 str: The formatted phone number. 744 745 Raises: 746 ValueError: If the phone number is invalid. 747 """ 748 phone_valid = phone_comp.search(phone) 749 if phone_valid: 750 return ( 751 f"+1-{phone_valid.group(1)}-{phone_valid.group(2)}-{phone_valid.group(3)}" 752 ) 753 raise ValueError(f"Invalid phone number: {phone}")
Format phone numbers to the US and Canadian standard format of +1-XXX-XXX-XXXX.
>>> get_phone("2029009019")
"+1-202-900-9019"
>>> get_phone("(202) 900-9019")
"+1-202-900-9019"
>>> get_phone("202-900-901")
ValueError: Invalid phone number: 202-900-901
Arguments:
- phone (str): The phone number to format.
Returns:
str: The formatted phone number.
Raises:
- ValueError: If the phone number is invalid.
1129def get_hours(value: str, no_wrap: bool = False) -> str: 1130 """Process opening hours strings into the OSM `opening_hours` format. 1131 1132 ```python 1133 >>> get_hours("Mo-Fr 08:00-12:00,13:00-17:30") 1134 "Mo-Fr 08:00-12:00,13:00-17:30" 1135 >>> get_hours("Monday to Friday 9am-5pm, Saturday 9am-12pm") 1136 "Mo-Fr 09:00-17:00; Sa 09:00-12:00" 1137 >>> get_hours("Closed") 1138 "off" 1139 >>> get_hours("Mo-Fr 09:00-17:00; PH off") 1140 "Mo-Fr 09:00-17:00; PH off" 1141 >>> get_hours("Mo-Fr sunrise-sunset") 1142 "Mo-Fr sunrise-sunset" 1143 >>> get_hours("Mo-Fr 9:00-5:00") 1144 "Mo-Fr 09:00-05:00" 1145 >>> get_hours("Mo-Fr 9:00-5:00", no_wrap=True) 1146 "Mo-Fr 09:00-17:00" 1147 >>> get_hours("Mo-Fr 13-2", no_wrap=True) 1148 "Mo-Fr 13:00-02:00" 1149 >>> get_hours("Mo-Fr 9-2", no_wrap=True) 1150 "Mo-Fr 09:00-14:00" 1151 >>> get_hours("Mo-Fr 16-14", no_wrap=True) 1152 Traceback (most recent call last): 1153 ... 1154 ValueError: Invalid time range: '16-14' ends before it starts, and isn't 1155 a plausible overnight closing time 1156 ``` 1157 1158 The solar keywords `dawn`, `dusk`, `sunrise`, and `sunset` are accepted 1159 in place of a clock time (on either or both sides of a time span), and 1160 are rendered in lowercase exactly as OSM expects. 1161 1162 `PH` (public holiday) is supported as a special, non-weekday indicator: 1163 it's recognized only as the exact token `PH` (no other aliases or 1164 forms), can never be part of an actual day range (e.g. `PH-Mo` is 1165 rejected), and always sorts after every other day/rule in the output, 1166 regardless of where it appeared in the input. 1167 1168 Calendar/date-based rules -- month names or specific dates (e.g. 1169 `"Jan 1"`), named holidays (e.g. `"Easter"`, `"Thanksgiving"`), and 1170 OSM's "nth weekday of month" notation (e.g. `"Th[4]"` for the fourth 1171 Thursday) -- aren't supported. Rather than risk silently mangling them, 1172 any input containing one of these raises `ValueError` instead of 1173 returning a partial or incorrect result. 1174 1175 Note: 1176 This function has a few quirks to be aware of: 1177 1178 - Only strings with English day names (and abbreviations) are 1179 supported; day names in other languages will not be recognized. 1180 - If the same day is mentioned more than once anywhere in the 1181 string, the later mention wins and silently overrides the 1182 earlier one (e.g. "Mo 09:00-17:00, Mo 10:00-14:00" resolves to 1183 just "Mo 10:00-14:00"). 1184 - If a string contains both a day range and a specific day that 1185 overlap (e.g. "Mo-Fr 09:00-17:00, We 10:00-14:00"), the explicit, 1186 more specific day definition takes precedence over the range for 1187 that day. 1188 - Days that are not mentioned anywhere in the input string are 1189 simply omitted from the output; they are not assumed to be 1190 `off`. 1191 - Bare, ambiguous times with no am/pm marker (e.g. "9-5") are, by 1192 default (`no_wrap=False`), assumed to be typical AM-to-PM 1193 business hours, so "9-5" resolves to "09:00-17:00" rather than 1194 being rejected or resolved another way. 1195 - A colon on its own does *not* make a time unambiguous -- it only 1196 fixes the minutes, not whether the hour means AM or PM. By 1197 default, a colon form with no am/pm marker (e.g. the "5:00" in 1198 "9:00-5:00") is instead assumed to already be correct 24-hour 1199 time, taken completely at face value. This is a common source of 1200 surprise: "Mo-Fr 9:00-5:00" resolves to "Mo-Fr 09:00-05:00" (open 1201 until 5 AM, not 5 PM) rather than the probably-intended 1202 "09:00-17:00". Use `no_wrap=True` to avoid this by resolving 1203 such times the same way bare digits are. 1204 1205 Args: 1206 value (str): The opening hours string to process. 1207 no_wrap (bool): If True, disables the "assume overnight" behavior 1208 for ambiguous times -- a bare digit (e.g. the "9" or "5" in 1209 "9-5") or a bare colon form (e.g. the "9:00" or "5:00" in 1210 "9:00-5:00") with no am/pm marker either way. Instead of ever 1211 wrapping such a span past midnight, each side is resolved so 1212 the times stay on the same day whenever that's possible: 1213 1214 - If the start hour is already > 12 (e.g. the "13" in "13-2"), 1215 there's no 12-hour reading of it, so neither side is 1216 adjusted and the span is taken at face value ("13:00-02:00" 1217 -- this is still an overnight span, but an intentional one, 1218 since the start hour couldn't have meant anything else). 1219 - Otherwise the start hour is assumed to be AM, and the end 1220 hour is only shifted to PM (by adding 12) when it's 1221 numerically less than or equal to the start hour -- just 1222 enough to keep the span from running backwards on the same 1223 day. So "9-5"/"9:00-5:00" becomes "09:00-17:00" (5 <= 9, 1224 shifted to PM), "9-14" stays "09:00-14:00" (14 is already 1225 later than 9, no shift needed), and "9-2" becomes 1226 "09:00-14:00" (2 <= 9, shifted to PM -- *not* "09:00-02:00", 1227 since nothing here signals an overnight span was intended). 1228 1229 This only affects ambiguous times. A real am/pm marker (e.g. 1230 "9am-5pm" or "10pm-2am") always resolves the same way whether 1231 or not `no_wrap` is set, and can still cross midnight when both 1232 sides are explicit, since that's an intentional signal rather 1233 than a guess. 1234 1235 `no_wrap` doesn't disable the separate, pre-existing check 1236 that rejects a span which still ends up looking backwards 1237 (end < start) once the end hour is too late in the day to be a 1238 plausible overnight close -- currently 6 AM or later (see 1239 `EARLY_MORNING_CUTOFF_HOUR`). So a bare "16-14" still raises 1240 `ValueError` with `no_wrap=True`, just as it does without it; 1241 `no_wrap` changes how an ambiguous hour is *interpreted*, not 1242 whether an implausible result is still caught. Defaults to 1243 False. 1244 1245 Returns: 1246 str: The formatted opening hours string. 1247 1248 Raises: 1249 ValueError: If the string cannot be parsed, or if it references a 1250 calendar/date-based rule that isn't supported. 1251 """ 1252 normalized = _normalize(value) 1253 if not normalized: 1254 raise ValueError("Empty opening hours string.") 1255 _reject_unsupported_calendar_refs(normalized) 1256 1257 stripped = normalized.strip() 1258 if closed_comp.fullmatch(stripped): 1259 return "off" 1260 if day_24_comp.fullmatch(stripped): 1261 return "24/7" 1262 1263 top_segments = [s for s in rule_split_comp.split(normalized) if s.strip()] 1264 top_segments = _merge_day_time_lines(top_segments) 1265 segments = [sub for top in top_segments for sub in _split_space_days(top)] 1266 segments = [sub for seg in segments for sub in _split_comma_days(seg)] 1267 rules = [_parse_segment(segment, no_wrap=no_wrap) for segment in segments] 1268 rules = _merge_duplicate_day_rules(rules) 1269 1270 # only coalesce/reorder when every rule specifies explicit days -- if any 1271 # rule applies to the whole week (e.g. "daily"), leave the input order 1272 # alone since day semantics may be intentionally layered 1273 if rules and all(rule.days for rule in rules): 1274 rules = _coalesce_rules(rules) 1275 1276 output = OpeningHours(rules=rules).to_osm() 1277 _validate_opening_hours_output(output) 1278 return output
Process opening hours strings into the OSM opening_hours format.
>>> get_hours("Mo-Fr 08:00-12:00,13:00-17:30")
"Mo-Fr 08:00-12:00,13:00-17:30"
>>> get_hours("Monday to Friday 9am-5pm, Saturday 9am-12pm")
"Mo-Fr 09:00-17:00; Sa 09:00-12:00"
>>> get_hours("Closed")
"off"
>>> get_hours("Mo-Fr 09:00-17:00; PH off")
"Mo-Fr 09:00-17:00; PH off"
>>> get_hours("Mo-Fr sunrise-sunset")
"Mo-Fr sunrise-sunset"
>>> get_hours("Mo-Fr 9:00-5:00")
"Mo-Fr 09:00-05:00"
>>> get_hours("Mo-Fr 9:00-5:00", no_wrap=True)
"Mo-Fr 09:00-17:00"
>>> get_hours("Mo-Fr 13-2", no_wrap=True)
"Mo-Fr 13:00-02:00"
>>> get_hours("Mo-Fr 9-2", no_wrap=True)
"Mo-Fr 09:00-14:00"
>>> get_hours("Mo-Fr 16-14", no_wrap=True)
Traceback (most recent call last):
...
ValueError: Invalid time range: '16-14' ends before it starts, and isn't
a plausible overnight closing time
The solar keywords dawn, dusk, sunrise, and sunset are accepted
in place of a clock time (on either or both sides of a time span), and
are rendered in lowercase exactly as OSM expects.
PH (public holiday) is supported as a special, non-weekday indicator:
it's recognized only as the exact token PH (no other aliases or
forms), can never be part of an actual day range (e.g. PH-Mo is
rejected), and always sorts after every other day/rule in the output,
regardless of where it appeared in the input.
Calendar/date-based rules -- month names or specific dates (e.g.
"Jan 1"), named holidays (e.g. "Easter", "Thanksgiving"), and
OSM's "nth weekday of month" notation (e.g. "Th[4]" for the fourth
Thursday) -- aren't supported. Rather than risk silently mangling them,
any input containing one of these raises ValueError instead of
returning a partial or incorrect result.
Note:
This function has a few quirks to be aware of:
- Only strings with English day names (and abbreviations) are supported; day names in other languages will not be recognized.
- If the same day is mentioned more than once anywhere in the string, the later mention wins and silently overrides the earlier one (e.g. "Mo 09:00-17:00, Mo 10:00-14:00" resolves to just "Mo 10:00-14:00").
- If a string contains both a day range and a specific day that overlap (e.g. "Mo-Fr 09:00-17:00, We 10:00-14:00"), the explicit, more specific day definition takes precedence over the range for that day.
- Days that are not mentioned anywhere in the input string are simply omitted from the output; they are not assumed to be
off.- Bare, ambiguous times with no am/pm marker (e.g. "9-5") are, by default (
no_wrap=False), assumed to be typical AM-to-PM business hours, so "9-5" resolves to "09:00-17:00" rather than being rejected or resolved another way.- A colon on its own does not make a time unambiguous -- it only fixes the minutes, not whether the hour means AM or PM. By default, a colon form with no am/pm marker (e.g. the "5:00" in "9:00-5:00") is instead assumed to already be correct 24-hour time, taken completely at face value. This is a common source of surprise: "Mo-Fr 9:00-5:00" resolves to "Mo-Fr 09:00-05:00" (open until 5 AM, not 5 PM) rather than the probably-intended "09:00-17:00". Use
no_wrap=Trueto avoid this by resolving such times the same way bare digits are.
Arguments:
- value (str): The opening hours string to process.
no_wrap (bool): If True, disables the "assume overnight" behavior for ambiguous times -- a bare digit (e.g. the "9" or "5" in "9-5") or a bare colon form (e.g. the "9:00" or "5:00" in "9:00-5:00") with no am/pm marker either way. Instead of ever wrapping such a span past midnight, each side is resolved so the times stay on the same day whenever that's possible:
- If the start hour is already > 12 (e.g. the "13" in "13-2"), there's no 12-hour reading of it, so neither side is adjusted and the span is taken at face value ("13:00-02:00" -- this is still an overnight span, but an intentional one, since the start hour couldn't have meant anything else).
- Otherwise the start hour is assumed to be AM, and the end hour is only shifted to PM (by adding 12) when it's numerically less than or equal to the start hour -- just enough to keep the span from running backwards on the same day. So "9-5"/"9:00-5:00" becomes "09:00-17:00" (5 <= 9, shifted to PM), "9-14" stays "09:00-14:00" (14 is already later than 9, no shift needed), and "9-2" becomes "09:00-14:00" (2 <= 9, shifted to PM -- not "09:00-02:00", since nothing here signals an overnight span was intended).
This only affects ambiguous times. A real am/pm marker (e.g. "9am-5pm" or "10pm-2am") always resolves the same way whether or not
no_wrapis set, and can still cross midnight when both sides are explicit, since that's an intentional signal rather than a guess.no_wrapdoesn't disable the separate, pre-existing check that rejects a span which still ends up looking backwards (end < start) once the end hour is too late in the day to be a plausible overnight close -- currently 6 AM or later (seeEARLY_MORNING_CUTOFF_HOUR). So a bare "16-14" still raisesValueErrorwithno_wrap=True, just as it does without it;no_wrapchanges how an ambiguous hour is interpreted, not whether an implausible result is still caught. Defaults to False.
Returns:
str: The formatted opening hours string.
Raises:
- ValueError: If the string cannot be parsed, or if it references a calendar/date-based rule that isn't supported.
1066def get_times(value: str) -> str: 1067 """Process point-in-time strings into the OSM format. 1068 1069 ```python 1070 >>> get_times("Mo-Fr 15:00,18:00,19:00,23:00; Sa 15:00; Su 10:30,23:00") 1071 "Mo-Fr 15:00,18:00,19:00,23:00; Sa 15:00; Su 10:30,23:00" 1072 >>> get_times("Monday to Friday 3pm and 6pm") 1073 "Mo-Fr 15:00,18:00" 1074 >>> get_times("Mo-Fr sunrise,sunset") 1075 "Mo-Fr sunrise,sunset" 1076 >>> get_times("Monday-Friday: 4:15pm Saturday: 1:00pm Sunday: Closed") 1077 "Mo-Fr 16:15; Sa 13:00" 1078 ``` 1079 1080 Point-in-time tags have no "closed" concept of their own -- a day with 1081 no scheduled times simply has no entry -- so a "closed"/"off" rule 1082 (e.g. `"Sunday: Closed"`) is dropped entirely rather than raising or 1083 fabricating a value. 1084 1085 The solar keywords `dawn`, `dusk`, `sunrise`, and `sunset` are accepted 1086 in place of a clock time, and are rendered in lowercase exactly as OSM 1087 expects. 1088 1089 Calendar/date-based rules -- month names or specific dates, named 1090 holidays, and OSM's "nth weekday of month" notation (e.g. `"Th[4]"`) 1091 -- aren't supported. Rather than risk silently mangling them, any input 1092 containing one of these raises `ValueError` instead of returning a 1093 partial or incorrect result. 1094 1095 Args: 1096 value (str): The point-in-time string to process. 1097 1098 Returns: 1099 str: The formatted point-in-time string. 1100 1101 Raises: 1102 ValueError: If the string cannot be parsed, or if it references a 1103 calendar/date-based rule that isn't supported. 1104 """ 1105 normalized = _normalize(value) 1106 if not normalized: 1107 raise ValueError("Empty collection/service times string.") 1108 _reject_unsupported_calendar_refs(normalized) 1109 1110 top_segments = [s for s in rule_split_comp.split(normalized) if s.strip()] 1111 top_segments = _merge_day_time_lines(top_segments) 1112 segments = [sub for top in top_segments for sub in _split_space_days(top)] 1113 segments = [sub for seg in segments for sub in _split_comma_days(seg)] 1114 rules = [ 1115 rule 1116 for rule in (_parse_point_segment(segment) for segment in segments) 1117 if rule is not None 1118 ] 1119 rules = _merge_duplicate_point_day_rules(rules) 1120 1121 if rules and all(rule.days for rule in rules): 1122 rules = _coalesce_point_rules(rules) 1123 1124 output = PointTimes(rules=rules).to_osm() 1125 _validate_point_times_output(output) 1126 return output
Process point-in-time strings into the OSM format.
>>> get_times("Mo-Fr 15:00,18:00,19:00,23:00; Sa 15:00; Su 10:30,23:00")
"Mo-Fr 15:00,18:00,19:00,23:00; Sa 15:00; Su 10:30,23:00"
>>> get_times("Monday to Friday 3pm and 6pm")
"Mo-Fr 15:00,18:00"
>>> get_times("Mo-Fr sunrise,sunset")
"Mo-Fr sunrise,sunset"
>>> get_times("Monday-Friday: 4:15pm Saturday: 1:00pm Sunday: Closed")
"Mo-Fr 16:15; Sa 13:00"
Point-in-time tags have no "closed" concept of their own -- a day with
no scheduled times simply has no entry -- so a "closed"/"off" rule
(e.g. "Sunday: Closed") is dropped entirely rather than raising or
fabricating a value.
The solar keywords dawn, dusk, sunrise, and sunset are accepted
in place of a clock time, and are rendered in lowercase exactly as OSM
expects.
Calendar/date-based rules -- month names or specific dates, named
holidays, and OSM's "nth weekday of month" notation (e.g. "Th[4]")
-- aren't supported. Rather than risk silently mangling them, any input
containing one of these raises ValueError instead of returning a
partial or incorrect result.
Arguments:
- value (str): The point-in-time string to process.
Returns:
str: The formatted point-in-time string.
Raises:
- ValueError: If the string cannot be parsed, or if it references a calendar/date-based rule that isn't supported.
210def abbrs(value: str) -> str: 211 """Bundle most common abbreviation expansion functions. 212 213 ```python 214 >>> abbrs("St. Francis") 215 "Saint Francis" 216 >>> abbrs("E Sewell Rd") 217 "East Sewell Road" 218 ``` 219 220 Note that `St` is left alone here, since it is ambiguous between `Saint` 221 and `Street` outside a known saint name. `_process_street` resolves it 222 once the token's position in the address is known. 223 224 Args: 225 value (str): String to expand. 226 227 Returns: 228 str: Expanded string. 229 """ 230 value = ord_replace(us_replace(mc_replace(get_title(value)))) 231 232 # change likely 'St' to 'Saint' 233 value = saint_comp.sub("Saint", value) 234 235 # expand common street and word abbreviations 236 value = abbr_word_comp.sub(_expand_word, value) 237 238 # expand directionals 239 value = dir_fill_comp.sub(direct_expand, value) 240 241 # normalize 'US' 242 value = us_replace(value) 243 244 # uppercase shortened street descriptors 245 value = cap_comp.sub(cap_match, value) 246 247 # remove unremoved abbr periods 248 if "." in value: 249 value = period_comp.sub(r"\1", value) 250 251 # expand 'SR' if no other street types 252 value = sr_comp.sub("State Route", value) 253 return value.strip(" .")
Bundle most common abbreviation expansion functions.
>>> abbrs("St. Francis")
"Saint Francis"
>>> abbrs("E Sewell Rd")
"East Sewell Road"
Note that St is left alone here, since it is ambiguous between Saint
and Street outside a known saint name. _process_street resolves it
once the token's position in the address is known.
Arguments:
- value (str): String to expand.
Returns:
str: Expanded string.
57def get_title(value: str, single_word: bool = False) -> str: 58 """Fix ALL-CAPS string. 59 60 ```python 61 >>> get_title("PALM BEACH") 62 "Palm Beach" 63 >>> get_title("BOSTON") 64 "BOSTON" 65 >>> get_title("BOSTON", single_word=True) 66 "Boston" 67 >>> get_title("KING'S BEACH") 68 "King's Beach" 69 ``` 70 71 Args: 72 value: String to fix. 73 single_word: Whether the string should be fixed even if it is a single word. 74 75 Returns: 76 str: Fixed string. 77 """ 78 if (value.isupper() and " " in value) or (value.isupper() and single_word): 79 return mc_replace(" ".join(x.capitalize() for x in value.split())) 80 return value
Fix ALL-CAPS string.
>>> get_title("PALM BEACH")
"Palm Beach"
>>> get_title("BOSTON")
"BOSTON"
>>> get_title("BOSTON", single_word=True)
"Boston"
>>> get_title("KING'S BEACH")
"King's Beach"
Arguments:
- value: String to fix.
- single_word: Whether the string should be fixed even if it is a single word.
Returns:
str: Fixed string.
100def mc_replace(value: str) -> str: 101 """Fix string containing improperly formatted Mc- prefix. 102 103 ```python 104 >>> mc_replace("Fort Mchenry") 105 "Fort McHenry" 106 ``` 107 108 Args: 109 value: String to fix. 110 111 Returns: 112 str: Fixed string. 113 """ 114 words = [] 115 for word in value.split(): 116 mc_match = word.partition("Mc") 117 words.append(mc_match[0] + mc_match[1] + mc_match[2].capitalize()) 118 return " ".join(words)
Fix string containing improperly formatted Mc- prefix.
>>> mc_replace("Fort Mchenry")
"Fort McHenry"
Arguments:
- value: String to fix.
Returns:
str: Fixed string.
83def us_replace(value: str) -> str: 84 """Fix string containing improperly formatted US. 85 86 ```python 87 >>> us_replace("U.S. Route 15") 88 "US Route 15" 89 ``` 90 91 Args: 92 value: String to fix. 93 94 Returns: 95 str: Fixed string. 96 """ 97 return value.replace("U.S.", "US").replace("U. S.", "US").replace("U S ", "US ")
Fix string containing improperly formatted US.
>>> us_replace("U.S. Route 15")
"US Route 15"
Arguments:
- value: String to fix.
Returns:
str: Fixed string.
121def ord_replace(value: str) -> str: 122 """Fix string containing improperly capitalized ordinal. 123 124 ```python 125 >>> ord_replace("3Rd St. NW") 126 "3rd St. NW" 127 ``` 128 129 Args: 130 value: String to fix. 131 132 Returns: 133 str: Fixed string. 134 """ 135 return ord_comp.sub(lower_match, value)
Fix string containing improperly capitalized ordinal.
>>> ord_replace("3Rd St. NW")
"3rd St. NW"
Arguments:
- value: String to fix.
Returns:
str: Fixed string.
256def remove_br_unicode(old: str) -> str: 257 """Clean the input string before sending to parser by removing newlines and unicode. 258 259 Args: 260 old (str): String to clean. 261 262 Returns: 263 str: Cleaned string. 264 """ 265 if "<br" in old: 266 old = br_comp.sub(",", old) 267 # the pattern only ever matches code points above 0x7F 268 if not old.isascii(): 269 old = unicode_comp.sub("", old) 270 return old
Clean the input string before sending to parser by removing newlines and unicode.
Arguments:
- old (str): String to clean.
Returns:
str: Cleaned string.