Skip to content

Commit 4bbceb0

Browse files
fix(parsers): honour HL7 v2 MSH delimiters, repetitions and escape sequences (#382)
The HL7 v2 parser hard-coded "|" and "^", so it ignored the delimiters a message declares in MSH-1/MSH-2. It never split "~" repetitions and never decoded escape sequences. As a result PID-3 "12345~98765" came out as the patient id, "\S\" / "\T\" stayed in text values, and a message using "#" as its field separator parsed to zero messages. Each MSH segment now sets the delimiters for its message: the field separator comes from seg[3] and the component, repetition, escape and subcomponent characters from MSH-2. Any that are missing default to ^~\&. Those delimiters stay in force until the next MSH. Segment ids are detected with seg[:3], and a recognised segment must be followed by the current field separator. Single-valued columns (ids, names, codes, dates, units, status) take the first repetition. OBX-5 is a repeating field, so the observation value keeps every repetition, joined with "~". Fields are split before escapes are decoded, so an escaped delimiter such as \R\ never splits a value. \F\ \S\ \T\ \R\ \E\ are decoded using the message's own delimiters, and other escapes (\H\, \N\, \Xhh\, \.br\) are kept verbatim. Whole-field values are re-emitted with the standard ^ / & / ~ separators. Standard |^~\& messages without escapes, and without repetitions outside OBX-5, produce the same frames, metadata and warnings as before. Closes #261
1 parent 93c3d88 commit 4bbceb0

2 files changed

Lines changed: 512 additions & 60 deletions

File tree

‎src/freshdata/parsers/hl7v2.py‎

Lines changed: 164 additions & 60 deletions
Original file line numberDiff line numberDiff line change
@@ -5,14 +5,25 @@
55
shaped for the healthcare domain pack. Observation code systems are mapped to their
66
canonical URIs (LOINC ``http://loinc.org``, SNOMED ``http://snomed.info/sct``, ICD-10).
77
8+
Delimiters are read from each message's MSH segment (HL7 v2 Chapter 2): MSH-1 is the
9+
field separator and MSH-2 holds the component, repetition, escape and subcomponent
10+
characters (``^~\\&`` when absent). They apply to every segment up to the next MSH.
11+
Single-valued output columns take the first repetition of a field; OBX-5 (observation
12+
value) is a repeating field and keeps every repetition. Fields are split first, then the
13+
delimiter escape sequences ``\\F\\ \\S\\ \\T\\ \\R\\ \\E\\`` are decoded to the literal
14+
characters. Other escapes (``\\H\\``, ``\\N\\``, ``\\X..\\``, ``\\.br\\`` ...) are left
15+
verbatim. Whole-field values (e.g. OBX-5) are re-emitted with the standard ``^`` / ``&``
16+
/ ``~`` component / subcomponent / repetition separators, so output does not depend on a
17+
message's custom delimiters.
18+
819
This is a structural parser for the common segments, not a full HL7 v2 conformance
920
engine: unrecognized segments are counted in :attr:`ParseResult.warnings`, and the OBX
1021
component layout follows the usual ORU convention.
1122
"""
1223

1324
from __future__ import annotations
1425

15-
from typing import Any
26+
from typing import Any, NamedTuple
1627

1728
import pandas as pd
1829

@@ -31,21 +42,90 @@
3142
}
3243

3344
# PV1-2 patient class -> human label.
34-
_PATIENT_CLASS = {"I": "inpatient", "O": "outpatient", "E": "emergency",
35-
"P": "preadmit", "R": "recurring", "B": "obstetrics"}
45+
_PATIENT_CLASS = {
46+
"I": "inpatient",
47+
"O": "outpatient",
48+
"E": "emergency",
49+
"P": "preadmit",
50+
"R": "recurring",
51+
"B": "obstetrics",
52+
}
3653

54+
_KNOWN_SEGMENTS = frozenset({"PID", "PV1", "OBR", "OBX"})
55+
56+
57+
class _Delimiters(NamedTuple):
58+
"""The five HL7 v2 message delimiters declared by MSH-1 / MSH-2."""
59+
60+
field: str = "|"
61+
component: str = "^"
62+
repetition: str = "~"
63+
escape: str = "\\"
64+
subcomponent: str = "&"
65+
66+
@classmethod
67+
def from_msh(cls, segment: str) -> _Delimiters:
68+
"""Read MSH-1 (``segment[3]``) and MSH-2; missing characters use the defaults."""
69+
default = cls()
70+
if len(segment) < 4:
71+
return default
72+
field_sep = segment[3]
73+
encoding = segment[4:].split(field_sep, 1)[0]
74+
chars = [encoding[i] if i < len(encoding) else d for i, d in enumerate(default[1:])]
75+
return cls(field_sep, *chars)
76+
77+
def decode(self, text: str) -> str:
78+
"""Decode the delimiter escape sequences; leave any other escape verbatim."""
79+
esc = self.escape
80+
if esc not in text:
81+
return text
82+
literal = {
83+
"F": self.field,
84+
"S": self.component,
85+
"T": self.subcomponent,
86+
"R": self.repetition,
87+
"E": esc,
88+
}
89+
out: list[str] = []
90+
i = 0
91+
while True:
92+
start = text.find(esc, i)
93+
end = text.find(esc, start + 1) if start != -1 else -1
94+
if end == -1:
95+
out.append(text[i:])
96+
return "".join(out)
97+
out.append(text[i:start])
98+
code = text[start + 1 : end]
99+
out.append(literal.get(code, text[start : end + 1]))
100+
i = end + 1
101+
102+
def value(self, field: str, *, repeating: bool = False) -> str:
103+
"""Decoded *field* re-emitted with the standard ``^`` / ``&`` / ``~`` separators.
104+
105+
Only the first repetition is kept unless *repeating* is true, in which case every
106+
repetition is kept and joined with ``~``. Splitting happens before escapes are
107+
decoded, so an escaped delimiter (e.g. ``\\R\\``) never splits the value.
108+
"""
109+
reps = field.strip().split(self.repetition)
110+
return "~".join(
111+
"^".join(
112+
"&".join(self.decode(sub) for sub in comp.split(self.subcomponent))
113+
for comp in rep.split(self.component)
114+
).strip()
115+
for rep in (reps if repeating else reps[:1])
116+
)
37117

38-
def _comp(field: str, n: int) -> str:
39-
"""1-based HL7 component *n* of a field (``DOE^JOHN`` -> 1='DOE')."""
40-
if not field:
41-
return ""
42-
parts = field.split("^")
43-
return parts[n - 1].strip() if 0 < n <= len(parts) else ""
118+
def component_of(self, field: str, n: int) -> str:
119+
"""1-based component *n* of the first repetition (``DOE^JOHN`` -> 1='DOE')."""
120+
if not field:
121+
return ""
122+
parts = field.strip().split(self.repetition, 1)[0].split(self.component)
123+
return self.decode(parts[n - 1]).strip() if 0 < n <= len(parts) else ""
44124

45125

46126
def _field(fields: list[str], n: int) -> str:
47-
"""Field *n* of a split segment (``fields[0]`` is the segment id)."""
48-
return fields[n].strip() if n < len(fields) else ""
127+
"""Raw field *n* of a split segment (``fields[0]`` is the segment id)."""
128+
return fields[n] if n < len(fields) else ""
49129

50130

51131
class HL7v2Parser(Parser):
@@ -57,8 +137,11 @@ class HL7v2Parser(Parser):
57137
def parse(self, source: Any) -> ParseResult:
58138
text = self.read_text(source)
59139
# HL7 segments are CR-separated; tolerate LF / CRLF too.
60-
segments = [s for s in text.replace("\r\n", "\r").replace("\n", "\r").split("\r")
61-
if s.strip()]
140+
segments = [
141+
s.lstrip()
142+
for s in text.replace("\r\n", "\r").replace("\n", "\r").split("\r")
143+
if s.strip()
144+
]
62145

63146
patients: list[dict[str, Any]] = []
64147
encounters: list[dict[str, Any]] = []
@@ -71,64 +154,85 @@ def parse(self, source: Any) -> ParseResult:
71154
current_pid: str | None = None
72155
current_order: str | None = None
73156
message_type = ""
157+
delim = _Delimiters()
74158

75159
for seg in segments:
76-
fields = seg.split("|")
77-
seg_id = fields[0].strip()
160+
seg_id = seg[:3]
161+
is_msh = seg_id == "MSH" and (len(seg) == 3 or not seg[3].isalnum())
78162

79-
if seg_id == "MSH":
163+
if is_msh:
164+
delim = _Delimiters.from_msh(seg)
80165
msg_index += 1
81166
current_pid = None
82167
current_order = None
83168
# MSH is offset by one (MSH-1 is the field separator itself), so
84-
# MSH-9 (message type, e.g. "ADT^A01") is fields[8].
85-
message_type = fields[8].strip() if len(fields) > 8 else ""
169+
# MSH-2 is fields[1] and MSH-9 (message type, e.g. "ADT^A01") is fields[8].
170+
message_type = delim.value(_field(seg.split(delim.field), 8))
171+
continue
172+
173+
d = delim
174+
fields = seg.split(d.field)
175+
if seg_id not in _KNOWN_SEGMENTS or (len(seg) > 3 and seg[3] != d.field):
176+
key = fields[0].strip()
177+
unknown[key] = unknown.get(key, 0) + 1
86178
elif seg_id == "PID":
87-
current_pid = _comp(_field(fields, 3), 1) or f"MSG{msg_index}"
88-
patients.append({
89-
"patient_id": current_pid,
90-
"family_name": _comp(_field(fields, 5), 1),
91-
"given_name": _comp(_field(fields, 5), 2),
92-
"birth_date": _field(fields, 7),
93-
"gender": _field(fields, 8),
94-
})
179+
current_pid = d.component_of(_field(fields, 3), 1) or f"MSG{msg_index}"
180+
patients.append(
181+
{
182+
"patient_id": current_pid,
183+
"family_name": d.component_of(_field(fields, 5), 1),
184+
"given_name": d.component_of(_field(fields, 5), 2),
185+
"birth_date": d.value(_field(fields, 7)),
186+
"gender": d.value(_field(fields, 8)),
187+
}
188+
)
95189
elif seg_id == "PV1":
96-
encounters.append({
97-
"patient_id": current_pid or f"MSG{msg_index}",
98-
"visit_number": _comp(_field(fields, 19), 1),
99-
"class_code": _field(fields, 2),
100-
"class": _PATIENT_CLASS.get(_field(fields, 2).upper(), _field(fields, 2)),
101-
"location": _comp(_field(fields, 3), 1),
102-
})
190+
class_code = d.value(_field(fields, 2))
191+
encounters.append(
192+
{
193+
"patient_id": current_pid or f"MSG{msg_index}",
194+
"visit_number": d.component_of(_field(fields, 19), 1),
195+
"class_code": class_code,
196+
"class": _PATIENT_CLASS.get(class_code.upper(), class_code),
197+
"location": d.component_of(_field(fields, 3), 1),
198+
}
199+
)
103200
elif seg_id == "OBR":
104201
service = _field(fields, 4)
105-
current_order = (_comp(_field(fields, 3), 1)
106-
or _comp(_field(fields, 2), 1) or None)
107-
orders.append({
108-
"patient_id": current_pid or f"MSG{msg_index}",
109-
"order_id": current_order,
110-
"placer_order": _comp(_field(fields, 2), 1),
111-
"filler_order": _comp(_field(fields, 3), 1),
112-
"service_code": _comp(service, 1),
113-
"service_display": _comp(service, 2),
114-
"service_system": _comp(service, 3),
115-
"observed_at": _field(fields, 7),
116-
})
117-
elif seg_id == "OBX":
118-
system = _comp(_field(fields, 3), 3)
119-
observations.append({
120-
"patient_id": current_pid or f"MSG{msg_index}",
121-
"order_id": current_order,
122-
"code": _comp(_field(fields, 3), 1),
123-
"display": _comp(_field(fields, 3), 2),
124-
"code_system": _CODE_SYSTEMS.get(system.upper(), system),
125-
"value": _field(fields, 5),
126-
"unit": _field(fields, 6),
127-
"status": _field(fields, 11),
128-
"observed_at": _field(fields, 14),
129-
})
130-
else:
131-
unknown[seg_id] = unknown.get(seg_id, 0) + 1
202+
current_order = (
203+
d.component_of(_field(fields, 3), 1)
204+
or d.component_of(_field(fields, 2), 1)
205+
or None
206+
)
207+
orders.append(
208+
{
209+
"patient_id": current_pid or f"MSG{msg_index}",
210+
"order_id": current_order,
211+
"placer_order": d.component_of(_field(fields, 2), 1),
212+
"filler_order": d.component_of(_field(fields, 3), 1),
213+
"service_code": d.component_of(service, 1),
214+
"service_display": d.component_of(service, 2),
215+
"service_system": d.component_of(service, 3),
216+
"observed_at": d.value(_field(fields, 7)),
217+
}
218+
)
219+
else: # OBX
220+
code = _field(fields, 3)
221+
system = d.component_of(code, 3)
222+
observations.append(
223+
{
224+
"patient_id": current_pid or f"MSG{msg_index}",
225+
"order_id": current_order,
226+
"code": d.component_of(code, 1),
227+
"display": d.component_of(code, 2),
228+
"code_system": _CODE_SYSTEMS.get(system.upper(), system),
229+
# OBX-5 repeats: keep every value, joined with "~".
230+
"value": d.value(_field(fields, 5), repeating=True),
231+
"unit": d.value(_field(fields, 6)),
232+
"status": d.value(_field(fields, 11)),
233+
"observed_at": d.value(_field(fields, 14)),
234+
}
235+
)
132236

133237
if unknown:
134238
listed = ", ".join(f"{k}({v})" for k, v in sorted(unknown.items()))

0 commit comments

Comments
 (0)