55shaped for the healthcare domain pack. Observation code systems are mapped to their
66canonical URIs (LOINC ``http://loinc.org``, SNOMED ``http://snomed.info/sct``, ICD-10).
77
8+ Delimiters are read from each message's MSH segment (HL7 v2 Chapter 2): MSH-1 is the
9+ field separator and MSH-2 holds the component, repetition, escape and subcomponent
10+ characters (``^~\\ &`` when absent). They apply to every segment up to the next MSH.
11+ Single-valued output columns take the first repetition of a field; OBX-5 (observation
12+ value) is a repeating field and keeps every repetition. Fields are split first, then the
13+ delimiter escape sequences ``\\ F\\ \\ S\\ \\ T\\ \\ R\\ \\ E\\ `` are decoded to the literal
14+ characters. Other escapes (``\\ H\\ ``, ``\\ N\\ ``, ``\\ X..\\ ``, ``\\ .br\\ `` ...) are left
15+ verbatim. Whole-field values (e.g. OBX-5) are re-emitted with the standard ``^`` / ``&``
16+ / ``~`` component / subcomponent / repetition separators, so output does not depend on a
17+ message's custom delimiters.
18+
819This is a structural parser for the common segments, not a full HL7 v2 conformance
920engine: unrecognized segments are counted in :attr:`ParseResult.warnings`, and the OBX
1021component layout follows the usual ORU convention.
1122"""
1223
1324from __future__ import annotations
1425
15- from typing import Any
26+ from typing import Any , NamedTuple
1627
1728import pandas as pd
1829
3142}
3243
3344# PV1-2 patient class -> human label.
34- _PATIENT_CLASS = {"I" : "inpatient" , "O" : "outpatient" , "E" : "emergency" ,
35- "P" : "preadmit" , "R" : "recurring" , "B" : "obstetrics" }
45+ _PATIENT_CLASS = {
46+ "I" : "inpatient" ,
47+ "O" : "outpatient" ,
48+ "E" : "emergency" ,
49+ "P" : "preadmit" ,
50+ "R" : "recurring" ,
51+ "B" : "obstetrics" ,
52+ }
3653
54+ _KNOWN_SEGMENTS = frozenset ({"PID" , "PV1" , "OBR" , "OBX" })
55+
56+
57+ class _Delimiters (NamedTuple ):
58+ """The five HL7 v2 message delimiters declared by MSH-1 / MSH-2."""
59+
60+ field : str = "|"
61+ component : str = "^"
62+ repetition : str = "~"
63+ escape : str = "\\ "
64+ subcomponent : str = "&"
65+
66+ @classmethod
67+ def from_msh (cls , segment : str ) -> _Delimiters :
68+ """Read MSH-1 (``segment[3]``) and MSH-2; missing characters use the defaults."""
69+ default = cls ()
70+ if len (segment ) < 4 :
71+ return default
72+ field_sep = segment [3 ]
73+ encoding = segment [4 :].split (field_sep , 1 )[0 ]
74+ chars = [encoding [i ] if i < len (encoding ) else d for i , d in enumerate (default [1 :])]
75+ return cls (field_sep , * chars )
76+
77+ def decode (self , text : str ) -> str :
78+ """Decode the delimiter escape sequences; leave any other escape verbatim."""
79+ esc = self .escape
80+ if esc not in text :
81+ return text
82+ literal = {
83+ "F" : self .field ,
84+ "S" : self .component ,
85+ "T" : self .subcomponent ,
86+ "R" : self .repetition ,
87+ "E" : esc ,
88+ }
89+ out : list [str ] = []
90+ i = 0
91+ while True :
92+ start = text .find (esc , i )
93+ end = text .find (esc , start + 1 ) if start != - 1 else - 1
94+ if end == - 1 :
95+ out .append (text [i :])
96+ return "" .join (out )
97+ out .append (text [i :start ])
98+ code = text [start + 1 : end ]
99+ out .append (literal .get (code , text [start : end + 1 ]))
100+ i = end + 1
101+
102+ def value (self , field : str , * , repeating : bool = False ) -> str :
103+ """Decoded *field* re-emitted with the standard ``^`` / ``&`` / ``~`` separators.
104+
105+ Only the first repetition is kept unless *repeating* is true, in which case every
106+ repetition is kept and joined with ``~``. Splitting happens before escapes are
107+ decoded, so an escaped delimiter (e.g. ``\\ R\\ ``) never splits the value.
108+ """
109+ reps = field .strip ().split (self .repetition )
110+ return "~" .join (
111+ "^" .join (
112+ "&" .join (self .decode (sub ) for sub in comp .split (self .subcomponent ))
113+ for comp in rep .split (self .component )
114+ ).strip ()
115+ for rep in (reps if repeating else reps [:1 ])
116+ )
37117
38- def _comp ( field : str , n : int ) -> str :
39- """1-based HL7 component *n* of a field (``DOE^JOHN`` -> 1='DOE')."""
40- if not field :
41- return ""
42- parts = field .split ("^" )
43- return parts [n - 1 ].strip () if 0 < n <= len (parts ) else ""
118+ def component_of ( self , field : str , n : int ) -> str :
119+ """1-based component *n* of the first repetition (``DOE^JOHN`` -> 1='DOE')."""
120+ if not field :
121+ return ""
122+ parts = field .strip (). split (self . repetition , 1 )[ 0 ]. split ( self . component )
123+ return self . decode ( parts [n - 1 ]) .strip () if 0 < n <= len (parts ) else ""
44124
45125
46126def _field (fields : list [str ], n : int ) -> str :
47- """Field *n* of a split segment (``fields[0]`` is the segment id)."""
48- return fields [n ]. strip () if n < len (fields ) else ""
127+ """Raw field *n* of a split segment (``fields[0]`` is the segment id)."""
128+ return fields [n ] if n < len (fields ) else ""
49129
50130
51131class HL7v2Parser (Parser ):
@@ -57,8 +137,11 @@ class HL7v2Parser(Parser):
57137 def parse (self , source : Any ) -> ParseResult :
58138 text = self .read_text (source )
59139 # HL7 segments are CR-separated; tolerate LF / CRLF too.
60- segments = [s for s in text .replace ("\r \n " , "\r " ).replace ("\n " , "\r " ).split ("\r " )
61- if s .strip ()]
140+ segments = [
141+ s .lstrip ()
142+ for s in text .replace ("\r \n " , "\r " ).replace ("\n " , "\r " ).split ("\r " )
143+ if s .strip ()
144+ ]
62145
63146 patients : list [dict [str , Any ]] = []
64147 encounters : list [dict [str , Any ]] = []
@@ -71,64 +154,85 @@ def parse(self, source: Any) -> ParseResult:
71154 current_pid : str | None = None
72155 current_order : str | None = None
73156 message_type = ""
157+ delim = _Delimiters ()
74158
75159 for seg in segments :
76- fields = seg . split ( "|" )
77- seg_id = fields [ 0 ]. strip ( )
160+ seg_id = seg [: 3 ]
161+ is_msh = seg_id == "MSH" and ( len ( seg ) == 3 or not seg [ 3 ]. isalnum () )
78162
79- if seg_id == "MSH" :
163+ if is_msh :
164+ delim = _Delimiters .from_msh (seg )
80165 msg_index += 1
81166 current_pid = None
82167 current_order = None
83168 # MSH is offset by one (MSH-1 is the field separator itself), so
84- # MSH-9 (message type, e.g. "ADT^A01") is fields[8].
85- message_type = fields [8 ].strip () if len (fields ) > 8 else ""
169+ # MSH-2 is fields[1] and MSH-9 (message type, e.g. "ADT^A01") is fields[8].
170+ message_type = delim .value (_field (seg .split (delim .field ), 8 ))
171+ continue
172+
173+ d = delim
174+ fields = seg .split (d .field )
175+ if seg_id not in _KNOWN_SEGMENTS or (len (seg ) > 3 and seg [3 ] != d .field ):
176+ key = fields [0 ].strip ()
177+ unknown [key ] = unknown .get (key , 0 ) + 1
86178 elif seg_id == "PID" :
87- current_pid = _comp (_field (fields , 3 ), 1 ) or f"MSG{ msg_index } "
88- patients .append ({
89- "patient_id" : current_pid ,
90- "family_name" : _comp (_field (fields , 5 ), 1 ),
91- "given_name" : _comp (_field (fields , 5 ), 2 ),
92- "birth_date" : _field (fields , 7 ),
93- "gender" : _field (fields , 8 ),
94- })
179+ current_pid = d .component_of (_field (fields , 3 ), 1 ) or f"MSG{ msg_index } "
180+ patients .append (
181+ {
182+ "patient_id" : current_pid ,
183+ "family_name" : d .component_of (_field (fields , 5 ), 1 ),
184+ "given_name" : d .component_of (_field (fields , 5 ), 2 ),
185+ "birth_date" : d .value (_field (fields , 7 )),
186+ "gender" : d .value (_field (fields , 8 )),
187+ }
188+ )
95189 elif seg_id == "PV1" :
96- encounters .append ({
97- "patient_id" : current_pid or f"MSG{ msg_index } " ,
98- "visit_number" : _comp (_field (fields , 19 ), 1 ),
99- "class_code" : _field (fields , 2 ),
100- "class" : _PATIENT_CLASS .get (_field (fields , 2 ).upper (), _field (fields , 2 )),
101- "location" : _comp (_field (fields , 3 ), 1 ),
102- })
190+ class_code = d .value (_field (fields , 2 ))
191+ encounters .append (
192+ {
193+ "patient_id" : current_pid or f"MSG{ msg_index } " ,
194+ "visit_number" : d .component_of (_field (fields , 19 ), 1 ),
195+ "class_code" : class_code ,
196+ "class" : _PATIENT_CLASS .get (class_code .upper (), class_code ),
197+ "location" : d .component_of (_field (fields , 3 ), 1 ),
198+ }
199+ )
103200 elif seg_id == "OBR" :
104201 service = _field (fields , 4 )
105- current_order = (_comp (_field (fields , 3 ), 1 )
106- or _comp (_field (fields , 2 ), 1 ) or None )
107- orders .append ({
108- "patient_id" : current_pid or f"MSG{ msg_index } " ,
109- "order_id" : current_order ,
110- "placer_order" : _comp (_field (fields , 2 ), 1 ),
111- "filler_order" : _comp (_field (fields , 3 ), 1 ),
112- "service_code" : _comp (service , 1 ),
113- "service_display" : _comp (service , 2 ),
114- "service_system" : _comp (service , 3 ),
115- "observed_at" : _field (fields , 7 ),
116- })
117- elif seg_id == "OBX" :
118- system = _comp (_field (fields , 3 ), 3 )
119- observations .append ({
120- "patient_id" : current_pid or f"MSG{ msg_index } " ,
121- "order_id" : current_order ,
122- "code" : _comp (_field (fields , 3 ), 1 ),
123- "display" : _comp (_field (fields , 3 ), 2 ),
124- "code_system" : _CODE_SYSTEMS .get (system .upper (), system ),
125- "value" : _field (fields , 5 ),
126- "unit" : _field (fields , 6 ),
127- "status" : _field (fields , 11 ),
128- "observed_at" : _field (fields , 14 ),
129- })
130- else :
131- unknown [seg_id ] = unknown .get (seg_id , 0 ) + 1
202+ current_order = (
203+ d .component_of (_field (fields , 3 ), 1 )
204+ or d .component_of (_field (fields , 2 ), 1 )
205+ or None
206+ )
207+ orders .append (
208+ {
209+ "patient_id" : current_pid or f"MSG{ msg_index } " ,
210+ "order_id" : current_order ,
211+ "placer_order" : d .component_of (_field (fields , 2 ), 1 ),
212+ "filler_order" : d .component_of (_field (fields , 3 ), 1 ),
213+ "service_code" : d .component_of (service , 1 ),
214+ "service_display" : d .component_of (service , 2 ),
215+ "service_system" : d .component_of (service , 3 ),
216+ "observed_at" : d .value (_field (fields , 7 )),
217+ }
218+ )
219+ else : # OBX
220+ code = _field (fields , 3 )
221+ system = d .component_of (code , 3 )
222+ observations .append (
223+ {
224+ "patient_id" : current_pid or f"MSG{ msg_index } " ,
225+ "order_id" : current_order ,
226+ "code" : d .component_of (code , 1 ),
227+ "display" : d .component_of (code , 2 ),
228+ "code_system" : _CODE_SYSTEMS .get (system .upper (), system ),
229+ # OBX-5 repeats: keep every value, joined with "~".
230+ "value" : d .value (_field (fields , 5 ), repeating = True ),
231+ "unit" : d .value (_field (fields , 6 )),
232+ "status" : d .value (_field (fields , 11 )),
233+ "observed_at" : d .value (_field (fields , 14 )),
234+ }
235+ )
132236
133237 if unknown :
134238 listed = ", " .join (f"{ k } ({ v } )" for k , v in sorted (unknown .items ()))
0 commit comments