Skip to content

Commit f80471b

Browse files
c-dilkstongtongcaoveronique
authored
feat: HIPO to NPZ converter (#1365)
* develop engine for CVT denoising * CVT Hits reconstruction stand-alone service. * update CVTDenoiseEngine with using BST::Hits and BMT::Hits as input * update byte for record of hit status by AI prediction * hipo to npz converter python script to view to converted output * hipo to npz converter python script to view to converted output * yaml file to create FML bank * refactor!: remove denoising stuff from this branch * feat: override JVM args * feat: default schame dir * feat: ci * ci: ls * ci: .hipo * refactor: hipoToNpz -> hipo2npz --------- Co-authored-by: tongtongcao <tongtongcao1@gmail.com> Co-authored-by: veronique <veronique@mac>
1 parent 8b153fa commit f80471b

4 files changed

Lines changed: 1189 additions & 4 deletions

File tree

‎.github/workflows/ci.yml‎

Lines changed: 35 additions & 4 deletions
Original file line numberDiff line numberDiff line change
@@ -225,11 +225,18 @@ jobs:
225225
run: |
226226
tar xzvf clara.tar.gz
227227
tar xzvf coatjava.tar.gz
228+
- run: ls
228229
- name: run test
229-
run: |
230-
ls -lhtr
231-
./bin/run-clara -y ./etc/services/rgd-clarode.yml -t 4 -n 500 -c ./clara -o ./tmp ./clas_018779.evio.00001
232-
ls -lhtr
230+
run: ./bin/run-clara -y ./etc/services/rgd-clarode.yml -t 4 -n 500 -c ./clara -o ./tmp ./clas_018779.evio.00001
231+
- name: ls tmp
232+
run: ls -lhtr tmp
233+
- name: rename
234+
run: mv -v tmp/rec_clas_018779.evio.00001.hipo rec.hipo
235+
- uses: actions/upload-artifact@v7
236+
with:
237+
name: test_clara_result
238+
retention-days: 1
239+
path: rec.hipo
233240

234241
test_coatjava:
235242
needs: [ build ]
@@ -329,6 +336,30 @@ jobs:
329336
- name: test run-groovy
330337
run: coatjava/bin/run-groovy validation/advanced-tests/test-run-groovy.groovy
331338

339+
test_qcddat:
340+
needs: test_clara
341+
runs-on: ubuntu-latest
342+
steps:
343+
- uses: actions/checkout@v7
344+
- name: Set up JDK
345+
uses: actions/setup-java@v5.5.0
346+
with:
347+
java-version: ${{ env.JAVA_VERSION }}
348+
distribution: ${{ env.java_distribution }}
349+
cache: maven
350+
- uses: actions/download-artifact@v8
351+
with:
352+
name: test_clara_result
353+
- uses: actions/download-artifact@v8
354+
with:
355+
name: build_ubuntu-latest
356+
- name: untar build
357+
run: tar xzvf coatjava.tar.gz
358+
- name: hipo2npz
359+
run: ./coatjava/bin/hipo2npz rec.hipo rec.npz RUN::config,REC::Event,REC::Particle
360+
- name: hipo2npz-dump
361+
run: ./coatjava/bin/hipo2npz-dump rec.npz 1
362+
332363
# documentation
333364
#############################################################################
334365

‎bin/hipo2npz‎

Lines changed: 12 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,12 @@
1+
#!/bin/bash
2+
3+
. `dirname $0`/../libexec/env.sh
4+
5+
split_cli $@
6+
7+
export MALLOC_ARENA_MAX=1
8+
9+
java ${JAVA_OPTS-} -Xmx1536m -Xms1024m -XX:+UseSerialGC ${jvm_options[@]} \
10+
-cp ${COATJAVA_CLASSPATH:-''} \
11+
org.jlab.io.hipo.Hipo2Npz \
12+
${class_options[@]}

‎bin/hipo2npz-dump‎

Lines changed: 166 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,166 @@
1+
#!/usr/bin/env python3
2+
"""
3+
Show values for a few events from all banks in an NPZ produced by hipo2npz.
4+
5+
Usage:
6+
python hipo2npz-dump file.npz
7+
python hipo2npz-dump file.npz 3
8+
"""
9+
10+
from __future__ import annotations
11+
12+
import ast
13+
import struct
14+
import sys
15+
from collections import defaultdict
16+
from pathlib import Path
17+
from zipfile import ZipFile
18+
19+
20+
def read_npy_from_bytes(blob: bytes):
21+
if len(blob) < 10 or blob[:6] != b"\x93NUMPY":
22+
raise ValueError("Not a valid .npy payload")
23+
24+
major = blob[6]
25+
minor = blob[7]
26+
27+
if major == 1:
28+
header_len = struct.unpack("<H", blob[8:10])[0]
29+
header_start = 10
30+
elif major in (2, 3):
31+
header_len = struct.unpack("<I", blob[8:12])[0]
32+
header_start = 12
33+
else:
34+
raise ValueError(f"Unsupported .npy version: {major}.{minor}")
35+
36+
header_end = header_start + header_len
37+
header_text = blob[header_start:header_end].decode("latin1").strip()
38+
header = ast.literal_eval(header_text)
39+
40+
descr = header["descr"]
41+
shape = header["shape"]
42+
fortran_order = header["fortran_order"]
43+
44+
if fortran_order:
45+
raise ValueError("Fortran-order arrays are not supported")
46+
47+
if len(shape) != 1:
48+
raise ValueError(f"Only 1D arrays are supported, got shape={shape}")
49+
50+
count = shape[0]
51+
data = blob[header_end:]
52+
53+
fmt_map = {
54+
"|i1": "b",
55+
"<i1": "b",
56+
"|u1": "B",
57+
"<u1": "B",
58+
"<i2": "h",
59+
"<u2": "H",
60+
"<i4": "i",
61+
"<u4": "I",
62+
"<i8": "q",
63+
"<u8": "Q",
64+
"<f4": "f",
65+
"<f8": "d",
66+
}
67+
68+
if descr not in fmt_map:
69+
raise ValueError(f"Unsupported dtype descriptor: {descr}")
70+
71+
fmt = fmt_map[descr]
72+
itemsize = struct.calcsize("<" + fmt)
73+
expected = count * itemsize
74+
if len(data) != expected:
75+
raise ValueError(
76+
f"Data length mismatch for {descr}: expected {expected}, got {len(data)}"
77+
)
78+
79+
values = struct.unpack("<" + str(count) + fmt, data)
80+
return values
81+
82+
83+
def load_npz(npz_path: Path):
84+
arrays = {}
85+
with ZipFile(npz_path, "r") as zf:
86+
for name in zf.namelist():
87+
if not name.endswith(".npy"):
88+
continue
89+
key = name[:-4]
90+
arrays[key] = read_npy_from_bytes(zf.read(name))
91+
return arrays
92+
93+
94+
def collect_banks(keys):
95+
banks = set()
96+
for key in keys:
97+
if key.endswith("__rows_per_event") or key.endswith("__offsets"):
98+
banks.add(key.rsplit("__", 1)[0])
99+
continue
100+
101+
parts = key.split("__")
102+
if len(parts) < 2:
103+
continue
104+
banks.add("__".join(parts[:-1]))
105+
return sorted(banks)
106+
107+
108+
def main():
109+
if len(sys.argv) < 2 or len(sys.argv) > 3:
110+
print("Usage: python3 show_all_npz_events_no_numpy.py <file.npz> [N_EVENTS]")
111+
sys.exit(2)
112+
113+
npz_path = Path(sys.argv[1])
114+
n_events = int(sys.argv[2]) if len(sys.argv) == 3 else 3
115+
116+
if not npz_path.exists():
117+
print(f"File not found: {npz_path}")
118+
sys.exit(1)
119+
120+
arrays = load_npz(npz_path)
121+
banks = collect_banks(arrays.keys())
122+
123+
for bank in banks:
124+
rows_key = f"{bank}__rows_per_event"
125+
offsets_key = f"{bank}__offsets"
126+
127+
if rows_key not in arrays or offsets_key not in arrays:
128+
continue
129+
130+
rows_per_event = arrays[rows_key]
131+
offsets = arrays[offsets_key]
132+
133+
column_keys = sorted(
134+
k for k in arrays.keys()
135+
if k.startswith(bank + "__") and k not in (rows_key, offsets_key)
136+
)
137+
138+
print("=" * 80)
139+
print(f"BANK: {bank}")
140+
print("COLUMNS:")
141+
for key in column_keys:
142+
print(f" {key[len(bank) + 2:]}")
143+
print()
144+
145+
nevt = min(n_events, len(rows_per_event))
146+
147+
for evt in range(nevt):
148+
start = int(offsets[evt])
149+
stop = int(offsets[evt + 1])
150+
nrows = int(rows_per_event[evt])
151+
152+
print(f"Event {evt}: rows={nrows} slice=[{start}:{stop}]")
153+
if nrows == 0:
154+
print(" <no rows>")
155+
print()
156+
continue
157+
158+
for key in column_keys:
159+
col = key[len(bank) + 2:]
160+
values = arrays[key][start:stop]
161+
print(f" {col}: {list(values)}")
162+
print()
163+
164+
165+
if __name__ == "__main__":
166+
main()

0 commit comments

Comments
 (0)