init public release
This commit is contained in:
@@ -0,0 +1,105 @@
|
||||
"""
|
||||
Reproduction: vanilla BioCypher corrupts already-written part files when a later
|
||||
batch of the same node type introduces a new property.
|
||||
|
||||
Run:
|
||||
pip install biocypher
|
||||
python reproduce_header_misalignment.py
|
||||
|
||||
Tested against BioCypher 0.15.1. Self-contained: writes its own config files
|
||||
into a temp dir, needs no Neo4j (offline CSV writer only).
|
||||
|
||||
Background
|
||||
----------
|
||||
MeDaX ingests data in successive batches and extends the schema on the fly:
|
||||
when a node/edge type or attribute first appears, it is added to the schema so
|
||||
no information is dropped. BioCypher's Neo4j batch writer rewrites the *header*
|
||||
file on every write_nodes() call, but leaves the *part files already on disk*
|
||||
untouched. As soon as a new attribute widens/reorders the header, the rows in
|
||||
earlier part files no longer line up with it -> silent column misalignment on
|
||||
neo4j-admin import.
|
||||
|
||||
Exact location upstream:
|
||||
biocypher/output/write/graph/_neo4j.py
|
||||
_Neo4jBatchWriter._write_node_headers() (and ._write_edge_headers())
|
||||
-> the `if os.path.exists(header_path): logger.warning("... Overwriting.")`
|
||||
branch rewrites the header but never reconciles existing -part*.csv files.
|
||||
"""
|
||||
|
||||
import glob
|
||||
import os
|
||||
import sys
|
||||
import tempfile
|
||||
|
||||
from biocypher import BioCypher
|
||||
|
||||
workdir = tempfile.mkdtemp(prefix="bc_repro_")
|
||||
os.chdir(workdir)
|
||||
|
||||
with open("schema_config.yaml", "w") as fh:
|
||||
fh.write(
|
||||
"patient:\n"
|
||||
" represented_as: node\n"
|
||||
" input_label: patient\n"
|
||||
" is_a: entity\n" # no fixed `properties:` -> inferred from data per batch
|
||||
)
|
||||
|
||||
with open("biocypher_config.yaml", "w") as fh:
|
||||
fh.write(
|
||||
"biocypher:\n"
|
||||
" dbms: neo4j\n"
|
||||
" offline: true\n"
|
||||
" strict_mode: false\n"
|
||||
" schema_config_path: schema_config.yaml\n"
|
||||
" head_ontology:\n"
|
||||
" url: https://raw.githubusercontent.com/biolink/biolink-model/v3.2.1/biolink-model.owl.ttl\n"
|
||||
" root_node: entity\n"
|
||||
)
|
||||
|
||||
bc = BioCypher(
|
||||
biocypher_config_path="biocypher_config.yaml",
|
||||
schema_config_path="schema_config.yaml",
|
||||
output_directory="out",
|
||||
)
|
||||
|
||||
# Batch 1: patients described by {name, age}
|
||||
bc.write_nodes(
|
||||
[
|
||||
("p1", "patient", {"name": "Alice", "age": 30}),
|
||||
("p2", "patient", {"name": "Bob", "age": 41}),
|
||||
],
|
||||
force=True,
|
||||
)
|
||||
|
||||
# Batch 2: same node type, now with an extra `diagnosis` attribute.
|
||||
bc.write_nodes(
|
||||
[("p3", "patient", {"name": "Carol", "age": 52, "diagnosis": "E11"})],
|
||||
force=True,
|
||||
)
|
||||
|
||||
# --- Inspect what ended up on disk -----------------------------------------
|
||||
header_path = "out/Patient-header.csv"
|
||||
header_cols = open(header_path).read().strip().split(";")
|
||||
n_header = len(header_cols)
|
||||
|
||||
print("HEADER :", header_cols, f"({n_header} columns)")
|
||||
ok = True
|
||||
for part in sorted(glob.glob("out/Patient-part*.csv")):
|
||||
for row in open(part).read().splitlines():
|
||||
n_row = len(row.split(";"))
|
||||
flag = "" if n_row == n_header else " <-- MISALIGNED"
|
||||
if n_row != n_header:
|
||||
ok = False
|
||||
print(f"{os.path.basename(part)}: {n_row} columns{flag} {row}")
|
||||
|
||||
print()
|
||||
if ok:
|
||||
print("PASS: every part file matches the header.")
|
||||
sys.exit(0)
|
||||
else:
|
||||
print(
|
||||
"FAIL: a part file written before the schema grew no longer matches the\n"
|
||||
" header. On import, its values land under the wrong columns\n"
|
||||
" (e.g. the node id ends up under `diagnosis`, :LABEL ends up empty)."
|
||||
)
|
||||
sys.exit(1)
|
||||
Reference in New Issue
Block a user