Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
108 changes: 84 additions & 24 deletions Makefile
Original file line number Diff line number Diff line change
@@ -1,19 +1,51 @@
datamodel:
poetry run gen-python kg_microbe_merge/schema/merge_schema.yaml > kg_microbe_merge/schema/merge_datamodel.py

kg-microbe-core:
kg-microbe-core: datamodel
poetry run kg merge -m duckdb -s "bacdive, mediadive, madin_etal, rhea_mappings, bactotraits, chebi, ec, envo, go, ncbitaxon, upa" --merge-label $@
poetry run python kg_microbe_merge/utils/edge_vs_node_check.py $@
@# If missing nodes were found, add them to the nodes file
@if [ -f data/merged/$@/$@_missing_nodes_with_category.tsv ]; then \
cd data/merged/$@ && \
mv merged-kg_nodes.tsv merged-kg_nodes_part.tsv && \
cat merged-kg_nodes_part.tsv $@_missing_nodes_with_category.tsv > merged-kg_nodes.tsv && \
rm $@_missing_nodes.tsv $@_missing_nodes_with_category.tsv merged-kg_nodes_part.tsv; \
fi

kg-microbe-function:
kg-microbe-function: datamodel
poetry run kg merge -m duckdb -s "bacdive, mediadive, madin_etal, rhea_mappings, bactotraits, chebi, ec, envo, go, ncbitaxon, upa, uniprot_functional_microbes" --merge-label $@
poetry run python kg_microbe_merge/utils/edge_vs_node_check.py $@
@# If missing nodes were found, add them to the nodes file
@if [ -f data/merged/$@/$@_missing_nodes_with_category.tsv ]; then \
cd data/merged/$@ && \
mv merged-kg_nodes.tsv merged-kg_nodes_part.tsv && \
cat merged-kg_nodes_part.tsv $@_missing_nodes_with_category.tsv > merged-kg_nodes.tsv && \
rm $@_missing_nodes.tsv $@_missing_nodes_with_category.tsv merged-kg_nodes_part.tsv; \
fi

kg-microbe-biomedical:
kg-microbe-biomedical: datamodel
poetry run kg merge -m duckdb -s "bacdive, mediadive, madin_etal, rhea_mappings, bactotraits, chebi, ec, envo, go, ncbitaxon, upa, hp, mondo, disbiome, ctd, wallen_etal, uniprot_human" --merge-label $@
poetry run python kg_microbe_merge/utils/edge_vs_node_check.py $@
@# If missing nodes were found, add them to the nodes file
@if [ -f data/merged/$@/$@_missing_nodes_with_category.tsv ]; then \
cd data/merged/$@ && \
mv merged-kg_nodes.tsv merged-kg_nodes_part.tsv && \
cat merged-kg_nodes_part.tsv $@_missing_nodes_with_category.tsv > merged-kg_nodes.tsv && \
rm $@_missing_nodes.tsv $@_missing_nodes_with_category.tsv merged-kg_nodes_part.tsv; \
fi

kg-microbe-biomedical-function:
kg-microbe-biomedical-function: datamodel
poetry run kg merge -m duckdb --merge-label $@
poetry run python kg_microbe_merge/utils/edge_vs_node_check.py $@
@# If missing nodes were found, add them to the nodes file
@if [ -f data/merged/$@/$@_missing_nodes_with_category.tsv ]; then \
cd data/merged/$@ && \
mv merged-kg_nodes.tsv merged-kg_nodes_part.tsv && \
cat merged-kg_nodes_part.tsv $@_missing_nodes_with_category.tsv > merged-kg_nodes.tsv && \
rm $@_missing_nodes.tsv $@_missing_nodes_with_category.tsv merged-kg_nodes_part.tsv; \
fi

kg-microbe-function-cat:
kg-microbe-function-cat: kg-microbe-core
cd data/raw/uniprot_functional_microbes && \
grep UniprotKB: nodes.tsv > nodes_UniprotKB.tsv && \
tail -n +2 edges.tsv | cut -f1,2,3 > edges_data_clean.tsv && \
Expand All @@ -22,37 +54,65 @@ kg-microbe-function-cat:
mkdir -p kg-microbe-function-cat && \
cd kg-microbe-core && \
tail -n +2 merged-kg_edges.tsv > edges_data.tsv && \
head -1 merged-kg_edges.tsv > edges_header.tsv && \
cd ../ && \
cd kg-microbe-function-cat && \
cat ../kg-microbe-core/merged-kg_nodes.tsv ../../raw/uniprot_functional_microbes/nodes_UniprotKB.tsv ../kg-microbe-core/kg-microbe-core_missing_nodes_with_category.tsv > merged-kg_nodes.tsv && \
cat ../kg-microbe-core/edges_header.tsv ../kg-microbe-core/edges_data.tsv ../../raw/uniprot_functional_microbes/edges_data_clean.tsv > merged-kg_edges.tsv && \
cd ../../../ && \
poetry run python kg_microbe_merge/utils/edge_vs_node_check.py kg-microbe-function-cat && \
cd data/merged/kg-microbe-function-cat && \
mv merged-kg_nodes.tsv merged-kg_nodes_part.tsv && \
cat merged-kg_nodes_part.tsv kg-microbe-function-cat_missing_nodes_with_category.tsv > merged-kg_nodes.tsv && \
cd ../../../

kg-microbe-biomedical-function-cat:

cat ../kg-microbe-core/merged-kg_nodes.tsv ../../raw/uniprot_functional_microbes/nodes_UniprotKB.tsv > merged-kg_nodes.tsv && \
cat ../kg-microbe-core/edges_header.tsv ../kg-microbe-core/edges_data.tsv ../../raw/uniprot_functional_microbes/edges_data_clean.tsv > merged-kg_edges.tsv
poetry run python kg_microbe_merge/utils/edge_vs_node_check.py kg-microbe-function-cat
@# If missing nodes were found, add them to the nodes file
@if [ -f data/merged/kg-microbe-function-cat/kg-microbe-function-cat_missing_nodes_with_category.tsv ]; then \
cd data/merged/kg-microbe-function-cat && \
mv merged-kg_nodes.tsv merged-kg_nodes_part.tsv && \
cat merged-kg_nodes_part.tsv kg-microbe-function-cat_missing_nodes_with_category.tsv > merged-kg_nodes.tsv && \
rm kg-microbe-function-cat_missing_nodes.tsv kg-microbe-function-cat_missing_nodes_with_category.tsv merged-kg_nodes_part.tsv; \
fi

kg-microbe-biomedical-function-cat: kg-microbe-biomedical
cd data/raw/uniprot_functional_microbes && \
grep UniprotKB: nodes.tsv > nodes_UniprotKB.tsv && \
tail -n +2 edges.tsv | cut -f1,2,3 > edges_data_clean.tsv && \
head -1 edges.tsv | cut -f1,2,3 > edges_header_clean.tsv && \
cd ../../merged && \
cd kg-microbe-biomedical && \
tail -n +2 merged-kg_edges.tsv > edges_data.tsv && \
head -1 merged-kg_edges.tsv > edges_header.tsv && \
cd ../ && \
mkdir -p kg-microbe-biomedical-function-cat && \
cd kg-microbe-biomedical-function-cat && \
cat ../kg-microbe-biomedical/merged-kg_nodes.tsv ../../raw/uniprot_functional_microbes/nodes_UniprotKB.tsv > merged-kg_nodes.tsv ../kg-microbe-biomedical/kg-microbe-biomedical_missing_nodes_with_category.tsv && \
cat ../kg-microbe-core/edges_header.tsv ../kg-microbe-biomedical/edges_data.tsv ../../raw/uniprot_functional_microbes/edges_data_clean.tsv > merged-kg_edges.tsv && \
cd ../../../ && \
poetry run python kg_microbe_merge/utils/edge_vs_node_check.py kg-microbe-biomedical-function-cat && \
cd data/merged/kg-microbe-biomedical-function-cat && \
mv merged-kg_nodes.tsv merged-kg_nodes_part.tsv && \
cat merged-kg_nodes_part.tsv kg-microbe-biomedical-function-cat_missing_nodes_with_category.tsv > merged-kg_nodes.tsv && \
cd ../../../
cat ../kg-microbe-biomedical/merged-kg_nodes.tsv ../../raw/uniprot_functional_microbes/nodes_UniprotKB.tsv > merged-kg_nodes.tsv && \
cat ../kg-microbe-biomedical/edges_header.tsv ../kg-microbe-biomedical/edges_data.tsv ../../raw/uniprot_functional_microbes/edges_data_clean.tsv > merged-kg_edges.tsv
poetry run python kg_microbe_merge/utils/edge_vs_node_check.py kg-microbe-biomedical-function-cat
@# If missing nodes were found, add them to the nodes file
@if [ -f data/merged/kg-microbe-biomedical-function-cat/kg-microbe-biomedical-function-cat_missing_nodes_with_category.tsv ]; then \
cd data/merged/kg-microbe-biomedical-function-cat && \
mv merged-kg_nodes.tsv merged-kg_nodes_part.tsv && \
cat merged-kg_nodes_part.tsv kg-microbe-biomedical-function-cat_missing_nodes_with_category.tsv > merged-kg_nodes.tsv && \
rm kg-microbe-biomedical-function-cat_missing_nodes.tsv kg-microbe-biomedical-function-cat_missing_nodes_with_category.tsv merged-kg_nodes_part.tsv; \
fi

clean:
# Remove generated datamodel
rm -f kg_microbe_merge/schema/merge_datamodel.py

# Remove all merged directories
rm -rf data/merged/kg-microbe-core
rm -rf data/merged/kg-microbe-function
rm -rf data/merged/kg-microbe-function-cat
rm -rf data/merged/kg-microbe-biomedical
rm -rf data/merged/kg-microbe-biomedical-function
rm -rf data/merged/kg-microbe-biomedical-function-cat

# Remove temporary files created during concatenation
rm -f data/raw/uniprot_functional_microbes/nodes_UniprotKB.tsv
rm -f data/raw/uniprot_functional_microbes/edges_data_clean.tsv
rm -f data/raw/uniprot_functional_microbes/edges_header_clean.tsv

# Remove any edge_data and edge_header files in merged directories
find data/merged -name "edges_data.tsv" -type f -delete 2>/dev/null || true
find data/merged -name "edges_header.tsv" -type f -delete 2>/dev/null || true

@echo "Cleaned all generated files"

include kg-microbe-merge.Makefile

20 changes: 20 additions & 0 deletions hpc/run_edge_vs_node_check.sl
Original file line number Diff line number Diff line change
@@ -0,0 +1,20 @@
#!/bin/bash
#SBATCH --account=m4689
#SBATCH --qos=regular
#SBATCH --constraint=cpu
#SBATCH --time=360
#SBATCH --mem=470GB
#SBATCH --ntasks=1
#SBATCH --output=%j.out
#SBATCH --error=%j.err
#SBATCH -N 1
#SBATCH --mail-type=BEGIN,END
#SBATCH --mail-user=MJoachimiak@lbl.gov

module load python/3.10

cd /global/cfs/cdirs/m4689/master/kg-microbe-merge
source venv-merge/bin/activate

time poetry run python kg_microbe_merge/utils/edge_vs_node_check.py

8 changes: 4 additions & 4 deletions kg_microbe_merge/run.py
Original file line number Diff line number Diff line change
Expand Up @@ -139,11 +139,11 @@ def merge(
merge_kg_object.merged_graph = merged_graph_object
if merge_tool == "duckdb":
if merge_label:
merged_nodes_output_path = MERGED_DATA_DIR / merge_label / "nodes.tsv"
merged_edges_output_path = MERGED_DATA_DIR / merge_label / "edges.tsv"
merged_nodes_output_path = MERGED_DATA_DIR / merge_label / f"merged-kg_nodes.tsv"
merged_edges_output_path = MERGED_DATA_DIR / merge_label / f"merged-kg_edges.tsv"
else:
merged_nodes_output_path = MERGED_DATA_DIR / "nodes.tsv"
merged_edges_output_path = MERGED_DATA_DIR / "edges.tsv"
merged_nodes_output_path = MERGED_DATA_DIR / "merged-kg_nodes.tsv"
merged_edges_output_path = MERGED_DATA_DIR / "merged-kg_edges.tsv"

duckdb_merge(
node_paths,
Expand Down
68 changes: 50 additions & 18 deletions kg_microbe_merge/utils/edge_vs_node_check.py
Original file line number Diff line number Diff line change
Expand Up @@ -26,13 +26,32 @@ def main(kg_path):
con = duckdb.connect()

# Load only the necessary columns from the TSV files into DuckDB tables
con.execute(
f"""
CREATE TABLE edges AS
SELECT subject AS subject_id, object AS object_id
FROM read_csv_auto('data/merged/{kg_path}/merged-kg_edges.tsv', delim='\t', null_padding=true)
"""
)
# Check which columns are available in the edges file
edges_df = con.execute(f"SELECT * FROM read_csv_auto('data/merged/{kg_path}/merged-kg_edges.tsv', delim='\t', null_padding=true) LIMIT 0").df()
columns = list(edges_df.columns)

# Handle different column configurations
if 'subject' in columns and 'object' in columns:
# Standard case with subject and object columns
con.execute(
f"""
CREATE TABLE edges AS
SELECT subject AS subject_id, object AS object_id
FROM read_csv_auto('data/merged/{kg_path}/merged-kg_edges.tsv', delim='\t', null_padding=true)
"""
)
elif 'predicate' in columns and 'object' in columns and len(columns) == 2:
# Case for concatenated files where subject column is missing
# We'll only check objects in this case
con.execute(
f"""
CREATE TABLE edges AS
SELECT object AS object_id
FROM read_csv_auto('data/merged/{kg_path}/merged-kg_edges.tsv', delim='\t', null_padding=true)
"""
)
else:
raise ValueError(f"Unexpected column structure in edges file: {columns}")
con.execute(
f"""
CREATE TABLE nodes AS
Expand All @@ -42,17 +61,30 @@ def main(kg_path):
)

# Check whether all subject and object IDs are represented as node IDs
query = """
WITH distinct_ids AS (
SELECT DISTINCT subject_id AS id FROM edges
UNION
SELECT DISTINCT object_id AS id FROM edges
)
SELECT distinct_ids.id
FROM distinct_ids
LEFT JOIN nodes ON distinct_ids.id = nodes.id
WHERE nodes.id IS NULL
"""
if 'subject' in columns and 'object' in columns:
# Standard case - check both subject and object IDs
query = """
WITH distinct_ids AS (
SELECT DISTINCT subject_id AS id FROM edges
UNION
SELECT DISTINCT object_id AS id FROM edges
)
SELECT distinct_ids.id
FROM distinct_ids
LEFT JOIN nodes ON distinct_ids.id = nodes.id
WHERE nodes.id IS NULL
"""
else:
# Concatenated case - only check object IDs
query = """
WITH distinct_ids AS (
SELECT DISTINCT object_id AS id FROM edges
)
SELECT distinct_ids.id
FROM distinct_ids
LEFT JOIN nodes ON distinct_ids.id = nodes.id
WHERE nodes.id IS NULL
"""

# Execute the query and fetch all results
missing_ids = con.execute(query).fetchall()
Expand Down
Loading