Skip to content

contentStats

contentStats #34

Workflow file for this run

# Workflow for checking the content of the files regarding primary and secondary IDs
name: contentStats
on:
workflow_dispatch:
pull_request: # tests whether it is working on PR
paths:
- '.github/workflows/contentStats.yml'
schedule:
- cron: "0 0 1,15 * *" # Run the workflow on the 1st and 15th day of each month
jobs:
contentStats:
runs-on: ubuntu-latest
permissions:
contents: write
# step 1: checkout the repository
steps:
- name: Download GitHub repo for the queries
uses: actions/checkout@v5
# step 2: run the bash commands to count the content for different databases
- name: ChEBI content
run: |
##Download ChEBI data
wget https://raw.githubusercontent.com/sec2pri/mapping_preprocessing/refs/heads/main/datasources/chebi/data/ChEBI_secID2priID.tsv
##count unique entries in first column
count=$(cut -f1 ChEBI_secID2priID.tsv | sort | uniq | wc -l) #for Primary IDs
echo "Count Primary IDs: $count"
##Count unique entries in second column
count=$(cut -f2 ChEBI_secID2priID.tsv | sort | uniq | wc -l) #for Secondary IDs
echo "Count Secondary IDs: $count"
##Count total number of lines for matches between Primary and Secondary IDs
echo "Total ID matches" |wc -l ChEBI_secID2priID.tsv
- name: HGNC content
run: |
##Download HGNC data
wget https://raw.githubusercontent.com/sec2pri/mapping_preprocessing/refs/heads/main/datasources/hgnc/data/HGNC_secID2priID.tsv
##count unique entries in first column
echo "Primary IDs" | cut -f1 HGNC_secID2priID.tsv | sort | uniq | wc -l #for Primary IDs
##Count unique entries in third column
echo "Secondary IDs" | cut -f3 HGNC_secID2priID.tsv | sort | uniq | wc -l #for Secondary IDs
##Count total number of lines for matches between Primary and Secondary IDs
echo "Total ID matches" |wc -l HGNC_secID2priID.tsv
- name: HMDB content
run: |
##Download HMDB data
wget https://raw.githubusercontent.com/sec2pri/mapping_preprocessing/refs/heads/main/datasources/hmdb/data/HMDB_secID2priID.tsv
##count unique entries in first column
echo "Primary IDs" | cut -f1 HMDB_secID2priID.tsv | sort | uniq | wc -l #for Primary IDs
##Count unique entries in third column
echo "Secondary IDs" | cut -f2 HMDB_secID2priID.tsv | sort | uniq | wc -l #for Secondary IDs
##Count total number of lines for matches between Primary and Secondary IDs
echo "Total ID matches" |wc -l HMDB_secID2priID.tsv
- name: Uniprot content
run: |
##Install program to unzip
sudo apt install unzip
##Download UniProt data
wget https://github.com/sec2pri/mapping_preprocessing/raw/refs/heads/main/datasources/uniprot/data/UniProt_secID2priID.zip
##Unzip the file
unzip UniProt_secID2priID.zip
##count unique entries in first column
echo "Primary IDs" | cut -f1 UniProt_secID2priID.tsv | sort | uniq | wc -l #for Primary IDs
##Count unique entries in third column
echo "Secondary IDs" | cut -f2 UniProt_secID2priID.tsv | sort | uniq | wc -l #for Secondary IDs
##Count total number of lines for matches between Primary and Secondary IDs
echo "Total ID matches" |wc -l UniProt_secID2priID.tsv