contentStats #34
This file contains hidden or bidirectional Unicode text that may be interpreted or compiled differently than what appears below. To review, open the file in an editor that reveals hidden Unicode characters.
Learn more about bidirectional Unicode characters
| # Workflow for checking the content of the files regarding primary and secondary IDs | |
| name: contentStats | |
| on: | |
| workflow_dispatch: | |
| pull_request: # tests whether it is working on PR | |
| paths: | |
| - '.github/workflows/contentStats.yml' | |
| schedule: | |
| - cron: "0 0 1,15 * *" # Run the workflow on the 1st and 15th day of each month | |
| jobs: | |
| contentStats: | |
| runs-on: ubuntu-latest | |
| permissions: | |
| contents: write | |
| # step 1: checkout the repository | |
| steps: | |
| - name: Download GitHub repo for the queries | |
| uses: actions/checkout@v5 | |
| # step 2: run the bash commands to count the content for different databases | |
| - name: ChEBI content | |
| run: | | |
| ##Download ChEBI data | |
| wget https://raw.githubusercontent.com/sec2pri/mapping_preprocessing/refs/heads/main/datasources/chebi/data/ChEBI_secID2priID.tsv | |
| ##count unique entries in first column | |
| count=$(cut -f1 ChEBI_secID2priID.tsv | sort | uniq | wc -l) #for Primary IDs | |
| echo "Count Primary IDs: $count" | |
| ##Count unique entries in second column | |
| count=$(cut -f2 ChEBI_secID2priID.tsv | sort | uniq | wc -l) #for Secondary IDs | |
| echo "Count Secondary IDs: $count" | |
| ##Count total number of lines for matches between Primary and Secondary IDs | |
| echo "Total ID matches" |wc -l ChEBI_secID2priID.tsv | |
| - name: HGNC content | |
| run: | | |
| ##Download HGNC data | |
| wget https://raw.githubusercontent.com/sec2pri/mapping_preprocessing/refs/heads/main/datasources/hgnc/data/HGNC_secID2priID.tsv | |
| ##count unique entries in first column | |
| echo "Primary IDs" | cut -f1 HGNC_secID2priID.tsv | sort | uniq | wc -l #for Primary IDs | |
| ##Count unique entries in third column | |
| echo "Secondary IDs" | cut -f3 HGNC_secID2priID.tsv | sort | uniq | wc -l #for Secondary IDs | |
| ##Count total number of lines for matches between Primary and Secondary IDs | |
| echo "Total ID matches" |wc -l HGNC_secID2priID.tsv | |
| - name: HMDB content | |
| run: | | |
| ##Download HMDB data | |
| wget https://raw.githubusercontent.com/sec2pri/mapping_preprocessing/refs/heads/main/datasources/hmdb/data/HMDB_secID2priID.tsv | |
| ##count unique entries in first column | |
| echo "Primary IDs" | cut -f1 HMDB_secID2priID.tsv | sort | uniq | wc -l #for Primary IDs | |
| ##Count unique entries in third column | |
| echo "Secondary IDs" | cut -f2 HMDB_secID2priID.tsv | sort | uniq | wc -l #for Secondary IDs | |
| ##Count total number of lines for matches between Primary and Secondary IDs | |
| echo "Total ID matches" |wc -l HMDB_secID2priID.tsv | |
| - name: Uniprot content | |
| run: | | |
| ##Install program to unzip | |
| sudo apt install unzip | |
| ##Download UniProt data | |
| wget https://github.com/sec2pri/mapping_preprocessing/raw/refs/heads/main/datasources/uniprot/data/UniProt_secID2priID.zip | |
| ##Unzip the file | |
| unzip UniProt_secID2priID.zip | |
| ##count unique entries in first column | |
| echo "Primary IDs" | cut -f1 UniProt_secID2priID.tsv | sort | uniq | wc -l #for Primary IDs | |
| ##Count unique entries in third column | |
| echo "Secondary IDs" | cut -f2 UniProt_secID2priID.tsv | sort | uniq | wc -l #for Secondary IDs | |
| ##Count total number of lines for matches between Primary and Secondary IDs | |
| echo "Total ID matches" |wc -l UniProt_secID2priID.tsv |