Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
1 change: 1 addition & 0 deletions config/settings.py
Original file line number Diff line number Diff line change
Expand Up @@ -142,6 +142,7 @@
'django.template.context_processors.request',
'django.contrib.auth.context_processors.auth',
'django.contrib.messages.context_processors.messages',
'rest_api.context_processors.op_settings',
],
},
},
Expand Down
2 changes: 1 addition & 1 deletion exports/config_template.py
Original file line number Diff line number Diff line change
Expand Up @@ -36,7 +36,7 @@

# Scoring files exports - from text files
scoring_file_from_file_config = {
'pmid': '<pubmed ID>', # e.g. 35501419
'publication_id': '<OPP ID>', # e.g. OPP000001
'input_dir_root': '<path_to_dir_containing_raw_scoring_files>',
'output_dir_root': '<path_to_omicspred_scoring_files_output_dir>'
}
28 changes: 17 additions & 11 deletions exports/metadata_build_export.py
Original file line number Diff line number Diff line change
Expand Up @@ -13,7 +13,9 @@ class MetadataExport:
def __init__(self, exports_dir:str, sqlite_dir:str, dataset:Dataset):
self.dataset = dataset
self.dataset_id = dataset.id
self.sqlite_dir = f'{sqlite_dir}/{self.dataset_id}'
self.sqlite_dir = None
if sqlite_dir:
self.sqlite_dir = f'{sqlite_dir}/{self.dataset_id}'
self.data = {
'dataset': self.get_data_attr(self.dataset,'Dataset'),
'publication': self.get_data_attr(self.dataset.publication,'Publication'),
Expand All @@ -26,15 +28,18 @@ def __init__(self, exports_dir:str, sqlite_dir:str, dataset:Dataset):

# Find corresponding SQLite file
self.sqlite_file = None
for s_file in os.listdir(self.sqlite_dir):
if s_file.startswith(dataset.id) and s_file.endswith('.db'):
self.sqlite_file = (s_file)
break
if not self.sqlite_file:
print(f"ERROR: Can't find a SQLite file for the dataset {self.dataset_id} in {self.sqlite_dir}")
exit()

filename = self.sqlite_file.replace('.db','_metadata.xlsx')
if self.sqlite_dir:
for s_file in os.listdir(self.sqlite_dir):
if s_file.startswith(dataset.id) and s_file.endswith('.db'):
self.sqlite_file = (s_file)
break
if not self.sqlite_file:
print(f"ERROR: Can't find a SQLite file for the dataset {self.dataset_id} in {self.sqlite_dir}")
exit()
if self.sqlite_file:
filename = self.sqlite_file.replace('.db','_metadata.xlsx')
else:
filename = f'{self.dataset_id}_metadata.xlsx'
self.filepath = f'{exports_dir}/{filename}'


Expand Down Expand Up @@ -192,7 +197,8 @@ def generate_metadata(self):
self.data['publication'] = self.get_data_attr(self.dataset.publication,'Publication')

# Fetch phi values from the SQLite export
self.fetch_phi_values_from_sqlite()
if self.sqlite_dir:
self.fetch_phi_values_from_sqlite()

# Add file urls to the dataset
self.add_dataset_file_urls()
Expand Down
42 changes: 37 additions & 5 deletions exports/scripts/generate_genetic_score_files_from_files.py
Original file line number Diff line number Diff line change
Expand Up @@ -20,18 +20,50 @@ def generate_scoring_file_header(score:Score, dataset:Dataset):
reported_trait = score.trait_reported
else:
reported_trait = score.trait_reported_id
return f'''#omicspred_id={score.id}

trait_type = dataset.platform.platform_master.type

# Setup 'trait_mapped'
molecular_traits = []
trait_mapped = set()
match trait_type:
case 'Proteomics':
molecular_traits = score.proteins.all()
case 'Transcriptomics':
molecular_traits = score.genes.all()
case 'Metabolomics':
molecular_traits = score.metabolites.all()
case _:
print("Can't find a valid platform type.") # Default case
exit()
for mt in molecular_traits:
if mt.name and mt.external_id:
trait_mapped.add(f'{mt.name} ({mt.external_id})')
elif mt.name:
trait_mapped.add(mt.name)
elif mt.external_id:
trait_mapped.add(mt.external_id)

platform_version = f' ({dataset.platform.version})' if dataset.platform.version else ''
return f'''###GENETIC SCORING FILE - see https://www.pgscatalog.org/downloads/#dl_ftp_scoring for additional information
#format_version=2.0
##GENETIC SCORE (OPGS) INFORMATION
#omicspred_id={score.id}
#pgs_name={score.name}
#trait_type=proteomics
#trait_type={trait_type}
#measurement_tissue={dataset.tissue.label} ({dataset.tissue.id})
#measurement_platform=Somalogic ({dataset.platform.version})
#measurement_platform={dataset.platform.name}{platform_version}
#trait_mapped={'|'.join(sorted(trait_mapped))}
#trait_reported={reported_trait}
#genome_build={score.variants_genomebuild}
#variants_number={score.variants_number}
##SOURCE INFORMATION
#pgp_id={dataset.publication.id}
#citation={publication_model.firstauthor} et al. {publication_model.journal} ({publication_model.pub_year}). doi:{publication_model.doi}
#license={score.license}'''



def write_scoring_file(header:str, content:str, filepath:str) -> None:
''' Write new scoring file with header '''
with open(filepath, 'w') as output_file:
Expand All @@ -56,7 +88,7 @@ def get_dataset_label(id:str, name:str) -> str:

def run():

pmid = scoring_file_from_file_config['pmid']
publication_id = scoring_file_from_file_config['publication_id']
input_dir_root = scoring_file_from_file_config['input_dir_root']
output_dir_root = scoring_file_from_file_config['output_dir_root']

Expand All @@ -69,7 +101,7 @@ def run():
print(f'-> input_dir_name > {scores_dir}: {scores_dir_path}')
input_dirs[scores_dir] = scores_dir_path

datasets = Dataset.objects.filter(publication__pmid=pmid)
datasets = Dataset.objects.filter(publication__id=publication_id)

for dataset in datasets:
print(f"# {dataset.name} ({dataset.id}) - {dataset.num}")
Expand Down
1 change: 1 addition & 0 deletions imports/README.md
Original file line number Diff line number Diff line change
Expand Up @@ -17,3 +17,4 @@ python manage.py runscript import_metadata
4. update_ensembl_proteins.py
5. download_reactome_data.py (optional - only for updated dataset from Reactome)
6. get_reactome_mappings.py
7. update_db_info_count.py
7 changes: 7 additions & 0 deletions imports/omicspred/models/performance.py
Original file line number Diff line number Diff line change
Expand Up @@ -82,7 +82,14 @@ def get_cohort_label(self, sample_model:Sample) -> str:


def update_eval_type(self):
eval_type_mapping = {
'Variant associations': 'Training',
'Score development': 'Training',
'Validation': 'External Validation'
}
eval_type = self.data['eval_type']
if eval_type in eval_type_mapping.keys():
eval_type = eval_type_mapping[eval_type]
eval_type_choices = Performance.eval_type.field.choices
eval_types = {}
for choice in eval_type_choices:
Expand Down
28 changes: 18 additions & 10 deletions imports/omicspred/models/publication.py
Original file line number Diff line number Diff line change
Expand Up @@ -6,26 +6,34 @@

class PublicationData(GenericData):

def __init__(self,pmid):
def __init__(self,pmid:int,data:dict):
GenericData.__init__(self)
self.pmid = int(pmid)
self.check_model_exist()
if not self.model:
self.fetch_publication_information()
if pmid and str(pmid).isdigit():
self.pmid = int(pmid)
self.check_model_exist()
if not self.model:
self.fetch_publication_information()
else:
self.pmid = None
self.data = data
self.firstauthor = data['firstauthor']


def check_model_exist(self):
'''
Check if a Publication model already exists.
'''
try:
publication = Publication.objects.get(pmid=self.pmid)
self.model = publication
except Publication.DoesNotExist:
if self.pmid:
try:
publication = Publication.objects.get(pmid=self.pmid)
self.model = publication
except Publication.DoesNotExist:
self.model = None
else:
self.model = None


def rest_api_call_to_epmc(self,query):
def rest_api_call_to_epmc(self,query:str):
'''
REST API call to EuropePMC
- query: the search query
Expand Down
2 changes: 1 addition & 1 deletion imports/omicspred/models/sample.py
Original file line number Diff line number Diff line change
Expand Up @@ -69,7 +69,7 @@ def create_model(self):
elif field == 'sample_percent_male':
# Remove % character
val_str = str(val)
if re.search('\%',val_str):
if re.search('\\%',val_str):
val_str = re.sub(r'\%', r'', val_str)
val_str = re.sub(r' ', r'', val_str)
val = float(val_str)
Expand Down
14 changes: 9 additions & 5 deletions imports/omicspred/spreadsheets/spreadsheet.py
Original file line number Diff line number Diff line change
Expand Up @@ -136,17 +136,20 @@ def extract_data(self):
model = 'Publication'
logger.info(f"Start to parse the {model} spreadsheet")
pmid = None
data = {}
pub_info = self.dataframe.iloc[0]
for col in pub_info.keys():
# m, f = self.get_model_field_from_schema(col,self.spreadsheet_schema)
m, f = self.spreadsheet_schema.loc[col][:2]
if f == 'pmid':
pmid = pub_info[col]
break
if pmid:
pub_data = PublicationData(pmid)
# break
elif f in ['doi','journal','date_publication','firstauthor']:
data[f] = pub_info[col]
pub_data = PublicationData(pmid,data)
if pmid and str(pmid).isdigit():
pub_data.fetch_publication_information()
self.parsed_data[pmid] = pub_data
self.parsed_data[pmid] = pub_data



Expand Down Expand Up @@ -286,7 +289,8 @@ def extract_data(self):
dataset_name_suffix = score_components[1]
#################################################

dataset_tag = f'{platform.name}_{platform_version}_{self.publication.pmid}_{dataset_tag_suffix}'
publication_tag = self.publication.pmid if self.publication.pmid else self.publication.firstauthor
dataset_tag = f'{platform.name}_{platform_version}_{publication_tag}_{dataset_tag_suffix}'
if dataset_tag in self.datasets.keys():
dataset = self.datasets[dataset_tag]
dataset.add_score()
Expand Down
33 changes: 33 additions & 0 deletions imports/scripts/update_db_info_count.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,33 @@
from omicspred.models import *

def run():
current_info = {}
for info in Info.objects.all():
info_name = info.name
current_info[info_name] = info

# Datasets
new_info = {
'datasets': Dataset.objects.count(),
'scores': Score.objects.count(),
'publications': Publication.objects.count(),
'platforms': PlatformMaster.objects.count(),
'pathways': Pathway.objects.count(),
'phenotypes': Phenotype.objects.count(),
'phewas': ScorePheWAS.objects.count(),
'tissues': Tissue.objects.filter(type='tissue').count()
}

for info_name in new_info.keys():
if info_name in current_info.keys():
current_info_model = current_info[info_name]
current_info_model.value = new_info[info_name]
current_info_model.save()
print(f"- Update '{info_name}' count ({new_info[info_name]})")
else:
info_model = Info(
name = info_name,
value = new_info[info_name]
)
info_model.save()
print(f"- Create '{info_name}' count ({new_info[info_name]})")
12 changes: 10 additions & 2 deletions omicspred/migrations/0001_initial.py
Original file line number Diff line number Diff line change
@@ -1,4 +1,4 @@
# Generated by Django 6.0.6 on 2026-07-03 13:41
# Generated by Django 6.0.8 on 2026-09-07 08:48

import django.contrib.postgres.fields.ranges
import django.core.validators
Expand Down Expand Up @@ -53,6 +53,14 @@ class Migration(migrations.Migration):
('url', models.CharField(max_length=100, verbose_name='External Source URL')),
],
),
migrations.CreateModel(
name='Info',
fields=[
('id', models.BigAutoField(auto_created=True, primary_key=True, serialize=False, verbose_name='ID')),
('name', models.CharField(max_length=50, verbose_name='Info Name')),
('value', models.IntegerField(verbose_name='Info Value')),
],
),
migrations.CreateModel(
name='Pathway',
fields=[
Expand Down Expand Up @@ -205,7 +213,7 @@ class Migration(migrations.Migration):
name='Performance',
fields=[
('id', models.BigAutoField(auto_created=True, primary_key=True, serialize=False, verbose_name='ID')),
('eval_type', models.CharField(choices=[('T', 'Training'), ('IV', 'Independent Validation'), ('EV', 'External Validation'), ('E', 'Evaluation')], default='', max_length=25, verbose_name='Evaluation Type')),
('eval_type', models.CharField(choices=[('T', 'Training'), ('IDV', 'Independent Validation'), ('IV', 'Internal Validation'), ('EV', 'External Validation'), ('E', 'Evaluation')], default='', max_length=25, verbose_name='Evaluation Type')),
('performance_additional', models.TextField(default='', verbose_name='Additional Information')),
('source_gwas_catalog', models.CharField(max_length=20, null=True, verbose_name='GWAS Catalog Study ID (GCST...)')),
('source_doi', models.CharField(max_length=100, null=True, verbose_name='Source DOI')),
Expand Down
11 changes: 9 additions & 2 deletions omicspred/models.py
Original file line number Diff line number Diff line change
Expand Up @@ -607,7 +607,8 @@ class Performance(models.Model):
# Evaluation Type
EVALUATION_CHOICES = [
('T', 'Training'),
('IV', 'Independent Validation'),
('IDV', 'Independent Validation'),
('IV', 'Internal Validation'),
('EV', 'External Validation'),
('E', 'Evaluation')
]
Expand Down Expand Up @@ -810,4 +811,10 @@ class ExternalSource(models.Model):
""" Class to hold ExternalSource values """
name = models.CharField('External Source Name', max_length=50)
version = models.CharField('External Source Version', max_length=50, null=True)
url = models.CharField('External Source URL', max_length=100)
url = models.CharField('External Source URL', max_length=100)


class Info(models.Model):
""" Class to hold overall information counts/values """
name = models.CharField('Info Name', max_length=50)
value = models.IntegerField(verbose_name='Info Value')
2 changes: 1 addition & 1 deletion requirements.txt
Original file line number Diff line number Diff line change
@@ -1,5 +1,5 @@
#### Django library ####
Django==6.0.7
Django==6.0.8
#### Python PostgreSQL library ####
# psycopg2-binary==2.9.11
psycopg[binary]==3.3.4
Expand Down
7 changes: 7 additions & 0 deletions rest_api/context_processors.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,7 @@
from django.conf import settings


def op_settings(request):
return {
'is_public_site' : settings.PUBLIC_SITE
}
Loading
Loading