diff --git a/config/settings.py b/config/settings.py index e32fd0e..4ecfd18 100644 --- a/config/settings.py +++ b/config/settings.py @@ -142,6 +142,7 @@ 'django.template.context_processors.request', 'django.contrib.auth.context_processors.auth', 'django.contrib.messages.context_processors.messages', + 'rest_api.context_processors.op_settings', ], }, }, diff --git a/exports/config_template.py b/exports/config_template.py index 4a93a8c..a4b21ad 100644 --- a/exports/config_template.py +++ b/exports/config_template.py @@ -36,7 +36,7 @@ # Scoring files exports - from text files scoring_file_from_file_config = { - 'pmid': '', # e.g. 35501419 + 'publication_id': '', # e.g. OPP000001 'input_dir_root': '', 'output_dir_root': '' } \ No newline at end of file diff --git a/exports/metadata_build_export.py b/exports/metadata_build_export.py index f20a43d..4bb50fb 100644 --- a/exports/metadata_build_export.py +++ b/exports/metadata_build_export.py @@ -13,7 +13,9 @@ class MetadataExport: def __init__(self, exports_dir:str, sqlite_dir:str, dataset:Dataset): self.dataset = dataset self.dataset_id = dataset.id - self.sqlite_dir = f'{sqlite_dir}/{self.dataset_id}' + self.sqlite_dir = None + if sqlite_dir: + self.sqlite_dir = f'{sqlite_dir}/{self.dataset_id}' self.data = { 'dataset': self.get_data_attr(self.dataset,'Dataset'), 'publication': self.get_data_attr(self.dataset.publication,'Publication'), @@ -26,15 +28,18 @@ def __init__(self, exports_dir:str, sqlite_dir:str, dataset:Dataset): # Find corresponding SQLite file self.sqlite_file = None - for s_file in os.listdir(self.sqlite_dir): - if s_file.startswith(dataset.id) and s_file.endswith('.db'): - self.sqlite_file = (s_file) - break - if not self.sqlite_file: - print(f"ERROR: Can't find a SQLite file for the dataset {self.dataset_id} in {self.sqlite_dir}") - exit() - - filename = self.sqlite_file.replace('.db','_metadata.xlsx') + if self.sqlite_dir: + for s_file in os.listdir(self.sqlite_dir): + if s_file.startswith(dataset.id) and s_file.endswith('.db'): + self.sqlite_file = (s_file) + break + if not self.sqlite_file: + print(f"ERROR: Can't find a SQLite file for the dataset {self.dataset_id} in {self.sqlite_dir}") + exit() + if self.sqlite_file: + filename = self.sqlite_file.replace('.db','_metadata.xlsx') + else: + filename = f'{self.dataset_id}_metadata.xlsx' self.filepath = f'{exports_dir}/{filename}' @@ -192,7 +197,8 @@ def generate_metadata(self): self.data['publication'] = self.get_data_attr(self.dataset.publication,'Publication') # Fetch phi values from the SQLite export - self.fetch_phi_values_from_sqlite() + if self.sqlite_dir: + self.fetch_phi_values_from_sqlite() # Add file urls to the dataset self.add_dataset_file_urls() diff --git a/exports/scripts/generate_genetic_score_files_from_files.py b/exports/scripts/generate_genetic_score_files_from_files.py index 95ecbe9..0f3cefb 100644 --- a/exports/scripts/generate_genetic_score_files_from_files.py +++ b/exports/scripts/generate_genetic_score_files_from_files.py @@ -20,18 +20,50 @@ def generate_scoring_file_header(score:Score, dataset:Dataset): reported_trait = score.trait_reported else: reported_trait = score.trait_reported_id - return f'''#omicspred_id={score.id} + + trait_type = dataset.platform.platform_master.type + + # Setup 'trait_mapped' + molecular_traits = [] + trait_mapped = set() + match trait_type: + case 'Proteomics': + molecular_traits = score.proteins.all() + case 'Transcriptomics': + molecular_traits = score.genes.all() + case 'Metabolomics': + molecular_traits = score.metabolites.all() + case _: + print("Can't find a valid platform type.") # Default case + exit() + for mt in molecular_traits: + if mt.name and mt.external_id: + trait_mapped.add(f'{mt.name} ({mt.external_id})') + elif mt.name: + trait_mapped.add(mt.name) + elif mt.external_id: + trait_mapped.add(mt.external_id) + + platform_version = f' ({dataset.platform.version})' if dataset.platform.version else '' + return f'''###GENETIC SCORING FILE - see https://www.pgscatalog.org/downloads/#dl_ftp_scoring for additional information +#format_version=2.0 +##GENETIC SCORE (OPGS) INFORMATION +#omicspred_id={score.id} #pgs_name={score.name} -#trait_type=proteomics +#trait_type={trait_type} #measurement_tissue={dataset.tissue.label} ({dataset.tissue.id}) -#measurement_platform=Somalogic ({dataset.platform.version}) +#measurement_platform={dataset.platform.name}{platform_version} +#trait_mapped={'|'.join(sorted(trait_mapped))} #trait_reported={reported_trait} #genome_build={score.variants_genomebuild} #variants_number={score.variants_number} +##SOURCE INFORMATION +#pgp_id={dataset.publication.id} #citation={publication_model.firstauthor} et al. {publication_model.journal} ({publication_model.pub_year}). doi:{publication_model.doi} #license={score.license}''' + def write_scoring_file(header:str, content:str, filepath:str) -> None: ''' Write new scoring file with header ''' with open(filepath, 'w') as output_file: @@ -56,7 +88,7 @@ def get_dataset_label(id:str, name:str) -> str: def run(): - pmid = scoring_file_from_file_config['pmid'] + publication_id = scoring_file_from_file_config['publication_id'] input_dir_root = scoring_file_from_file_config['input_dir_root'] output_dir_root = scoring_file_from_file_config['output_dir_root'] @@ -69,7 +101,7 @@ def run(): print(f'-> input_dir_name > {scores_dir}: {scores_dir_path}') input_dirs[scores_dir] = scores_dir_path - datasets = Dataset.objects.filter(publication__pmid=pmid) + datasets = Dataset.objects.filter(publication__id=publication_id) for dataset in datasets: print(f"# {dataset.name} ({dataset.id}) - {dataset.num}") diff --git a/imports/README.md b/imports/README.md index 3e73890..d687838 100644 --- a/imports/README.md +++ b/imports/README.md @@ -17,3 +17,4 @@ python manage.py runscript import_metadata 4. update_ensembl_proteins.py 5. download_reactome_data.py (optional - only for updated dataset from Reactome) 6. get_reactome_mappings.py +7. update_db_info_count.py diff --git a/imports/omicspred/models/performance.py b/imports/omicspred/models/performance.py index ae68c7d..0db88d5 100644 --- a/imports/omicspred/models/performance.py +++ b/imports/omicspred/models/performance.py @@ -82,7 +82,14 @@ def get_cohort_label(self, sample_model:Sample) -> str: def update_eval_type(self): + eval_type_mapping = { + 'Variant associations': 'Training', + 'Score development': 'Training', + 'Validation': 'External Validation' + } eval_type = self.data['eval_type'] + if eval_type in eval_type_mapping.keys(): + eval_type = eval_type_mapping[eval_type] eval_type_choices = Performance.eval_type.field.choices eval_types = {} for choice in eval_type_choices: diff --git a/imports/omicspred/models/publication.py b/imports/omicspred/models/publication.py index 8c5c60a..21ea791 100644 --- a/imports/omicspred/models/publication.py +++ b/imports/omicspred/models/publication.py @@ -6,26 +6,34 @@ class PublicationData(GenericData): - def __init__(self,pmid): + def __init__(self,pmid:int,data:dict): GenericData.__init__(self) - self.pmid = int(pmid) - self.check_model_exist() - if not self.model: - self.fetch_publication_information() + if pmid and str(pmid).isdigit(): + self.pmid = int(pmid) + self.check_model_exist() + if not self.model: + self.fetch_publication_information() + else: + self.pmid = None + self.data = data + self.firstauthor = data['firstauthor'] def check_model_exist(self): ''' Check if a Publication model already exists. ''' - try: - publication = Publication.objects.get(pmid=self.pmid) - self.model = publication - except Publication.DoesNotExist: + if self.pmid: + try: + publication = Publication.objects.get(pmid=self.pmid) + self.model = publication + except Publication.DoesNotExist: + self.model = None + else: self.model = None - def rest_api_call_to_epmc(self,query): + def rest_api_call_to_epmc(self,query:str): ''' REST API call to EuropePMC - query: the search query diff --git a/imports/omicspred/models/sample.py b/imports/omicspred/models/sample.py index 02cfad8..8b7650c 100644 --- a/imports/omicspred/models/sample.py +++ b/imports/omicspred/models/sample.py @@ -69,7 +69,7 @@ def create_model(self): elif field == 'sample_percent_male': # Remove % character val_str = str(val) - if re.search('\%',val_str): + if re.search('\\%',val_str): val_str = re.sub(r'\%', r'', val_str) val_str = re.sub(r' ', r'', val_str) val = float(val_str) diff --git a/imports/omicspred/spreadsheets/spreadsheet.py b/imports/omicspred/spreadsheets/spreadsheet.py index af3cf70..c003612 100644 --- a/imports/omicspred/spreadsheets/spreadsheet.py +++ b/imports/omicspred/spreadsheets/spreadsheet.py @@ -136,17 +136,20 @@ def extract_data(self): model = 'Publication' logger.info(f"Start to parse the {model} spreadsheet") pmid = None + data = {} pub_info = self.dataframe.iloc[0] for col in pub_info.keys(): # m, f = self.get_model_field_from_schema(col,self.spreadsheet_schema) m, f = self.spreadsheet_schema.loc[col][:2] if f == 'pmid': pmid = pub_info[col] - break - if pmid: - pub_data = PublicationData(pmid) + # break + elif f in ['doi','journal','date_publication','firstauthor']: + data[f] = pub_info[col] + pub_data = PublicationData(pmid,data) + if pmid and str(pmid).isdigit(): pub_data.fetch_publication_information() - self.parsed_data[pmid] = pub_data + self.parsed_data[pmid] = pub_data @@ -286,7 +289,8 @@ def extract_data(self): dataset_name_suffix = score_components[1] ################################################# - dataset_tag = f'{platform.name}_{platform_version}_{self.publication.pmid}_{dataset_tag_suffix}' + publication_tag = self.publication.pmid if self.publication.pmid else self.publication.firstauthor + dataset_tag = f'{platform.name}_{platform_version}_{publication_tag}_{dataset_tag_suffix}' if dataset_tag in self.datasets.keys(): dataset = self.datasets[dataset_tag] dataset.add_score() diff --git a/imports/scripts/update_db_info_count.py b/imports/scripts/update_db_info_count.py new file mode 100644 index 0000000..eab604b --- /dev/null +++ b/imports/scripts/update_db_info_count.py @@ -0,0 +1,33 @@ +from omicspred.models import * + +def run(): + current_info = {} + for info in Info.objects.all(): + info_name = info.name + current_info[info_name] = info + + # Datasets + new_info = { + 'datasets': Dataset.objects.count(), + 'scores': Score.objects.count(), + 'publications': Publication.objects.count(), + 'platforms': PlatformMaster.objects.count(), + 'pathways': Pathway.objects.count(), + 'phenotypes': Phenotype.objects.count(), + 'phewas': ScorePheWAS.objects.count(), + 'tissues': Tissue.objects.filter(type='tissue').count() + } + + for info_name in new_info.keys(): + if info_name in current_info.keys(): + current_info_model = current_info[info_name] + current_info_model.value = new_info[info_name] + current_info_model.save() + print(f"- Update '{info_name}' count ({new_info[info_name]})") + else: + info_model = Info( + name = info_name, + value = new_info[info_name] + ) + info_model.save() + print(f"- Create '{info_name}' count ({new_info[info_name]})") \ No newline at end of file diff --git a/omicspred/migrations/0001_initial.py b/omicspred/migrations/0001_initial.py index 09fda48..7d19e18 100644 --- a/omicspred/migrations/0001_initial.py +++ b/omicspred/migrations/0001_initial.py @@ -1,4 +1,4 @@ -# Generated by Django 6.0.6 on 2026-07-03 13:41 +# Generated by Django 6.0.8 on 2026-09-07 08:48 import django.contrib.postgres.fields.ranges import django.core.validators @@ -53,6 +53,14 @@ class Migration(migrations.Migration): ('url', models.CharField(max_length=100, verbose_name='External Source URL')), ], ), + migrations.CreateModel( + name='Info', + fields=[ + ('id', models.BigAutoField(auto_created=True, primary_key=True, serialize=False, verbose_name='ID')), + ('name', models.CharField(max_length=50, verbose_name='Info Name')), + ('value', models.IntegerField(verbose_name='Info Value')), + ], + ), migrations.CreateModel( name='Pathway', fields=[ @@ -205,7 +213,7 @@ class Migration(migrations.Migration): name='Performance', fields=[ ('id', models.BigAutoField(auto_created=True, primary_key=True, serialize=False, verbose_name='ID')), - ('eval_type', models.CharField(choices=[('T', 'Training'), ('IV', 'Independent Validation'), ('EV', 'External Validation'), ('E', 'Evaluation')], default='', max_length=25, verbose_name='Evaluation Type')), + ('eval_type', models.CharField(choices=[('T', 'Training'), ('IDV', 'Independent Validation'), ('IV', 'Internal Validation'), ('EV', 'External Validation'), ('E', 'Evaluation')], default='', max_length=25, verbose_name='Evaluation Type')), ('performance_additional', models.TextField(default='', verbose_name='Additional Information')), ('source_gwas_catalog', models.CharField(max_length=20, null=True, verbose_name='GWAS Catalog Study ID (GCST...)')), ('source_doi', models.CharField(max_length=100, null=True, verbose_name='Source DOI')), diff --git a/omicspred/models.py b/omicspred/models.py index 9c8e3ed..37d3603 100644 --- a/omicspred/models.py +++ b/omicspred/models.py @@ -607,7 +607,8 @@ class Performance(models.Model): # Evaluation Type EVALUATION_CHOICES = [ ('T', 'Training'), - ('IV', 'Independent Validation'), + ('IDV', 'Independent Validation'), + ('IV', 'Internal Validation'), ('EV', 'External Validation'), ('E', 'Evaluation') ] @@ -810,4 +811,10 @@ class ExternalSource(models.Model): """ Class to hold ExternalSource values """ name = models.CharField('External Source Name', max_length=50) version = models.CharField('External Source Version', max_length=50, null=True) - url = models.CharField('External Source URL', max_length=100) \ No newline at end of file + url = models.CharField('External Source URL', max_length=100) + + +class Info(models.Model): + """ Class to hold overall information counts/values """ + name = models.CharField('Info Name', max_length=50) + value = models.IntegerField(verbose_name='Info Value') \ No newline at end of file diff --git a/requirements.txt b/requirements.txt index c5cf57a..38b0ea0 100644 --- a/requirements.txt +++ b/requirements.txt @@ -1,5 +1,5 @@ #### Django library #### -Django==6.0.7 +Django==6.0.8 #### Python PostgreSQL library #### # psycopg2-binary==2.9.11 psycopg[binary]==3.3.4 diff --git a/rest_api/context_processors.py b/rest_api/context_processors.py new file mode 100644 index 0000000..53da405 --- /dev/null +++ b/rest_api/context_processors.py @@ -0,0 +1,7 @@ +from django.conf import settings + + +def op_settings(request): + return { + 'is_public_site' : settings.PUBLIC_SITE + } \ No newline at end of file diff --git a/rest_api/fixtures/db_test.default.json b/rest_api/fixtures/db_test.default.json index 2bbf967..78251d8 100644 --- a/rest_api/fixtures/db_test.default.json +++ b/rest_api/fixtures/db_test.default.json @@ -1044,5 +1044,69 @@ "variants_number_used": 19, "variants_fraction_found": 1 } + }, + { + "model": "omicspred.Info", + "pk": 1, + "fields": { + "name": "datasets", + "value": 216 + } + }, + { + "model": "omicspred.Info", + "pk": 2, + "fields": { + "name": "scores", + "value": 3339610 + } + }, + { + "model": "omicspred.Info", + "pk": 3, + "fields": { + "name": "publications", + "value": 7 + } + }, + { + "model": "omicspred.Info", + "pk": 4, + "fields": { + "name": "platforms", + "value": 6 + } + }, + { + "model": "omicspred.Info", + "pk": 5, + "fields": { + "name": "pathways", + "value": 2748 + } + }, + { + "model": "omicspred.Info", + "pk": 6, + "fields": { + "name": "phenotypes", + "value": 927 + } + }, + { + "model": "omicspred.Info", + "pk": 7, + "fields": { + "name": "phewas", + "value": 369024 + } + }, + { + "model": "omicspred.Info", + "pk": 8, + "fields": { + "name": "tissues", + "value": 53 + } } ] \ No newline at end of file diff --git a/rest_api/static/rest_api/images/favicon.ico b/rest_api/static/rest_api/images/favicon.ico index 3927d37..3c7bff9 100644 Binary files a/rest_api/static/rest_api/images/favicon.ico and b/rest_api/static/rest_api/images/favicon.ico differ diff --git a/rest_api/templates/robots.txt b/rest_api/templates/robots.txt index 5b96075..65e5aa4 100644 --- a/rest_api/templates/robots.txt +++ b/rest_api/templates/robots.txt @@ -1,3 +1,5 @@ User-agent: * +{% if is_public_site == 'True' %} Allow: /$ +{% endif %} Disallow: / \ No newline at end of file diff --git a/rest_api/views.py b/rest_api/views.py index ae6d4a6..2d566e6 100644 --- a/rest_api/views.py +++ b/rest_api/views.py @@ -1819,21 +1819,14 @@ class RestInfo(generics.RetrieveAPIView): """ def get(self, request): - + data_count = {} + for info in Info.objects.all().order_by('id'): + data_count[info.name] = info.value data = { 'rest_api': { "version": settings.REST_API_VERSION }, - 'data_count': { - 'datasets': Dataset.objects.count(), - 'scores': Score.objects.count(), - 'publications': Publication.objects.count(), - 'platforms': PlatformMaster.objects.count(), - 'pathways': Pathway.objects.count(), - 'phenotypes': Phenotype.objects.count(), - 'phewas': ScorePheWAS.objects.count(), - 'tissues': Tissue.objects.filter(type='tissue').count() - } + 'data_count': data_count } return Response(data)