Finding.
','volume':'2','issue':'3','page':'1-8'}]}} + r=parse_crossref(raw)[0] + assert r.metadata.year==2023 and r.abstract=='Finding.' + assert r.metadata.authors[0].family=='Tester' and not r.metadata.preprint + + +def test_europe_pmc_open_access_ids(): + raw={'resultList':{'result':[{'title':'Paper','id':'123','source':'MED','pmid':'123','pmcid':'PMC456','pubYear':'2024','abstractText':'Summary','authorList':{'author':[{'lastName':'Example','firstName':'A'}]},'journalInfo':{'journal':{'title':'Journal'}}}]}} + r=parse_europe(raw)[0] + assert r.metadata.pmcid=='PMC456' and r.metadata.content_scope=='abstract' + + +def test_duplicate_merging_preserves_screening_and_provenance(): + first=papers();first[0].decision='exclude';first[0].reason='Not eligible' + second=papers();second[0].metadata.metadata_sources=['crossref'] + result,count=merge_records(first,second) + assert len(result)==2 and count==2 + assert result[0].decision=='exclude' and result[0].reason=='Not eligible' + assert set(result[0].metadata.metadata_sources)=={'pubmed','crossref'} + + +@pytest.mark.parametrize('value,expected',[('https://doi.org/10.9999/ABC','10.9999/abc'),('doi: 10.9999/test','10.9999/test'),('not-a-doi','')]) +def test_doi_normalization(value,expected):assert doi(value)==expected + + +def test_research_dates_are_validated(): + with pytest.raises(ValueError):ResearchPlan(year_from=2025,year_to=2020) + + +def test_search_transport_uses_correct_eutils_endpoints(): + requests=[] + def handler(request): + requests.append(request) + if request.url.path.endswith('esearch.fcgi'):return httpx.Response(200,json={'esearchresult':{'count':'4','idlist':['123']}}) + return httpx.Response(200,content=PUBMED) + c=ScholarlyClient(transport=httpx.MockTransport(handler)) + records,log=c.search('pubmed','test query',ResearchPlan(per_database=1,year_from=2020)) + assert len(records)==1 and log['truncated'] and log['total_hits']==4 + assert 'Date - Publication' in log['query'] + assert len(requests)==2 + + +def test_partial_database_failures_are_logged_not_empty_success(tmp_path): + class Partial(FakeScholar): + def search(self,db,query,plan): + if db=='semantic_scholar':raise ValueError('Semantic Scholar returned HTTP 429') + return super().search(db,query,plan) + c=ScienceClients();c.scholarly_factory=Partial + p=Project(mode='science');p.research.plan=ResearchPlan(question='Memory',databases=['pubmed','semantic_scholar']) + search_literature(p,c,None,Job(p.id,'science-search')) + assert len(p.research.records)==2 + assert [x['status'] for x in p.research.searches]==['ok','failed'] + assert '429' in p.research.searches[-1]['error'] + + +def prepared(): + p=Project(mode='science',title='A test manuscript') + p.research.records=papers() + for r in p.research.records:r.decision='include';record_source(p,r) + p.draft='# A test manuscript\n\n## Introduction\n\nThe result was uncertain. [S1] [S2]' + p.research.abstract='An abstract of this synthetic workflow.' + p.research.plan=ResearchPlan(question='Test',author_names='Alex Example',affiliation='Test Institute',funding='None',conflicts='None',data_availability='Synthetic material',ethics='Not applicable') + p.research.searches=[{'database':'pubmed','query':'test','status':'ok','truncated':False}] + p.review.draft_hash=digest(p.draft) + return p + + +def test_apa_author_counts_and_initials(): + authors=[ResearchAuthor(family=f'Tester{i}',given='Alex Bea') for i in range(21)] + value=authors_reference(authors) + assert 'Tester18' in value and 'Tester19' not in value and 'Tester20' in value and '. . .' in value + assert authors_reference(authors[:2])=='Tester0, A. B., & Tester1, A. B.' + + +def test_apa_same_year_disambiguation_and_adjacent_citations(): + p=prepared();p.sources[1].scholarly.year=2024 + mapping=citation_map(p) + assert {mapping['S1']['year'],mapping['S2']['year']}=={'2024a','2024b'} + value=inline_citations(p,'Evidence [S2] [S1].') + assert value=='Evidence (Example & Researcher, 2024a, 2024b).' + assert 'UNRESOLVED' in inline_citations(p,'[S99]') + + +def test_apa_word_dimensions_styles_header_and_references(): + from docx import Document + from docx.shared import Inches + p=prepared();doc=Document(io.BytesIO(docx_bytes(p))) + assert doc.sections[0].page_width==Inches(8.5) + assert doc.sections[0].top_margin==Inches(1) + assert doc.styles['Normal'].paragraph_format.line_spacing==2 + assert doc.styles['Normal'].font.name=='Times New Roman' + assert 'PAGE' in doc.sections[0].header._element.xml + paragraphs=[x for x in doc.paragraphs if 'https://doi.org' in x.text] + assert len(paragraphs)==2 and paragraphs[0].paragraph_format.first_line_indent==Inches(-.5) + assert any(r.italic and 'Test Fixture Journal' in r.text for r in paragraphs[0].runs) + assert '[S1]' not in '\n'.join(x.text for x in doc.paragraphs) + + +def test_submission_is_blocked_until_current_author_checks(): + p=prepared() + with pytest.raises(ValueError):export_science(p,'submission') + p.research.acknowledgements={k:True for k in readiness(p)['author_checks']} + p.research.confirmation_hash=digest(research_fingerprint(p)+p.draft+p.research.abstract) + assert readiness(p)['submission_allowed'] + data,_,_=export_science(p,'submission') + with zipfile.ZipFile(io.BytesIO(data)) as z:assert {'manuscript.docx','references.bib','search-log.json','evidence.csv'}<=set(z.namelist()) + p.draft+=' Changed.' + assert not readiness(p)['submission_allowed'] + + +def test_systematic_review_cannot_hide_capped_search(): + p=prepared();p.research.plan.article_type='systematic_review';p.research.searches[0]['truncated']=True + assert any('incomplete' in b.lower() for b in readiness(p)['blockers']) + + +def test_csv_formula_safety(): + p=prepared();p.research.records[0].metadata.title='=1+1' + assert "'=1+1" in evidence_csv(p) + + +def wait_job(client,jid): + for _ in range(250): + j=client.get('/api/jobs/'+jid).json() + if j['state'] not in {'queued','running'}: + assert j['state']=='completed',j + return j + time.sleep(.02) + raise AssertionError('Job did not finish') + + +def test_science_project_full_workflow_and_export(client,app): + app.state.runner.clients_factory=ScienceClients + app.state.store.set_settings(Settings(model='test',allow_cloud=True,refine=False).model_dump()) + p=client.post('/api/projects',json={'title':'Science test','mode':'science'}).json();base='/api/projects/'+p['id'] + assert p['mode']=='science' and p['brief']['format']=='APA scientific manuscript' + p=client.put(base+'/research/plan',json={'version':p['version'],'plan':{'question':'Memory and sleep','databases':['pubmed']}}).json() + def run(action):wait_job(client,client.post(base+'/jobs',json={'action':action}).json()['id']) + run('science-search');p=client.get(base).json();assert len(p['research']['records'])==2 + p=client.post(base+'/research/screen',json={'version':p['version'],'ids':[r['id'] for r in p['research']['records']],'decision':'include'}).json() + run('science-appraise');run('science-outline');run('science-draft') + p=client.get(base).json();assert p['research']['abstract'] and '[S1]' in p['draft'] + assert p['research']['records'][0]['appraisal']['claims'][0]['exact_match'] + preview=client.get(base+'/research/preview').json()['markdown'] + assert '(Example & Researcher, 2024,' in preview and '## References' in preview + for kind in ['docx','md','bib','ris','evidence','search-log','research-package']: + response=client.get(base+'/export/'+kind);assert response.status_code==200,(kind,response.text[:400]) + assert client.get(base+'/export/submission').status_code==400 + + +def test_screening_and_project_boundaries(client): + p=client.post('/api/projects',json={'mode':'science'}).json() + result=client.put('/api/projects/'+p['id']+'/research/plan',json={'version':99,'plan':{}}) + assert result.status_code==409 + n=client.post('/api/projects',json={'mode':'nonfiction'}).json() + assert client.put('/api/projects/'+n['id']+'/research/plan',json={'version':0,'plan':{}}).status_code==400 diff --git a/tests/test_science_integrity.py b/tests/test_science_integrity.py new file mode 100644 index 0000000..4e06ddc --- /dev/null +++ b/tests/test_science_integrity.py @@ -0,0 +1,58 @@ +import io +from docx import Document +from graphpaper.science_export import docx_bytes +from graphpaper.apa import inline_citations,reference_parts,citation_map +from graphpaper.science import require_evidence,record_source +from graphpaper.models import Project,Source +from .test_scholarly_research import prepared +from .science_fixtures import papers +import pytest + + +def test_apa_reference_follows_clause_and_precedes_period(): + p=prepared() + assert inline_citations(p,'The result was uncertain. [S1] [S2]')=='The result was uncertain (Example & Researcher, 2024, 2025).' + + +def test_major_sections_use_page_break_before_not_empty_break_paragraphs(): + document=Document(io.BytesIO(docx_bytes(prepared()))) + assert sum(bool(p.paragraph_format.page_break_before) for p in document.paragraphs)==3 + assert 'Opening your writing studio…
Choose independently for the writer, editor and extractor. Provider default sends no override. Load models to see their advertised levels; Codex may offer xhigh, max or ultra depending on the model and account.
Search does not need a writing-model key. Semantic Scholar may require its own key or throttle anonymous traffic. PubMed accepts an optional NCBI key. Failures are recorded, never presented as zero results.
Search, screen, appraise and write. Keep the source behind every claim.
${esc(plan.article_type.replaceAll('_',' '))} · ${plan.databases.map(d=>scienceNames[d]).join(' / ')} · up to ${plan.per_database} records per query per database
${esc(authors)}${m.authors.length>4?' et al.':''} · ${m.year||'Year not supplied'} · ${esc(m.journal||m.publication_type||'Venue not supplied')}
${esc(record.abstract.slice(0,420))}${record.abstract.length>420?'…':''}
${record.reason?`Decision: ${esc(record.reason)}
`:''}${record.fulltext_error?`${esc(log.query)}${esc(log.searched_at)} · ${log.retrieved||0} retrieved / ${log.total_hits??'unknown'} total hits · ${log.duplicates_merged||0} duplicates merged${log.truncated?' · RESULT CAP REACHED':''}
${log.error?`${esc(log.error)}
`:''}Planned searches are not recorded here until executed.
'}| Study | Design / sample | Findings | Limits / opposing evidence | Access |
|---|---|---|---|---|
| ${esc(r.appraisal.design||'Not appraised')} ${esc(r.appraisal.sample||'')} | ${esc(r.appraisal.findings||'Not established')} | ${esc(Array.isArray(r.appraisal.limitations)?r.appraisal.limitations.join('; '):r.appraisal.limitations||'')} ${esc(r.appraisal.contrary_evidence||'')} | ${esc(r.metadata.content_scope.replaceAll('_',' '))} ${esc(r.appraisal.coverage||'')} |
Include papers and run appraisal to build the evidence matrix.
':''}An override is used instead of the general query for that database. PubMed supports MeSH/field syntax; arXiv supports all:, ti:, abs: and cat:.
${esc(Array.isArray(a[k])?a[k].join('; '):a[k])}
Include this paper and run evidence appraisal. Reading an abstract does not certify study quality.
'}${(a.claims||[]).map(c=>`${esc(c.claim)}
${esc(c.quote)}${c.exact_match?'Exact source match':'Unverified quote'}
Verify compound surnames, group authors and name order. Databases supplying only a full name may need correction.
${esc(x)}
`).join('')||'No automated blockers found.
'}${esc(x)}
`).join('')||'No additional automated warnings.
'}Internal [S#] links become APA author–year citations in exports. References are generated from the checked database metadata.
${scienceUI.selected.size} selected records.
`,btn('Save screening decision','science-save-screen','check','primary'));return;}if(a==='science-save-screen'){await flush();state.p=await api('/projects/'+state.p.id+'/research/screen',{version:state.p.version,ids:[...scienceUI.selected],decision:scienceUI.decision,reason:$('#science-reason').value});scienceUI.selected.clear();closeModal();syncSummary();render();return;}if(a==='science-save-reference'){await flush();const r=state.p.research.records.find(x=>x.id===scienceUI.editRecord),meta={...r.metadata};for(const key of ['title','journal','volume','issue','pages','doi','url','reference_override'])meta[key]=$('#ref-'+key).value.trim();meta.year=$('#ref-year').value?Number($('#ref-year').value):null;meta.authors=$('#ref-authors').value.split('\n').filter(x=>x.trim()).map(line=>{const [family='',given='',literal='']=line.split('|').map(x=>x.trim());return {family,given,literal};});state.p=await api('/projects/'+state.p.id+'/research/records/'+r.id,{version:state.p.version,metadata:meta},'PATCH');closeModal();render();return;}if(a==='science-save-attach'){await flush();state.p=await api('/projects/'+state.p.id+'/research/records/'+scienceUI.editRecord+'/attach',{version:state.p.version,source_id:$('#paper-source').value,confirm_identity:$('#paper-identity').checked,full_text:$('#paper-fulltext').checked});closeModal();render();return;}if(a==='science-confirm'){await flush();const checks=Object.fromEntries($$('.science-confirm').map(x=>[x.value,x.checked]));state.p=await api('/projects/'+state.p.id+'/research/confirm',{version:state.p.version,checks});closeModal();render();toast('Author confirmations recorded for this manuscript version.');return;}}; +document.addEventListener('change',event=>{const el=event.target;if(el.classList.contains('science-paper-select')){if(el.checked)scienceUI.selected.add(el.value);else scienceUI.selected.delete(el.value);}if(el.id==='s-provider'||['s-model','s-editor_model','s-extraction_model'].includes(el.id))refreshEfforts();if(el.id==='project-mode'&&el.value==='science')$('#project-words').value=4000;}); +document.addEventListener('input',event=>{if(['s-model','s-editor_model','s-extraction_model'].includes(event.target.id))refreshEfforts();}); +if(state.p||state.projects.length)render();