-
Notifications
You must be signed in to change notification settings - Fork 0
Expand file tree
/
Copy pathscript.js
More file actions
2631 lines (2372 loc) · 114 KB
/
Copy pathscript.js
File metadata and controls
2631 lines (2372 loc) · 114 KB
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
225
226
227
228
229
230
231
232
233
234
235
236
237
238
239
240
241
242
243
244
245
246
247
248
249
250
251
252
253
254
255
256
257
258
259
260
261
262
263
264
265
266
267
268
269
270
271
272
273
274
275
276
277
278
279
280
281
282
283
284
285
286
287
288
289
290
291
292
293
294
295
296
297
298
299
300
301
302
303
304
305
306
307
308
309
310
311
312
313
314
315
316
317
318
319
320
321
322
323
324
325
326
327
328
329
330
331
332
333
334
335
336
337
338
339
340
341
342
343
344
345
346
347
348
349
350
351
352
353
354
355
356
357
358
359
360
361
362
363
364
365
366
367
368
369
370
371
372
373
374
375
376
377
378
379
380
381
382
383
384
385
386
387
388
389
390
391
392
393
394
395
396
397
398
399
400
401
402
403
404
405
406
407
408
409
410
411
412
413
414
415
416
417
418
419
420
421
422
423
424
425
426
427
428
429
430
431
432
433
434
435
436
437
438
439
440
441
442
443
444
445
446
447
448
449
450
451
452
453
454
455
456
457
458
459
460
461
462
463
464
465
466
467
468
469
470
471
472
473
474
475
476
477
478
479
480
481
482
483
484
485
486
487
488
489
490
491
492
493
494
495
496
497
498
499
500
501
502
503
504
505
506
507
508
509
510
511
512
513
514
515
516
517
518
519
520
521
522
523
524
525
526
527
528
529
530
531
532
533
534
535
536
537
538
539
540
541
542
543
544
545
546
547
548
549
550
551
552
553
554
555
556
557
558
559
560
561
562
563
564
565
566
567
568
569
570
571
572
573
574
575
576
577
578
579
580
581
582
583
584
585
586
587
588
589
590
591
592
593
594
595
596
597
598
599
600
601
602
603
604
605
606
607
608
609
610
611
612
613
614
615
616
617
618
619
620
621
622
623
624
625
626
627
628
629
630
631
632
633
634
635
636
637
638
639
640
641
642
643
644
645
646
647
648
649
650
651
652
653
654
655
656
657
658
659
660
661
662
663
664
665
666
667
668
669
670
671
672
673
674
675
676
677
678
679
680
681
682
683
684
685
686
687
688
689
690
691
692
693
694
695
696
697
698
699
700
701
702
703
704
705
706
707
708
709
710
711
712
713
714
715
716
717
718
719
720
721
722
723
724
725
726
727
728
729
730
731
732
733
734
735
736
737
738
739
740
741
742
743
744
745
746
747
748
749
750
751
752
753
754
755
756
757
758
759
760
761
762
763
764
765
766
767
768
769
770
771
772
773
774
775
776
777
778
779
780
781
782
783
784
785
786
787
788
789
790
791
792
793
794
795
796
797
798
799
800
801
802
803
804
805
806
807
808
809
810
811
812
813
814
815
816
817
818
819
820
821
822
823
824
825
826
827
828
829
830
831
832
833
834
835
836
837
838
839
840
841
842
843
844
845
846
847
848
849
850
851
852
853
854
855
856
857
858
859
860
861
862
863
864
865
866
867
868
869
870
871
872
873
874
875
876
877
878
879
880
881
882
883
884
885
886
887
888
889
890
891
892
893
894
895
896
897
898
899
900
901
902
903
904
905
906
907
908
909
910
911
912
913
914
915
916
917
918
919
920
921
922
923
924
925
926
927
928
929
930
931
932
933
934
935
936
937
938
939
940
941
942
943
944
945
946
947
948
949
950
951
952
953
954
955
956
957
958
959
960
961
962
963
964
965
966
967
968
969
970
971
972
973
974
975
976
977
978
979
980
981
982
983
984
985
986
987
988
989
990
991
992
993
994
995
996
997
998
999
1000
/* =========================================================
ICONS
========================================================= */
const ICONS = {
home: '<path d="M4 11l8-7 8 7v9a1 1 0 0 1-1 1h-5v-6H10v6H5a1 1 0 0 1-1-1z"/>',
folder: '<path d="M4 6a1 1 0 0 1 1-1h4l2 2h8a1 1 0 0 1 1 1v10a1 1 0 0 1-1 1H5a1 1 0 0 1-1-1z"/>',
upload: '<path d="M12 16V6M12 6l-4 4M12 6l4 4"/><path d="M5 18h14"/>',
stack: '<path d="M12 4l8 4-8 4-8-4z"/><path d="M4 12l8 4 8-4M4 16l8 4 8-4"/>',
pin: '<circle cx="12" cy="10" r="3"/><path d="M12 21s7-6.5 7-11a7 7 0 1 0-14 0c0 4.5 7 11 7 11z"/>',
user: '<circle cx="12" cy="8" r="4"/><path d="M4 20c1.5-4 5-6 8-6s6.5 2 8 6"/>',
gear: '<circle cx="12" cy="12" r="3"/><path d="M12 2v3M12 19v3M4.2 4.2l2.1 2.1M17.7 17.7l2.1 2.1M2 12h3M19 12h3M4.2 19.8l2.1-2.1M17.7 6.3l2.1-2.1"/>',
help: '<circle cx="12" cy="12" r="9"/><path d="M9.5 9a2.5 2.5 0 0 1 4.8 1c0 1.5-2.3 1.7-2.3 3.5"/><circle cx="12" cy="17" r=".6" fill="currentColor"/>',
menu: '<path d="M4 7h16M4 12h16M4 17h16"/>',
search: '<circle cx="11" cy="11" r="7"/><path d="M21 21l-4.3-4.3"/>',
sun: '<circle cx="12" cy="12" r="4"/><path d="M12 2v2M12 20v2M4.9 4.9l1.4 1.4M17.7 17.7l1.4 1.4M2 12h2M20 12h2M4.9 19.1l1.4-1.4M17.7 6.3l1.4-1.4"/>',
moon: '<path d="M20 14.5A8.5 8.5 0 1 1 9.5 4a7 7 0 0 0 10.5 10.5z"/>',
eye: '<path d="M2 12s3.5-7 10-7 10 7 10 7-3.5 7-10 7-10-7-10-7z"/><circle cx="12" cy="12" r="3"/>',
eyeOff: '<path d="M3 3l18 18"/><path d="M10.6 5.2A10.9 10.9 0 0 1 12 5c6.5 0 10 7 10 7a17.6 17.6 0 0 1-3.2 4.1M6.6 6.6C4 8.3 2 12 2 12s3.5 7 10 7a10.6 10.6 0 0 0 4.2-.9"/><path d="M9.5 9.5a3 3 0 0 0 4.2 4.2"/>',
chevron: '<path d="M6 9l6 6 6-6"/>',
file: '<path d="M6 3h8l4 4v14a1 1 0 0 1-1 1H6a1 1 0 0 1-1-1V4a1 1 0 0 1 1-1z"/><path d="M14 3v4h4"/>'
};
function renderIcons(root = document) {
root.querySelectorAll('i[data-ic]').forEach(el => {
const name = el.getAttribute('data-ic');
if (!ICONS[name]) return;
el.innerHTML = `<svg width="18" height="18" viewBox="0 0 24 24" fill="none" stroke="currentColor" stroke-width="1.8" stroke-linecap="round" stroke-linejoin="round">${ICONS[name]}</svg>`;
});
}
/* =========================================================
CONFIG
========================================================= */
let CONFIG = null;
async function loadConfig() {
try {
const res = await fetch('data.json');
CONFIG = await res.json();
} catch (e) {
CONFIG = {
app: { name: 'Stash', tagline: 'Your school papers, sorted', cafePinExpiryMinutes: 10 },
plans: {
free: { label: 'Free', maxDocuments: 50, maxStorageMB: 2048, ocrPerDay: 5 },
pro: { label: 'Pro', maxDocuments: null, maxStorageMB: 20480, ocrPerDay: null }
},
upload: {
maxFileSizeMB: 10,
acceptedTypes: ['application/pdf', 'image/jpeg', 'image/jpg', 'image/png'],
acceptedExtensions: ['.pdf', '.jpg', '.jpeg', '.png'],
maxOcrPdfPages: 5
},
categories: [
{ id: 'Receipts', label: 'Receipts / Invoices', badge: 'RCT', color: 'accent', keywords: ['receipt', 'payment', 'paid', 'invoice', 'fee'] },
{ id: 'ClassPDFs', label: 'Class PDFs', badge: 'PDF', color: 'accent-2', keywords: ['lecture', 'syllabus', 'course outline'] },
{ id: 'Images', label: 'Images', badge: 'IMG', color: 'accent-3', keywords: ['photo', 'scan'] }
],
categoryMigration: { Receipt: 'Receipts', Docket: 'ClassPDFs', Admin: 'Images' },
levels: ['100L', '200L', '300L', '400L', '500L'],
semesters: ['First', 'Second'],
referenceTypes: ['RRR', 'Receipt No.', 'Invoice No.', 'Reference No.', 'Other'],
extraction: {
institutionMarkers: ['university', 'polytechnic', 'college', 'institute'],
referenceKeywords: [{ match: 'reference no', type: 'Reference No.' }],
studentIdKeywords: ['matric no', 'student id', 'reg no'],
semesterKeywords: { First: ['first semester'], Second: ['second semester'] }
},
clearanceRequirements: [],
defaultProfile: { name: '', matric: '', email: '', department: '', level: '100L', avatar: null },
defaultSettings: { theme: 'dark', offlineAccess: true, ocrAutofill: true },
pricingPlans: [],
faqs: []
};
}
return CONFIG;
}
/* =========================================================
FILE STORE (IndexedDB) — holds the actual file bytes.
localStorage (used for everything else) tops out around 5-10MB
per origin in most browsers, which can't honestly hold even a
handful of real 10MB uploads, let alone a 2GB/20GB quota. Blobs
live here; only small metadata lives in the DB/localStorage layer.
========================================================= */
const FileStore = {
_db: null,
_urlCache: new Map(),
open() {
if (this._db) return Promise.resolve(this._db);
if (typeof indexedDB === 'undefined') return Promise.reject(new Error('IndexedDB unavailable'));
return new Promise((resolve, reject) => {
const req = indexedDB.open('stash_files_db', 1);
req.onupgradeneeded = () => {
if (!req.result.objectStoreNames.contains('files')) {
req.result.createObjectStore('files');
}
};
req.onsuccess = () => { this._db = req.result; resolve(this._db); };
req.onerror = () => reject(req.error || new Error('Could not open local file storage'));
});
},
async put(id, blob) {
const db = await this.open();
return new Promise((resolve, reject) => {
const tx = db.transaction('files', 'readwrite');
tx.objectStore('files').put(blob, id);
tx.oncomplete = () => resolve(true);
tx.onerror = () => reject(tx.error || new Error('Could not save file'));
});
},
async get(id) {
const db = await this.open();
return new Promise((resolve, reject) => {
const tx = db.transaction('files', 'readonly');
const req = tx.objectStore('files').get(id);
req.onsuccess = () => resolve(req.result || null);
req.onerror = () => reject(req.error || new Error('Could not read file'));
});
},
async delete(id) {
const db = await this.open();
const cached = this._urlCache.get(id);
if (cached) { URL.revokeObjectURL(cached); this._urlCache.delete(id); }
return new Promise((resolve, reject) => {
const tx = db.transaction('files', 'readwrite');
tx.objectStore('files').delete(id);
tx.oncomplete = () => resolve(true);
tx.onerror = () => reject(tx.error || new Error('Could not delete file'));
});
},
// Object URLs are cached per document id so repeated preview/thumbnail
// renders don't keep allocating new blob: URLs.
async getObjectURL(id) {
if (this._urlCache.has(id)) return this._urlCache.get(id);
const blob = await this.get(id);
if (!blob) return null;
const url = URL.createObjectURL(blob);
this._urlCache.set(id, url);
return url;
}
};
/* =========================================================
PLANS — Free/Pro limits. Billing itself isn't connected yet
(that lands in a later phase); every account is on 'free' by
default and these limits are enforced for real regardless.
========================================================= */
const Plans = {
of(user) { return (user && user.plan) || 'free'; },
limits(user) {
const id = this.of(user);
return CONFIG.plans[id] || CONFIG.plans.free;
},
ocrPerDay(user) { return this.limits(user).ocrPerDay; }
};
/* =========================================================
DOCUMENT CONTENT EXTRACTION
Pulls real, searchable text out of an uploaded file (native PDF
text first, OCR only when there's no text layer), then applies
deterministic pattern-matching — no AI, no guessing — to surface
metadata like amount, date, reference number, institution, etc.
Anything that isn't confidently found is left blank rather than
invented.
========================================================= */
if (typeof pdfjsLib !== 'undefined') {
pdfjsLib.GlobalWorkerOptions.workerSrc = 'https://cdnjs.cloudflare.com/ajax/libs/pdf.js/3.11.174/pdf.worker.min.js';
}
const PDF_MIN_TEXT_LENGTH = 25; // below this, treat the PDF as having no usable text layer
async function extractPdfText(file) {
if (typeof pdfjsLib === 'undefined') return { text: '', pdf: null };
const buf = await file.arrayBuffer();
const pdf = await pdfjsLib.getDocument({ data: buf }).promise;
let text = '';
const pageCount = Math.min(pdf.numPages, 15); // cap for performance on huge files
for (let i = 1; i <= pageCount; i++) {
const page = await pdf.getPage(i);
const content = await page.getTextContent();
text += content.items.map(it => it.str).join(' ') + '\n';
}
return { text: text.trim(), pdf };
}
// Runs Tesseract with settings tuned for document/receipt text (a single
// uniform block, rather than the default multi-column-aware mode), using
// the explicit worker API so the page-segmentation mode actually applies.
async function runTesseractOCR(image) {
const worker = await Tesseract.createWorker('eng');
try {
await worker.setParameters({ tessedit_pageseg_mode: '6' }); // assume a single uniform block of text
const { data } = await worker.recognize(image);
return (data.text || '').trim();
} finally {
await worker.terminate();
}
}
// Basic, dependency-free preprocessing to help Tesseract on photographed
// receipts: upscale small images (more pixels for the recognizer to work
// with), convert to grayscale, and stretch contrast so faint text stands
// out — without the aggressive thresholding that destroys thin strokes.
// Drawing through <img>/canvas already respects EXIF orientation in
// current browsers, so no separate rotation step is needed here.
function preprocessImageForOCR(file) {
return new Promise((resolve, reject) => {
const img = new Image();
const objectUrl = URL.createObjectURL(file);
img.onload = () => {
try {
const { width, height } = img;
const minDimension = 1200; // upscale small photos for better recognition
const maxDimension = 3000; // cap so a huge photo can't hang the browser
let scale = 1;
if (Math.max(width, height) < minDimension) scale = minDimension / Math.max(width, height);
if (Math.max(width, height) * scale > maxDimension) scale = maxDimension / Math.max(width, height);
const targetW = Math.max(1, Math.round(width * scale));
const targetH = Math.max(1, Math.round(height * scale));
const canvas = document.createElement('canvas');
canvas.width = targetW;
canvas.height = targetH;
const ctx = canvas.getContext('2d');
ctx.drawImage(img, 0, 0, targetW, targetH);
const imageData = ctx.getImageData(0, 0, targetW, targetH);
const d = imageData.data;
let min = 255, max = 0;
for (let i = 0; i < d.length; i += 4) {
const gray = 0.299 * d[i] + 0.587 * d[i + 1] + 0.114 * d[i + 2];
d[i] = d[i + 1] = d[i + 2] = gray;
if (gray < min) min = gray;
if (gray > max) max = gray;
}
const range = Math.max(1, max - min); // avoid divide-by-zero on a flat/blank image
for (let i = 0; i < d.length; i += 4) {
const stretched = ((d[i] - min) / range) * 255;
d[i] = d[i + 1] = d[i + 2] = stretched;
}
ctx.putImageData(imageData, 0, 0);
resolve(canvas);
} catch (e) {
reject(e);
} finally {
URL.revokeObjectURL(objectUrl);
}
};
img.onerror = () => { URL.revokeObjectURL(objectUrl); reject(new Error('Could not load image for preprocessing')); };
img.src = objectUrl;
});
}
// OCRs a scanned PDF (no usable text layer) page by page, up to a
// configurable limit so a huge document can't freeze the browser or
// silently consume unbounded OCR resources. Pages are combined in order;
// if the document has more pages than the limit, that's reported back
// rather than silently pretending the whole thing was processed.
async function ocrScannedPdf(pdf, maxPages) {
const pageCount = Math.min(pdf.numPages, maxPages);
const parts = [];
for (let i = 1; i <= pageCount; i++) {
const page = await pdf.getPage(i);
const viewport = page.getViewport({ scale: 2 });
const canvas = document.createElement('canvas');
canvas.width = viewport.width;
canvas.height = viewport.height;
await page.render({ canvasContext: canvas.getContext('2d'), viewport }).promise;
const pageText = await runTesseractOCR(canvas);
if (pageText) parts.push(pageText);
}
return {
text: parts.join('\n\n').trim(),
pagesProcessed: pageCount,
totalPages: pdf.numPages,
truncated: pdf.numPages > pageCount
};
}
// Renders a PDF's pages as stacked canvases inside the given pane — an
// in-app preview using the existing pdf.js library, instead of handing
// the file off to the browser's native PDF viewer (which navigates away
// or triggers a download depending on the browser).
async function renderPdfIntoPane(pane, blob) {
if (typeof pdfjsLib === 'undefined') {
pane.innerHTML = `<div class="doc-preview-placeholder"><i data-ic="file"></i><p>PDF preview engine unavailable offline.</p></div>`;
renderIcons(pane);
return;
}
pane.innerHTML = `<div class="pdf-preview-scroll" id="pdfPreviewScroll"></div>`;
const scrollEl = document.getElementById('pdfPreviewScroll');
try {
const buf = await blob.arrayBuffer();
const pdf = await pdfjsLib.getDocument({ data: buf }).promise;
const pageCount = Math.min(pdf.numPages, 30); // sane cap so a huge PDF can't hang the preview
const targetWidth = Math.max(200, scrollEl.clientWidth || 380);
for (let i = 1; i <= pageCount; i++) {
const page = await pdf.getPage(i);
const unscaled = page.getViewport({ scale: 1 });
const scale = Math.min(2.5, targetWidth / unscaled.width);
const viewport = page.getViewport({ scale });
const canvas = document.createElement('canvas');
canvas.className = 'pdf-page-canvas';
canvas.width = viewport.width;
canvas.height = viewport.height;
scrollEl.appendChild(canvas);
await page.render({ canvasContext: canvas.getContext('2d'), viewport }).promise;
}
if (pdf.numPages > pageCount) {
const note = document.createElement('p');
note.className = 'muted pdf-preview-note';
note.textContent = `Showing the first ${pageCount} of ${pdf.numPages} pages. Download to view all.`;
scrollEl.appendChild(note);
}
} catch (e) {
pane.innerHTML = `<div class="doc-preview-placeholder"><i data-ic="file"></i><p>Couldn't render this PDF for preview. You can still download it below.</p></div>`;
renderIcons(pane);
}
}
const Extractor = {
// Finds a currency amount, but only when there's real contextual evidence
// — a currency symbol/code, or an explicit label like "Amount"/"Total".
// A bare number is never treated as an amount, since that's how phone
// numbers, student IDs and reference numbers get misread as money.
findAmount(text) {
// Unambiguous currency markers — safe to trust directly.
const strongMatch = text.match(/(?:\u20a6|NGN|\$|USD|\u00a3|GBP|\u20ac|EUR)\s?([\d,]{3,12}(?:\.\d{1,2})?)/i);
if (strongMatch) { const v = this._cleanAmount(strongMatch[1]); if (v) return v; }
// Bare "N" for Naira is ambiguous (collides with room numbers, IDs,
// footnotes) — only trust it when it's comma-grouped or has decimals,
// which is how real amounts are actually written.
const shorthandMatch = text.match(/\bN\s?(\d{1,3}(?:,\d{3})+(?:\.\d{1,2})?|\d+\.\d{2})\b/);
if (shorthandMatch) { const v = this._cleanAmount(shorthandMatch[1]); if (v) return v; }
// Explicit wording near a number.
const wordMatch = text.match(/(?:amount|total|sum paid|amount paid|fee)s?\s*(?:paid|due)?\s*[:\-]?\s*(?:\u20a6|N|\$|\u00a3|\u20ac)?\s?([\d,]{3,12}(?:\.\d{1,2})?)/i);
if (wordMatch) { const v = this._cleanAmount(wordMatch[1]); if (v) return v; }
return '';
},
_cleanAmount(raw) {
const cleaned = String(raw || '').replace(/,/g, '');
const num = Number(cleaned);
if (!cleaned || isNaN(num) || num <= 0) return '';
return cleaned;
},
// Finds a date and normalizes it to yyyy-mm-dd, but only when it passes
// a real calendar-validity check (correct month/day range, a plausible
// year). This stops random number sequences from being read as dates.
findDate(text) {
const months = ['january', 'february', 'march', 'april', 'may', 'june', 'july', 'august', 'september', 'october', 'november', 'december'];
const monthWordMatch = text.match(/\b(\d{1,2})\s+(jan|feb|mar|apr|may|jun|jul|aug|sep|oct|nov|dec)[a-z]*\s+(\d{4})\b/i)
|| text.match(/\b(jan|feb|mar|apr|may|jun|jul|aug|sep|oct|nov|dec)[a-z]*\s+(\d{1,2}),?\s+(\d{4})\b/i);
if (monthWordMatch) {
let day, monthWord, year;
if (/^\d/.test(monthWordMatch[0])) { [, day, monthWord, year] = monthWordMatch; }
else { [, monthWord, day, year] = monthWordMatch; }
const mIdx = months.findIndex(m => m.startsWith(monthWord.toLowerCase().slice(0, 3)));
if (mIdx > -1 && this._isValidDate(year, mIdx + 1, day)) {
return `${year}-${String(mIdx + 1).padStart(2, '0')}-${String(day).padStart(2, '0')}`;
}
}
const isoMatch = text.match(/\b(20\d{2})-(\d{2})-(\d{2})\b/);
if (isoMatch && this._isValidDate(isoMatch[1], isoMatch[2], isoMatch[3])) {
return `${isoMatch[1]}-${isoMatch[2]}-${isoMatch[3]}`;
}
const numericMatch = text.match(/\b(\d{1,2})[\/\-.](\d{1,2})[\/\-.](\d{2,4})\b/);
if (numericMatch) {
let [, a, b, y] = numericMatch;
if (y.length === 2) y = '20' + y;
// Try day/month first (common outside the US); fall back to month/day.
if (this._isValidDate(y, b, a)) return `${y}-${String(b).padStart(2, '0')}-${String(a).padStart(2, '0')}`;
if (this._isValidDate(y, a, b)) return `${y}-${String(a).padStart(2, '0')}-${String(b).padStart(2, '0')}`;
}
return '';
},
_isValidDate(y, m, d) {
y = Number(y); m = Number(m); d = Number(d);
if (!y || !m || !d) return false;
if (m < 1 || m > 12 || d < 1 || d > 31) return false;
if (y < 1990 || y > 2100) return false; // sane range for academic documents
const date = new Date(y, m - 1, d);
return date.getFullYear() === y && date.getMonth() === m - 1 && date.getDate() === d;
},
// Reference number ONLY when a real label ("Receipt No", "RRR", etc.) is
// found nearby. No blind digit-pattern fallback — a grouped-digit string
// with no label is exactly as likely to be a phone or account number.
findReference(text, config) {
const lower = text.toLowerCase();
for (const { match, type } of (config.referenceKeywords || [])) {
const idx = lower.indexOf(match);
if (idx === -1) continue;
const after = text.slice(idx + match.length, idx + match.length + 40);
const codeMatch = after.match(/^[:\-\s]*([A-Za-z0-9][A-Za-z0-9\/\- ]{2,24})/);
if (!codeMatch) continue;
const raw = codeMatch[1].trim().split(/\s{2,}/)[0]; // stop at an accidental run into the next word
const digitsOnlyLen = raw.replace(/[^A-Za-z0-9]/g, '').length;
if (/\d/.test(raw) && digitsOnlyLen >= 5 && digitsOnlyLen <= 20) {
return { referenceNumber: raw, referenceType: type };
}
}
return { referenceNumber: '', referenceType: '' };
},
// A line is only treated as an institution name when it contains a
// recognizable institution-type word AND is more than a single stray word.
findInstitution(text, markers) {
const lines = text.split(/\n|(?<=\.)\s+/).map(l => l.trim()).filter(Boolean);
const lower = markers.map(m => m.toLowerCase());
const hit = lines.find(line => {
const l = line.toLowerCase();
return lower.some(m => l.includes(m)) && line.split(/\s+/).length >= 2;
});
return hit ? hit.slice(0, 80) : '';
},
findStudentId(text, keywords) {
const lower = text.toLowerCase();
for (const kw of keywords) {
const idx = lower.indexOf(kw);
if (idx === -1) continue;
const after = text.slice(idx + kw.length, idx + kw.length + 30);
const codeMatch = after.match(/[:\-\s]*([A-Za-z0-9\/\-]{4,20})/);
if (codeMatch) return codeMatch[1].trim();
}
return '';
},
// Two consecutive years (e.g. "2023/2024") — but only when the second
// year is genuinely the first plus one, so two unrelated 4-digit numbers
// separated by a slash don't get read as an academic session.
findAcademicSession(text) {
const m = text.match(/\b(20\d{2})\s?[\/\-]\s?(20\d{2})\b/);
if (!m) return '';
if (Number(m[2]) !== Number(m[1]) + 1) return '';
return `${m[1]}/${m[2]}`;
},
findSemester(text, semesterKeywords) {
const lower = text.toLowerCase();
for (const [semester, phrases] of Object.entries(semesterKeywords || {})) {
if (phrases.some(p => lower.includes(p))) return semester;
}
return '';
},
// Only a direct match against a configured level string (in either
// "200L" or "200 Level" form). No numeric "Year N" guessing — a stray
// "2" near the word "year" is not credible evidence of academic level.
findLevel(text, levels) {
for (const lvl of levels) {
const digits = String(lvl).match(/\d+/);
if (digits) {
const re = new RegExp(`\\b${digits[0]}\\s?(?:level|l)\\b`, 'i');
if (re.test(text)) return lvl;
} else if (text.toLowerCase().includes(String(lvl).toLowerCase())) {
return lvl;
}
}
return '';
},
// Looks for a short, confident-looking document title near the top of
// the extracted text, using a confidence score rather than a rigid
// casing rule — position near the top, short length, title-like
// casing, and document-title vocabulary ("receipt", "form", etc.) all
// add confidence; things that look like a labeled field, address,
// date, phone/reference number, or body prose are excluded outright.
// Returns '' when nothing clears the confidence threshold — this only
// ever feeds a filename suggestion, never a forced rename on weak
// evidence.
findHeading(text) {
const smallWords = new Set(['of', 'and', 'the', 'for', 'in', 'on', 'a', 'an', 'to']);
const titleVocabulary = ['receipt', 'invoice', 'certificate', 'letter', 'form', 'registration', 'statement', 'transcript', 'admission', 'clearance', 'confirmation', 'notice', 'slip', 'record', 'report', 'card', 'schedule', 'result', 'syllabus', 'outline'];
const lines = text.split(/\n/).map(l => l.trim()).filter(Boolean).slice(0, 12);
let best = '', bestScore = 0;
lines.forEach((line, idx) => {
if (line.length < 6 || line.length > 60) return;
const words = line.split(/\s+/).filter(Boolean);
if (words.length < 2 || words.length > 9) return;
if (!/[A-Za-z]/.test(line)) return;
// Hard exclusions — these disqualify a line outright regardless of score.
if (/\d{4,}/.test(line)) return; // reference/account/phone-shaped
if (/^\d/.test(line)) return; // starts with a digit — date/address/reference
if (/@|https?:\/\/|www\./i.test(line)) return; // email/url
if (/:\s*\S/.test(line) && /\d/.test(line)) return; // a labeled field, e.g. "Date: 12/03/2024"
if (/\b(street|st\.?|road|rd\.?|avenue|ave\.?|close|crescent|drive|lane)\b/i.test(line)) return; // address
if (/^[A-Z][a-zA-Z'-]*,\s*[A-Z]/.test(line) && words.length <= 4) return; // "City, State" style locale line
if ((line.match(/\d/g) || []).length > 3) return; // too many stray digits for a clean title
if (/[.!?]\s+[A-Z]/.test(line)) return; // multiple sentences — prose, not a title
if (!words.some(w => w.replace(/[^A-Za-z]/g, '').length >= 5)) return; // needs at least one real word, not just short abbreviations
const alpha = line.replace(/[^A-Za-z]/g, '');
if (alpha.length < 5) return; // not enough letters to judge casing confidently
const upper = alpha.replace(/[^A-Z]/g, '');
const isMostlyUpper = (upper.length / alpha.length) > 0.7;
const significantWords = words.filter(w => !smallWords.has(w.toLowerCase()));
const capitalizedWords = significantWords.filter(w => /^[A-Z]/.test(w));
const isTitleCase = significantWords.length > 0 && capitalizedWords.length >= Math.ceil(significantWords.length * 0.7);
// Casing is a hard requirement, not just a scoring input — otherwise
// vocabulary + position alone could let plain lowercase prose (e.g.
// "this receipt confirms that payment has been received") through.
// The score below only ranks among lines that already look like a
// real heading, choosing the best candidate rather than just the
// first line that happens to match.
if (!isMostlyUpper && !isTitleCase) return;
let score = 0;
score += Math.max(0, 4 - idx); // position near the top is a strong signal
if (words.length <= 6) score += 2; // short, title-length lines
else score += 1; // still plausible up to 9 words, just less confident
if (isMostlyUpper) score += 3;
if (isTitleCase) score += 3;
if (titleVocabulary.some(v => line.toLowerCase().includes(v))) score += 2; // document-title vocabulary
if (score > bestScore) { bestScore = score; best = line; }
});
return bestScore >= 6 ? best : '';
},
// A category is only suggested when one category clearly leads on
// keyword evidence — at least two distinct keyword hits, and not tied
// with the runner-up. A single incidental word is not enough evidence.
// Category priority: (1) strong Receipts/Invoices evidence — either
// multiple keyword hits, or one strong signal (an amount and a
// reference number together, or a keyword hit alongside either) —
// (2) strong Class PDFs evidence, where being a PDF itself counts as a
// supporting signal alongside at least one content keyword, (3) a
// plain image with no stronger evidence falls back to Images, (4)
// otherwise left blank. The three category IDs are referenced directly
// because this priority logic is specific to what each of Stash's
// three categories means; the keyword word-lists themselves stay fully
// configurable via data.json.
guessCategory(text, categories, { amount, referenceNumber, isPdf, isImage } = {}) {
const lower = (text || '').toLowerCase();
const scoreFor = id => {
const cat = categories.find(c => c.id === id);
if (!cat) return 0;
return (cat.keywords || []).reduce((s, k) => s + (lower.includes(k.toLowerCase()) ? 1 : 0), 0);
};
const receiptScore = scoreFor('Receipts');
const classScore = scoreFor('ClassPDFs');
const strongReceipt = receiptScore >= 2 || (receiptScore >= 1 && (amount || referenceNumber)) || (amount && referenceNumber);
if (strongReceipt) return 'Receipts';
const strongClass = classScore >= 2 || (isPdf && classScore >= 1);
if (strongClass) return 'ClassPDFs';
if (isImage) return 'Images';
return '';
},
// Runs every extractor against a block of text and returns only the
// fields it found reasonable evidence for. Anything not found comes
// back as '' rather than a guess — never invented. Category is the one
// field that can still get a value with no text at all (a plain image
// file falls back to "Images" on file type alone).
analyze(text, fileType) {
const cfg = CONFIG.extraction || {};
const hasText = !!(text && text.trim());
const ref = hasText ? this.findReference(text, cfg) : { referenceNumber: '', referenceType: '' };
const result = {
amount: hasText ? this.findAmount(text) : '',
date: hasText ? this.findDate(text) : '',
referenceNumber: ref.referenceNumber,
referenceType: ref.referenceType,
institution: hasText ? this.findInstitution(text, cfg.institutionMarkers || []) : '',
studentId: hasText ? this.findStudentId(text, cfg.studentIdKeywords || []) : '',
academicSession: hasText ? this.findAcademicSession(text) : '',
semester: hasText ? this.findSemester(text, cfg.semesterKeywords || {}) : '',
level: hasText ? this.findLevel(text, CONFIG.levels || []) : ''
};
result.category = this.guessCategory(hasText ? text : '', CONFIG.categories || [], {
amount: result.amount,
referenceNumber: result.referenceNumber,
isPdf: fileType === 'application/pdf',
isImage: (fileType || '').startsWith('image/')
});
return result;
}
};
const DB = {
_read(key, fallback) {
try {
const raw = localStorage.getItem(key);
return raw ? JSON.parse(raw) : fallback;
} catch (e) { return fallback; }
},
_write(key, value) { localStorage.setItem(key, JSON.stringify(value)); },
async getUsers() { return this._read('stash_users', {}); },
async saveUsers(users) { this._write('stash_users', users); return users; },
async findUserByIdentifier(identifier) {
const users = await this.getUsers();
const id = (identifier || '').toLowerCase();
return Object.values(users).find(u => u.matric.toLowerCase() === id || u.email.toLowerCase() === id) || null;
},
async createUser({ name, matric, email, password, level }) {
const users = await this.getUsers();
const id = 'u_' + Date.now().toString(36) + Math.random().toString(36).slice(2, 6);
const profile = { ...CONFIG.defaultProfile, name, matric, email, level: level || '100L' };
const settings = { ...CONFIG.defaultSettings };
const user = { id, name, matric, email, password, plan: 'free', profile, settings, documents: [] };
users[id] = user;
await this.saveUsers(users);
return user;
},
async updateUser(id, patch) {
const users = await this.getUsers();
if (!users[id]) return null;
users[id] = { ...users[id], ...patch };
await this.saveUsers(users);
return users[id];
},
// Used when editing a profile, to stop two accounts from ending up with
// the same matric number or email (which would break identifier login).
async findConflictingUser(matric, email, excludeId) {
const users = await this.getUsers();
const m = (matric || '').toLowerCase();
const e = (email || '').toLowerCase();
return Object.values(users).find(u =>
u.id !== excludeId && ((m && u.matric.toLowerCase() === m) || (e && u.email.toLowerCase() === e))
) || null;
},
async deleteUser(id) {
const users = await this.getUsers();
const docs = (users[id] && users[id].documents) || [];
delete users[id];
await this.saveUsers(users);
await Promise.all(docs.map(d => FileStore.delete(d.id).catch(() => {})));
// Revoke any cafe PINs the deleted account had issued so a stale link
// can't keep exposing a document after the account is gone.
const pins = await this.getPins();
let changed = false;
Object.keys(pins).forEach(p => {
if (pins[p].userId === id) { delete pins[p]; changed = true; }
});
if (changed) await this.savePins(pins);
},
async getDocuments(userId) {
const users = await this.getUsers();
const docs = (users[userId] && users[userId].documents) || [];
// One-time migration: earlier builds embedded files as base64 directly
// in the document record. Move any of those into FileStore so they
// don't sit uncounted (or get lost) under the new storage model.
const legacy = docs.filter(d => d.fileData && !d.hasFile);
if (legacy.length) {
for (const d of legacy) {
try {
const blob = dataURLtoBlob(d.fileData);
await FileStore.put(d.id, blob);
d.hasFile = true;
d.size = blob.size;
d.fileType = d.fileType || blob.type;
d.uploadedAt = d.uploadedAt || new Date().toISOString();
d.modifiedAt = d.modifiedAt || d.uploadedAt;
d.referenceNumber = d.referenceNumber || d.rrr || '';
d.referenceType = d.referenceType || 'RRR';
d.ocrStatus = d.ocrStatus || (d.ocrText ? 'done' : 'skipped');
delete d.fileData;
delete d.rrr;
} catch (e) { /* leave this one as-is; it'll just show "no preview" */ }
}
await this.saveDocuments(userId, docs);
}
// Migrate documents saved under the old category names (Receipt /
// Docket / Admin) to the current category IDs. Old documents keep
// their category — the meaning is preserved, only the ID changes.
const catMap = CONFIG.categoryMigration || {};
const needsCatMigration = docs.filter(d => d.category && catMap[d.category]);
if (needsCatMigration.length) {
needsCatMigration.forEach(d => { d.category = catMap[d.category]; });
await this.saveDocuments(userId, docs);
}
// Safe in-memory defaults for documents saved before this correction
// pass, so older records don't break new UI that expects these fields.
return docs.map(d => ({
originalFilename: d.name,
autoFilled: [],
...d
}));
},
async saveDocuments(userId, docs) {
const users = await this.getUsers();
if (!users[userId]) return [];
users[userId].documents = docs;
try {
await this.saveUsers(users);
} catch (e) {
throw new Error('Could not save changes — local storage may be full.');
}
return docs;
},
async getProfile(userId) {
const users = await this.getUsers();
return (users[userId] && users[userId].profile) || { ...CONFIG.defaultProfile };
},
async saveProfile(userId, profile) {
const users = await this.getUsers();
if (!users[userId]) return profile;
users[userId].profile = profile;
// Keep the top-level lookup fields (used for login) in sync with the profile
// so editing your name/matric/email in Profile doesn't desync from what's
// shown on the dashboard or used to log back in.
users[userId].name = profile.name || users[userId].name;
users[userId].matric = profile.matric || users[userId].matric;
users[userId].email = profile.email || users[userId].email;
await this.saveUsers(users);
return profile;
},
async getSettings(userId) {
const users = await this.getUsers();
return (users[userId] && users[userId].settings) || { ...CONFIG.defaultSettings };
},
async saveSettings(userId, settings) {
const users = await this.getUsers();
if (!users[userId]) return settings;
users[userId].settings = settings;
await this.saveUsers(users);
return settings;
},
todayKey() { return new Date().toISOString().slice(0, 10); },
// Usage is tracked per calendar day and rolls over automatically —
// if the stored date isn't today, the count is treated as 0.
async usageToday(userId, kind) {
const users = await this.getUsers();
const usage = users[userId] && users[userId].usage && users[userId].usage[kind];
if (!usage || usage.date !== this.todayKey()) return 0;
return usage.count || 0;
},
async incrementUsage(userId, kind) {
const users = await this.getUsers();
if (!users[userId]) return 0;
const today = this.todayKey();
const current = users[userId].usage && users[userId].usage[kind];
const count = (current && current.date === today) ? current.count + 1 : 1;
users[userId].usage = { ...(users[userId].usage || {}), [kind]: { date: today, count } };
await this.saveUsers(users);
return count;
},
getSession() {
const local = this._read('stash_session', null);
if (local) return local;
try {
const raw = sessionStorage.getItem('stash_session');
return raw ? JSON.parse(raw) : null;
} catch (e) { return null; }
},
setSession(userId, remember) {
const payload = { userId };
if (remember) {
localStorage.setItem('stash_session', JSON.stringify(payload));
sessionStorage.removeItem('stash_session');
} else {
sessionStorage.setItem('stash_session', JSON.stringify(payload));
localStorage.removeItem('stash_session');
}
},
clearSession() {
localStorage.removeItem('stash_session');
sessionStorage.removeItem('stash_session');
},
async getPins() { return this._read('stash_cafe_pins', {}); },
async savePins(pins) { this._write('stash_cafe_pins', pins); return pins; },
async createPin(userId, doc) {
const pins = await this.getPins();
let pin;
do { pin = String(Math.floor(1000 + Math.random() * 9000)); } while (pins[pin]);
const expiresAt = Date.now() + CONFIG.app.cafePinExpiryMinutes * 60 * 1000;
pins[pin] = { userId, doc, expiresAt };
await this.savePins(pins);
return { pin, expiresAt };
},
async consumePin(pin) {
const pins = await this.getPins();
const entry = pins[pin];
if (!entry) return null;
if (Date.now() > entry.expiresAt) { delete pins[pin]; await this.savePins(pins); return null; }
return entry;
}
};
/* =========================================================
UI HELPERS
========================================================= */
const UI = {
toast(msg) {
const t = document.getElementById('toast');
t.textContent = msg;
t.classList.add('show');
clearTimeout(t._timer);
t._timer = setTimeout(() => t.classList.remove('show'), 2600);
},
openSidebar() {
document.getElementById('sidebar').classList.add('open');
document.getElementById('scrim').classList.add('show');
},
closeSidebar() {
document.getElementById('sidebar').classList.remove('open');
document.getElementById('scrim').classList.remove('show');
},
setTheme(mode) {
document.body.classList.toggle('light-mode', mode === 'light');
localStorage.setItem('stash_theme', mode);
document.querySelectorAll('.seg-btn').forEach(b => b.classList.toggle('active', b.dataset.theme === mode));
},
initTheme() { this.setTheme(localStorage.getItem('stash_theme') || 'dark'); },
toggleTheme() { this.setTheme(document.body.classList.contains('light-mode') ? 'dark' : 'light'); },
openModal(id) { document.getElementById(id).classList.remove('hidden'); },
closeModal(id) { document.getElementById(id).classList.add('hidden'); },
showFieldError(id, message) {
const el = document.getElementById(id);
if (!el) return;
if (message) el.textContent = message;
el.classList.remove('hidden');
},
hideFieldError(id) {
const el = document.getElementById(id);
if (el) el.classList.add('hidden');
}
};
function isValidEmail(str) {
return /^[^\s@]+@[^\s@]+\.[^\s@]+$/.test((str || '').trim());
}
function escapeHTML(str) {
const div = document.createElement('div');
div.textContent = str == null ? '' : String(str);
return div.innerHTML;
}
function formatFileSize(bytes) {
if (!bytes) return '0 KB';
if (bytes < 1024 * 1024) return `${Math.max(1, Math.round(bytes / 1024))} KB`;
return `${(bytes / (1024 * 1024)).toFixed(1)} MB`;
}
function formatMB(mb) {
if (!mb) return '0 MB';
if (mb < 100) return `${mb.toFixed(2)} MB`;
return `${Math.round(mb)} MB`;
}
function toTitleCase(str) {
return str.toLowerCase().replace(/\b[a-z]/g, c => c.toUpperCase());
}
function sanitizeSuggestedName(str) {
return str
.replace(/[\\/:*?"<>|]/g, '') // filesystem/browser-unsafe characters
.replace(/\s+/g, ' ')
.trim()
.slice(0, 80);
}
function dataURLtoBlob(dataURL) {
const [header, base64] = dataURL.split(',');
const mime = (header.match(/data:(.*?);base64/) || [])[1] || 'application/octet-stream';
const bin = atob(base64);
const bytes = new Uint8Array(bin.length);
for (let i = 0; i < bin.length; i++) bytes[i] = bin.charCodeAt(i);
return new Blob([bytes], { type: mime });
}
function fileToDataURL(file) {
return new Promise((resolve, reject) => {
const reader = new FileReader();
reader.onload = () => resolve(reader.result);
reader.onerror = reject;
reader.readAsDataURL(file);
});
}
/* =========================================================
ROUTER
========================================================= */
const MARKETING = ['landing', 'pricing'];
const AUTH_ONLY = ['login', 'signup'];
const PROTECTED = ['dashboard', 'documents', 'upload', 'clearance', 'cafepin', 'profile', 'settings', 'help'];
const PUBLIC_STANDALONE = ['print-portal'];
function currentRoute() {
const h = (window.location.hash || '').replace('#', '');
return h || 'landing';
}
async function router() {
let route = currentRoute();
const session = DB.getSession();
if (PROTECTED.includes(route) && !session) { window.location.hash = '#login'; route = 'login'; }
else if (AUTH_ONLY.includes(route) && session) { window.location.hash = '#dashboard'; route = 'dashboard'; }
else if (!MARKETING.includes(route) && !AUTH_ONLY.includes(route) && !PROTECTED.includes(route) && !PUBLIC_STANDALONE.includes(route)) {
window.location.hash = '#landing'; route = 'landing';
}
document.getElementById('marketingShell').classList.toggle('hidden', !MARKETING.includes(route));
document.getElementById('authShell').classList.toggle('hidden', !AUTH_ONLY.includes(route));
document.getElementById('appShell').classList.toggle('hidden', !PROTECTED.includes(route));
document.getElementById('printShell').classList.toggle('hidden', !PUBLIC_STANDALONE.includes(route));
if (MARKETING.includes(route)) {
document.getElementById('view-landing').hidden = route !== 'landing';
document.getElementById('view-pricing').hidden = route !== 'pricing';
if (route === 'pricing') Views.renderPricing();
if (route === 'landing') renderIcons(document.getElementById('view-landing'));
}
if (AUTH_ONLY.includes(route)) {
document.getElementById('view-login').hidden = route !== 'login';
document.getElementById('view-signup').hidden = route !== 'signup';
}
if (PROTECTED.includes(route)) {
PROTECTED.forEach(r => { document.getElementById('view-' + r).hidden = r !== route; });
document.querySelectorAll('.nav-item').forEach(a => a.classList.toggle('active', a.dataset.route === route));
UI.closeSidebar();
// Keep the topbar avatar/name correct no matter which protected view we land on.
const user = await Auth.currentUser();
if (user) {
const profile = await DB.getProfile(user.id);
Views.setAvatarDisplay(profile);
}
if (route === 'dashboard') await Views.renderDashboard();
if (route === 'documents') await Views.renderDocuments();
if (route === 'upload') await Views.renderUploadCapacity();
if (route === 'clearance') await Views.renderClearance();
if (route === 'cafepin') await Views.populateCafeSelect();
if (route === 'profile') await Views.renderProfile();
if (route === 'settings') await Views.renderSettings();
if (route === 'help') Views.renderHelp();
}
document.title = ({
landing: 'Stash — Your school papers, sorted',
pricing: 'Pricing · Stash',
login: 'Log in · Stash',
signup: 'Create your account · Stash',
dashboard: 'Dashboard · Stash',
documents: 'My Documents · Stash',
upload: 'Scan & Upload · Stash',
clearance: 'Clearance Export · Stash',
cafepin: 'Cafe Quick-Print · Stash',
profile: 'Profile · Stash',
settings: 'Settings · Stash',
help: 'Help & Support · Stash',
'print-portal': 'Cafe Print · Stash'
})[route] || 'Stash';
}
/* =========================================================
AUTH
========================================================= */
const Auth = {
async login(identifier, password, remember) {
const user = await DB.findUserByIdentifier(identifier);
if (!user || user.password !== password) {
return { ok: false, error: 'That matric number/email and password don\u2019t match.' };
}
DB.setSession(user.id, remember);