Repository navigation
Expand file tree
/
Copy pathshared.js
More file actions
1168 lines (1111 loc) · 55.6 KB
/
Copy pathshared.js
File metadata and controls
1168 lines (1111 loc) · 55.6 KB
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
225
226
227
228
229
230
231
232
233
234
235
236
237
238
239
240
241
242
243
244
245
246
247
248
249
250
251
252
253
254
255
256
257
258
259
260
261
262
263
264
265
266
267
268
269
270
271
272
273
274
275
276
277
278
279
280
281
282
283
284
285
286
287
288
289
290
291
292
293
294
295
296
297
298
299
300
301
302
303
304
305
306
307
308
309
310
311
312
313
314
315
316
317
318
319
320
321
322
323
324
325
326
327
328
329
330
331
332
333
334
335
336
337
338
339
340
341
342
343
344
345
346
347
348
349
350
351
352
353
354
355
356
357
358
359
360
361
362
363
364
365
366
367
368
369
370
371
372
373
374
375
376
377
378
379
380
381
382
383
384
385
386
387
388
389
390
391
392
393
394
395
396
397
398
399
400
401
402
403
404
405
406
407
408
409
410
411
412
413
414
415
416
417
418
419
420
421
422
423
424
425
426
427
428
429
430
431
432
433
434
435
436
437
438
439
440
441
442
443
444
445
446
447
448
449
450
451
452
453
454
455
456
457
458
459
460
461
462
463
464
465
466
467
468
469
470
471
472
473
474
475
476
477
478
479
480
481
482
483
484
485
486
487
488
489
490
491
492
493
494
495
496
497
498
499
500
501
502
503
504
505
506
507
508
509
510
511
512
513
514
515
516
517
518
519
520
521
522
523
524
525
526
527
528
529
530
531
532
533
534
535
536
537
538
539
540
541
542
543
544
545
546
547
548
549
550
551
552
553
554
555
556
557
558
559
560
561
562
563
564
565
566
567
568
569
570
571
572
573
574
575
576
577
578
579
580
581
582
583
584
585
586
587
588
589
590
591
592
593
594
595
596
597
598
599
600
601
602
603
604
605
606
607
608
609
610
611
612
613
614
615
616
617
618
619
620
621
622
623
624
625
626
627
628
629
630
631
632
633
634
635
636
637
638
639
640
641
642
643
644
645
646
647
648
649
650
651
652
653
654
655
656
657
658
659
660
661
662
663
664
665
666
667
668
669
670
671
672
673
674
675
676
677
678
679
680
681
682
683
684
685
686
687
688
689
690
691
692
693
694
695
696
697
698
699
700
701
702
703
704
705
706
707
708
709
710
711
712
713
714
715
716
717
718
719
720
721
722
723
724
725
726
727
728
729
730
731
732
733
734
735
736
737
738
739
740
741
742
743
744
745
746
747
748
749
750
751
752
753
754
755
756
757
758
759
760
761
762
763
764
765
766
767
768
769
770
771
772
773
774
775
776
777
778
779
780
781
782
783
784
785
786
787
788
789
790
791
792
793
794
795
796
797
798
799
800
801
802
803
804
805
806
807
808
809
810
811
812
813
814
815
816
817
818
819
820
821
822
823
824
825
826
827
828
829
830
831
832
833
834
835
836
837
838
839
840
841
842
843
844
845
846
847
848
849
850
851
852
853
854
855
856
857
858
859
860
861
862
863
864
865
866
867
868
869
870
871
872
873
874
875
876
877
878
879
880
881
882
883
884
885
886
887
888
889
890
891
892
893
894
895
896
897
898
899
900
901
902
903
904
905
906
907
908
909
910
911
912
913
914
915
916
917
918
919
920
921
922
923
924
925
926
927
928
929
930
931
932
933
934
935
936
937
938
939
940
941
942
943
944
945
946
947
948
949
950
951
952
953
954
955
956
957
958
959
960
961
962
963
964
965
966
967
968
969
970
971
972
973
974
975
976
977
978
979
980
981
982
983
984
985
986
987
988
989
990
991
992
993
994
995
996
997
998
999
1000
//Shared Variables between popup and context scripts
var smartcopyurl = "https://historylink.herokuapp.com"; //helpful for local testing to switch from https to http
var genifamily, focusid, tabuniondatalink;
var familystatus = [], genifamilydata = {};
var focusgender = "unknown";
var uniondata = [];
// #213: Deep Research controls (persistent on/off setting + a live per-run
// skip), declared here rather than popup.js since every collections/*.js
// tab-fetch queue needs to read them directly as bare globals, and
// shared.js is the one script guaranteed to load before both popup.js and
// every collections/*.js file. deepResearchOn mirrors chrome.storage.local's
// 'deepresearchenabled' key (see popup.js) and is fixed for the popup's
// lifetime otherwise. deepResearchSkipRun is never persisted - it's reset to
// false at the start of every new "read family data" run so a skip never
// silently carries over onto the next profile.
var deepResearchOn = true;
var deepResearchSkipRun = false;
// #231 follow-up (live-reported): the eyeball icon - even after being
// fixed to reflect current state rather than the click action - was
// still confusing at a glance (an eye-open vs. eye-slashed glyph
// requires interpreting a small color/shape difference). Replaced with
// a plain text toggle: a disclosure-triangle glyph (▶ collapsed, ▼
// expanded) paired with a label naming the click action itself ("Show
// all" / "Show less") - the triangle still reflects current state
// (Apple HIG convention, same reasoning #231 already established), the
// text just makes the action unambiguous without requiring any icon
// interpretation at all.
//
// popup.html's own static <span id="focusshowhide"> tag can't
// reference a JS constant (it's plain HTML, not templated) - its
// initial text is SHOW_ALL_LABEL hardcoded directly, matching the
// default collapsed state ("Hide Empty Fields" defaults on). Keep both
// in sync if this label ever changes.
// Leading space is intentional and load-bearing, not stray whitespace -
// the focus-profile placement (popup.html) sits this span directly after
// plain "Update Profile" text with no space of its own, and CSS
// padding-left alone wasn't rendering enough visual separation (live-
// reported: "Update Profile▶ Show all" ran together).
// Small solid triangles (▸/▾), not full-size (▶/▼) - live-reported the
// full-size glyphs read as bold regardless of font-weight, since
// font-weight has no effect on a filled Unicode symbol's shape; these
// are a dedicated smaller variant, still filled rather than outline.
var SHOW_ALL_LABEL = ' ▸ Show all fields';
var SHOW_LESS_LABEL = ' ▾ Hide unused fields';
// Registry of abort callbacks for every Deep Research tab fetch currently
// in flight (one entry per open tab, across all four collections). Skipping
// mid-run only needs to stop FUTURE tabs from starting (the runNext*TabFetch
// gate above handles that) - clicking skip while a tab is already open and
// polling would otherwise still wait out that tab's own multi-second
// timeout before the run actually finishes. Each run*TabFetch() pushes its
// own abort function here when its tab opens and removes it once settled;
// the skip button (popup.js) calls every entry immediately on click.
var deepResearchInFlightAborts = [];
// #208: mirrors chrome.storage.local's 'estimatebirthyears' key (see
// popup.js), default OFF - this feature writes inferred, not sourced,
// data. Not read directly inside the pure estimation functions in
// buildform.js (getMemberSpouse()/getChildGroupAnchorYear()/
// estimateBirthYear()) - those stay DOM/global-free on purpose so they can
// be extracted and run standalone in this project's synthetic test
// harnesses; the actual on/off gate is checked via
// $('#estimatebirthyearsonoffswitch').prop('checked') at the two call
// sites in buildForm() instead, matching how most other per-feature
// toggles are already read directly in buildform.js (e.g.
// birthonoffswitch). This global exists for consistency with the Deep
// Research pattern above and for any future use outside buildForm()'s
// own scope.
var estimateBirthYearsOn = true;
// #223/#224 follow-up: mirrors chrome.storage.local's
// 'familysearchplaces' key (see popup.js). This queries an
// UNAUTHENTICATED FamilySearch beta endpoint (apibeta.familysearch.org)
// that has no registered API key backing it and isn't a documented/
// sanctioned access path (see issue #224 - FamilySearch's own Solution
// Provider application, the only route to a real production key,
// explicitly rejects a project shaped like this one). FamilySearch's own
// docs describe this tier as an "older production data snapshot" whose
// "availability varies" - it could change or disappear without notice,
// so every call site built on this must degrade silently to the existing
// Google/raw-string fallback, never block location parsing on it.
// #229: DEFAULT true for a fresh install as of here, per explicit
// decision - FamilySearch's date-aware historical resolution
// outperformed Google's in live testing throughout #224/#227/#228, and
// Google requires a per-user paid API key to do anything at all, which
// FamilySearch doesn't. The risk above is real and unchanged - just
// judged worth it as the new default, not eliminated. Read directly via
// $('#familysearchplacesonoffswitch').prop('checked') at its one call
// site in parse-location.js's queryGeo(), matching how the Google geo
// toggle itself is read (geoqueryCheck()) rather than off this global -
// this global exists for consistency with the other feature-flag
// globals in this file.
var familysearchPlacesOn = true;
// #241: burial locations conventionally show the place name as it's
// known TODAY (useful for someone actually visiting the grave), not the
// historic name at time of death - the opposite of every other event,
// which deliberately resolves to the period-correct historical name (see
// familysearchPlacesOn's own comment). Off by default - an opt-in
// override, not a change to the base behavior. Only affects the
// FamilySearch lookup's query year for burial specifically; the source
// burial date and #232's own estimated burial date are untouched. Read
// directly via $('#burialcurrentlocationonoffswitch').prop('checked') at
// its call sites in buildform.js, same convention as
// familysearchPlacesOn above.
var burialCurrentLocationOn = false;
// #247: some source records fold the cemetery name into the death
// location instead of recording it separately as the burial location. Off
// by default - this rewrites already-scraped text (moves a detected
// cemetery segment from death to burial location), not just how a lookup
// is queried, so it's opt-in rather than an automatic default like
// #244's cemetery-abbreviation normalization. Read directly via
// $('#extractburialfromdeathonoffswitch').prop('checked') at its one call
// site in buildform.js's updateGeo(), same convention as
// burialCurrentLocationOn above.
var extractBurialFromDeathLocationOn = false;
// Run script as soon as the document's DOM is ready.
if (typeof String.prototype.startsWith != 'function') {
String.prototype.startsWith = function (str) {
if (typeof str === "undefined") {
return false;
}
return this.slice(0, str.length) == str;
}
}
if (typeof String.prototype.endsWith != 'function') {
String.prototype.endsWith = function (str) {
if (typeof str === "undefined") {
return false;
}
return this.substring(this.length - str.length, this.length) === str;
}
}
if (!String.prototype.contains) {
String.prototype.contains = function () {
return String.prototype.indexOf.apply(this, arguments) !== -1;
}
}
function exists(object) {
return (typeof object !== "undefined" && object !== null);
}
// #211: chrome.i18n.getMessage() has no fallback option of its own - if the
// browser's active locale (e.g. es/fi/he, all badly incomplete as of this
// writing) is missing a key, it returns "" rather than falling back to
// manifest.json's declared default_locale ("en"), even though Chrome does
// use default_locale when NO locale folder at all matches the browser's
// language. That fallback only ever applies at the whole-locale level, never
// per-key. There's also no API to ask chrome.i18n for a specific locale's
// text at runtime - the only way to get a per-key fallback is a JS-side copy
// of the English text this wrapper can reach for when Chrome's own lookup
// comes back empty. EN_FALLBACK_MESSAGES (locale_fallback_en.js, loaded
// before this file) is that copy, generated from the real
// _locales/en/messages.json by scripts/generate-locale-fallback.js rather
// than hand-duplicated.
//
// This is now the one canonical _() - it used to be defined identically
// (a bare chrome.i18n.getMessage() passthrough, no fallback) in popup.js,
// research.js, and content.js separately; centralized here since all three
// contexts already load shared.js.
function _(messageName, substitutions) {
var result = chrome.i18n.getMessage(messageName, substitutions);
if (result !== "") {
return result;
}
if (typeof EN_FALLBACK_MESSAGES === "undefined" || !EN_FALLBACK_MESSAGES.hasOwnProperty(messageName)) {
return result;
}
return applyLocaleFallbackSubstitutions(EN_FALLBACK_MESSAGES[messageName], substitutions);
}
// Replays chrome.i18n's own message-substitution algorithm
// (https://developer.chrome.com/docs/extensions/reference/api/i18n#placeholders)
// against a raw messages.json entry, since the fallback path bypasses
// chrome.i18n.getMessage() entirely (that's the whole point - it's the one
// that just returned "") and so never gets Chrome's own substitution
// handling for free.
function applyLocaleFallbackSubstitutions(entry, substitutions) {
var message = entry.message;
var placeholders = entry.placeholders || {};
// $placeholderName$ -> the placeholder's own "content" template (e.g.
// "$1") - matched case-insensitively per Chrome's spec, hence the
// lowercase lookup against placeholder keys (which the real
// messages.json always defines in lowercase).
message = message.replace(/\$([A-Za-z0-9_@]+)\$/g, function (match, name) {
var key = name.toLowerCase();
return placeholders.hasOwnProperty(key) ? placeholders[key].content : match;
});
// $1, $2, ... -> the caller's substitutions. Chrome's own API accepts
// either a single string (treated as just $1) or an array - matched
// here for parity.
var subs = Array.isArray(substitutions) ? substitutions : (exists(substitutions) ? [substitutions] : []);
message = message.replace(/\$(\d+)/g, function (match, num) {
var index = parseInt(num, 10) - 1;
return exists(subs[index]) ? subs[index] : match;
});
// $$ -> a literal $ - Chrome's escape mechanism. Applied last so it
// can't interfere with the $name$/$digit patterns matched above.
message = message.replace(/\$\$/g, "$");
return message;
}
function isValidDate(d) {
return d instanceof Date && !isNaN(d);
}
// #212: moved here from popup.js (#223/#224 follow-up) - also needed by
// extractDateYear() below, not just datesAreEquivalent() (popup.js).
// Optional trailing "the " handles a real case confirmed live - "After
// the 1st September 1919" - a qualifier followed by an article before the
// date itself, not just "After 1919".
var DATE_QUALIFIER_PATTERN = /^(circa|about|after|before)\s+(the\s+)?/i;
// #223/#224 follow-up: live-confirmed getGeoDedupKey()/FamilySearch's date
// query both silently lost the year entirely for any qualified date
// ("After 20 Jan 1891") - moment() requires the ENTIRE string to match one
// of dateformatter's formats, and none of them account for a leading
// qualifier word at all, so a real, parseable date failed outright just
// because of the "After " prefix. Strips the qualifier (reusing the exact
// pattern datesAreEquivalent() already relies on for the same reason) and
// the ordinal suffix ("1st" -> "1") before handing off to moment - the
// single shared place both getGeoDedupKey() and FamilySearch's
// queryFamilySearchPlaces() (parse-location.js) now get a year from,
// instead of two separate inline copies of the same moment() call.
function extractDateYear(dateval) {
if (!exists(dateval) || dateval === "") {
return undefined;
}
var stripped = dateval.replace(DATE_QUALIFIER_PATTERN, "").replace(/(\d+)(st|nd|rd|th)\b/i, "$1").trim();
var dt = moment(stripped, getDateFormat(stripped));
if (dt.isValid() && !isNaN(dt.get('year'))) {
return dt.get('year');
}
return undefined;
}
// #223/#224 follow-up: previously deduped purely by the raw location
// STRING, harmless for Google (date-blind - two events sharing the same
// place string always got the identical result regardless of processing
// order) but a real correctness bug once FamilySearch Places entered the
// picture: it resolves the SAME place string to DIFFERENT, historically
// correct entities depending on the associated date (e.g. a birth vs a
// death decades apart at the same place name). Live-confirmed: a death
// location was silently getting the geocoded result computed for a
// different, earlier event that happened to share the identical raw place
// string - whichever one was processed first "won" the unique slot, and
// every later duplicate just copied its result wholesale (see the
// geocleanup/geounique matching loop in updateFamily(), buildform.js).
// Folding the event's own year into the dedup key fixes this for both
// sources - the common Google-only case is unaffected, this only costs a
// handful of extra (still fully correct) redundant lookups in the rare
// same-string-different-year edge case.
//
// Reads .dateForFsLookup in preference to .date (see
// attachDateForFsLookup() below) - live-confirmed at least one parser
// (collections/onlineofb.js) never puts an event's date and location on
// the same array element at all, so .date alone is undefined here far
// more often than expected. Lives here (not buildform.js) as a general-
// purpose, reusable utility - depends only on exists()/moment/
// getDateFormat(), all globally available by the time this actually runs,
// regardless of script load order (function declarations resolve their
// references at call time, not at parse time).
function getGeoDedupKey(entry) {
var dateval = exists(entry.dateForFsLookup) ? entry.dateForFsLookup : entry.date;
var year = extractDateYear(dateval);
return entry.location + "|" + (exists(year) ? year : "");
}
// Live-confirmed bug (issue #224): collections/onlineofb.js pushes an
// event's date and location as TWO SEPARATE array elements
// (data.push({date:...}) then data.push({id:geoid, location:...})), never
// merged onto one object - so a location-bearing element's OWN .date is
// undefined even when the event genuinely has a real date sitting right
// next to it in the same array. FamilySearch's date-aware lookup (and the
// dedup key above) need SOME associated year; this scans the same
// event-type array (there's normally at most one real date per event) for
// the first element that actually has one, and attaches it as a NEW field
// - deliberately not named/aliased as .date itself, so nothing else that
// already reads .date (rendering, the 95-year check, etc.) is affected by
// this at all.
// fallbackobj (optional): requested live - burial events frequently have
// no date of their own at all anywhere in their own array, only a
// location. Assuming burial happened the same year as death is a
// reasonable genealogical default (rather than leaving the FamilySearch
// lookup entirely date-blind, or wrongly reusing some unrelated event's
// year via dedup collision) - callers pass the person's own "death" array
// as fallbackobj specifically for a "burial" location; omitted entirely
// for every other event type.
function attachDateForFsLookup(memberobj, locationEntry, fallbackobj) {
if (exists(locationEntry.date) && locationEntry.date !== "") {
return; // already has its own real date, nothing to borrow
}
for (var i in memberobj) if (memberobj.hasOwnProperty(i)) {
if (exists(memberobj[i].date) && memberobj[i].date !== "") {
locationEntry.dateForFsLookup = memberobj[i].date;
return;
}
}
if (exists(fallbackobj)) {
for (var j in fallbackobj) if (fallbackobj.hasOwnProperty(j)) {
if (exists(fallbackobj[j].date) && fallbackobj[j].date !== "") {
locationEntry.dateForFsLookup = fallbackobj[j].date;
return;
}
}
}
}
// #224: last-resort tier of the same chain, applied after
// attachDateForFsLookup() above has already had its shot (own date ->
// sibling scrape -> burial-borrows-scraped-death) and come up empty -
// fills in a year resolved from Geni's existing data or a genealogical
// ballpark heuristic (see resolveFsLookupYears() in buildform.js: birth ->
// #208's estimated year, marriage -> birth+30, burial -> death year).
// Never overwrites anything attachDateForFsLookup() already set - only
// used to scope the FamilySearch date filter, never written back to the
// form, so an approximate year here is fine; landing in the right decade
// is enough to prefer the correct historical jurisdiction over a modern
// one.
function applyFsLookupYearFallback(locationEntry, year) {
if (exists(year) &&
(!exists(locationEntry.dateForFsLookup) || locationEntry.dateForFsLookup === "") &&
(!exists(locationEntry.date) || locationEntry.date === "")) {
locationEntry.dateForFsLookup = String(year);
}
}
// #224: strips German jurisdiction-level abbreviations/words ("Kr."/
// "Kreis", "Landkreis", "Amt", "Bezirk", "Regierungsbezirk", "Provinz",
// "Königreich", "Grafschaft") and parenthetical qualifiers (e.g. "(Mark)"
// in "Storkow (Mark)") before comparing a raw scraped segment against a
// resolved city/county/state/country value - the two rarely match as
// exact strings even when they clearly refer to the same place ("Kr.
// Beeskow-Storkow" vs a resolved county of "Beeskow-Storkow"). Not an
// exhaustive list - built from the actual qualifier words seen live on
// 19th-century Prussian-era records, the case this feature was built
// against; add to it as further cases turn up.
var PLACE_SEGMENT_QUALIFIER_PATTERN = /\((?:[^)]*)\)|\b(kr\.?|kreis|landkreis|amt|bez\.?|bezirk|regierungsbezirk|provinz|province|königreich|koenigreich|grafschaft)\b/gi;
// #253 follow-up (found while investigating a live-reported leftover-text
// bug): the old /[^\w\s]/g punctuation strip is ASCII-only - \w never
// matches an accented letter, so "México" got mangled to "m xico" (a
// stray space where the "é" was), silently breaking a match against a
// resolved "Mexico" field. Two-part fix: NFD-normalize and strip
// combining marks so an accented letter folds to its plain-ASCII base
// ("México" -> "Mexico") BEFORE the punctuation strip - live-confirmed
// this exact mismatch (FamilySearch returns the unaccented "Mexico", the
// scraped source text has "México") would otherwise leave a real match
// undetected. Unicode property escapes (\p{L}/\p{N}) then keep any
// remaining real letter/digit in any script the fold didn't cover, only
// stripping actual punctuation.
// #260 follow-up (live-reported): common country abbreviations share NO
// substring with FamilySearch/Google's own resolved full name ("USA" vs
// "United States"), so a raw segment like "Price Hill, Hamilton County,
// Ohio, USA" left "USA" surviving as leftover text - previously invisible
// (hidden behind the geoicon toggle), now shown by default since #260,
// which is what actually surfaced this as a real, visible bug. Exact
// whole-segment lookup only (never a substring replace), so this can't
// corrupt an unrelated word that merely contains "us" (e.g. "Russia").
// Not an exhaustive list - built from live-confirmed cases; extend as
// further ones turn up, matching this project's usual evidence-based
// discipline for this kind of table.
// #260 follow-up: the country-abbreviation gap above turned out to be
// inconsistent for US state abbreviations too, not just missing outright
// - "IL" happens to already match a resolved "Illinois" by sheer
// substring coincidence ("illinois" starts with "il"), but "NY" does NOT
// match "New York" (no "ny" substring exists once the space is in the
// way), confirmed directly. Rather than leave it working for some states
// and not others depending on spelling luck, the standard USPS
// two-letter codes are unambiguous, standardized data (not a guessed
// translation) - safe to include in full rather than wait for a report
// naming each individual unlucky state one at a time.
var PLACE_SEGMENT_EQUIVALENTS = {
"usa": "united states",
"us": "united states",
"united states of america": "united states",
"al": "alabama", "ak": "alaska", "az": "arizona", "ar": "arkansas",
"ca": "california", "co": "colorado", "ct": "connecticut", "de": "delaware",
"dc": "district of columbia", "fl": "florida", "ga": "georgia", "hi": "hawaii",
"id": "idaho", "il": "illinois", "in": "indiana", "ia": "iowa",
"ks": "kansas", "ky": "kentucky", "la": "louisiana", "me": "maine",
"md": "maryland", "ma": "massachusetts", "mi": "michigan", "mn": "minnesota",
"ms": "mississippi", "mo": "missouri", "mt": "montana", "ne": "nebraska",
"nv": "nevada", "nh": "new hampshire", "nj": "new jersey", "nm": "new mexico",
"ny": "new york", "nc": "north carolina", "nd": "north dakota", "oh": "ohio",
"ok": "oklahoma", "or": "oregon", "pa": "pennsylvania", "ri": "rhode island",
"sc": "south carolina", "sd": "south dakota", "tn": "tennessee", "tx": "texas",
"ut": "utah", "vt": "vermont", "va": "virginia", "wa": "washington",
"wv": "west virginia", "wi": "wisconsin", "wy": "wyoming", "pr": "puerto rico"
};
// #280/#282 (live-reported, DanCornett): expands a burial-venue
// abbreviation to its full word - "Cem"/"Cem."->"Cemetery" (also the
// misspelled "Cemetary"), "Mem."/"Mem" immediately before "Garden(s)"/
// "Park"->"Memorial" (a bare "Mem" alone is too ambiguous to expand
// unconditionally, matching the same discipline already applied to a
// bare "Cem" - see PLACE_NAME_KEYWORD_PATTERN's own comment,
// parse-location.js). The negative lookahead on "cem" prevents matching
// its own prefix inside the word "Cemetery" once already expanded (or
// already spelled that way to begin with) - without it, "Cemetery" would
// corrupt into "Cemeteryetery" on a second pass. Shared between
// normalizeCemeteryAbbreviation() (parse-location.js, builds the Place
// Name field's display text and pre-splits the raw location string
// before segmenting it for the FamilySearch query) and this file's own
// normalizePlaceSegmentForMatch() (decides whether a raw segment already
// duplicates a resolved field) - without sharing this, an abbreviated
// raw segment like "Liberty Cem" was never recognized as the SAME thing
// as an already-expanded "Liberty Cemetery" field value, so it survived
// into computeLeftoverPlaceName()'s leftover text as a spurious near-
// duplicate.
function expandBurialVenueAbbreviation(text) {
var expanded = String(text || "").replace(/\bcemetary\b/i, "Cemetery");
expanded = expanded.replace(/\bmem\.?(\s+(?:garden|park))/i, "Memorial$1");
return expanded.replace(/\bcem\.?(?![a-z])/i, "Cemetery");
}
function normalizePlaceSegmentForMatch(text) {
var normalized = expandBurialVenueAbbreviation(String(text || ""))
.replace(PLACE_SEGMENT_QUALIFIER_PATTERN, " ")
.normalize("NFD").replace(/[\u0300-\u036f]/g, "")
.replace(/[^\p{L}\p{N}\s]/gu, " ")
.replace(/\s+/g, " ")
.trim()
.toLowerCase();
return PLACE_SEGMENT_EQUIVALENTS.hasOwnProperty(normalized) ? PLACE_SEGMENT_EQUIVALENTS[normalized] : normalized;
}
// #224: true when segment is either empty after normalizing (nothing but
// a qualifier word/parenthetical - no real content to preserve) or is an
// exact match (after normalizing - accent-folding, qualifier-stripping,
// and the abbreviation/equivalence lookup all still apply, e.g. "IL"
// against a resolved "Illinois") of one of the already-resolved fields.
// Checks each field's own individual comma-parts separately (a field can
// itself be compound, e.g. geo.city "Price Hill, Cincinnati" - a
// neighborhood folded into its enclosing city, #234) rather than the
// field as one whole string - so a plain segment naming just one part
// ("Cincinnati" alone) still correctly matches.
// #249 (live-reported, DanCornett): this used to also accept a substring
// match either direction, which correctly caught minor wording
// differences ("Storkow (Mark)" vs resolved "Storkow" - though that
// specific case is actually handled by the qualifier-pattern stripping
// above, not this) but ALSO wrongly discarded an entire segment that
// merely CONTAINED a resolved field name glued onto genuine extra prefix
// text with no comma to split on - "at the crossroads 1 mile south of
// Springfield" (City: "Springfield") was being thrown away whole instead
// of kept as real, useful leftover context. DanCornett's own call:
// simpler and safer to keep the whole segment intact (accepting
// "Springfield" appearing twice, once in City and once here) than to try
// to algorithmically split out just the redundant part - the string-
// position heuristics considered for that were flagged as easy to get
// subtly wrong on edge cases.
// #249 follow-up (live-confirmed regression against #260's own existing
// tests): a raw segment often carries a "County"/"Parish" suffix
// ("Hamilton County", "Kings County") that a resolved field sometimes
// doesn't (a bare Google Geocoding county, e.g. "Hamilton") and
// sometimes does (FamilySearch's own results - see the " County"/"
// Parish" suffix logic in familySearchPlaceToGeoLocation(),
// parse-location.js) - needs to match regardless of which side, if
// either, carries the suffix. Stripped only from the very END of the
// string (never mid-word, unlike PLACE_SEGMENT_QUALIFIER_PATTERN above)
// since that's the only place this specific suffix convention actually
// appears - a targeted, evidence-based exception, not a reversion to
// the general substring matching #249's fix above just removed.
function stripTrailingCountySuffix(normalizedText) {
return normalizedText.replace(/\s+(county|parish)$/, "");
}
// #270 (live-reported, DanCornett): FamilySearch resolves some US records
// to a historically-accurate predecessor country name instead of "United
// States" - "Republic of Texas" (the 1836-1845 independence era) and
// "British Colonial America" (pre-1776) are the two this codebase already
// tracks (FS_US_COUNTY_SUFFIX_COUNTRIES, parse-location.js - kept as a
// separate literal list here rather than a cross-file reference, since
// shared.js loads before parse-location.js and nothing else here depends
// on that file). A raw segment reading "USA"/"US" - how most source
// records are actually written, regardless of the event's own historical
// era - needs to be recognized as ALSO representing one of these, not
// just literal "United States", or it shows up as spurious leftover
// Place text right next to a correctly period-resolved Country field.
// DanCornett's own note: "this is going to be a VERY common situation."
// One-directional only (a segment normalizing to "united states" matches
// any of these) - no live-evidenced case yet of raw text literally
// spelling out "Republic of Texas" needing the reverse.
var US_HISTORICAL_COUNTRY_EQUIVALENTS = ["united states", "british colonial america", "republic of texas"];
function segmentMatchesAnyField(segment, fields) {
var normSeg = normalizePlaceSegmentForMatch(segment);
if (normSeg === "") {
return true;
}
var normSegBare = stripTrailingCountySuffix(normSeg);
for (var i = 0; i < fields.length; i++) {
if (!exists(fields[i]) || fields[i] === "") {
continue;
}
var fieldParts = fields[i].split(",");
for (var j = 0; j < fieldParts.length; j++) {
var normPart = normalizePlaceSegmentForMatch(fieldParts[j]);
if (normPart !== "" && (normSeg === normPart || normSegBare === stripTrailingCountySuffix(normPart) ||
(normSeg === "united states" && US_HISTORICAL_COUNTRY_EQUIVALENTS.indexOf(normPart) !== -1))) {
return true;
}
}
}
return false;
}
// #224: live-reported - once city/county/state/country resolve to real
// values, the RAW scraped string (Geni's "location_string"/Place Name
// field) still gets suggested as-is, duplicating the exact same
// information Geni's display then shows twice over ("Storkow (Mark), Kr.
// Beeskow-Storkow, Potsdam, Brandenburg, Preussen, Storkow, Beeskow-
// Storkow, Brandenburg, Germany"). This computes what's actually LEFT
// OVER after removing every raw segment that's already represented in the
// resolved geo fields - e.g. "Potsdam" (a jurisdiction level that gets
// dropped during the 5-levels-into-4-fields mapping, see
// familySearchPlaceToGeoLocation()'s own comment) or "Preussen" (a
// historical name that won't match a modern-day resolved country like
// "Germany") legitimately survive as real, non-redundant context; "Storkow
// (Mark)"/"Kr. Beeskow-Storkow"/"Brandenburg" don't, since they're already
// captured by city/county/state. Returns "" when nothing is left over (the
// common case for a location that resolved cleanly) - never returns the
// raw string unchanged, and never fires at all unless the caller already
// confirmed real geo fields exist (see buildform.js's hasGeoFields).
function computeLeftoverPlaceName(rawLocation, geo) {
if (!exists(rawLocation) || rawLocation.trim() === "" || !exists(geo)) {
return "";
}
var segments = rawLocation.split(",").map(function (s) { return s.trim(); }).filter(function (s) { return s !== ""; });
var fields = [geo.place, geo.city, geo.county, geo.state, geo.country];
var leftover = segments.filter(function (seg) { return !segmentMatchesAnyField(seg, fields); });
return leftover.join(", ");
}
// #260 follow-up (live-reported, DanCornett - screenshots confirmed the
// exact symptom): buildform.js's visible-by-default "Place: " row
// (title:location:place_name_geo, fed by geo.place alone) only ever
// showed whatever extractPlaceNameSegments() stripped as a recognized
// venue keyword before searching - genuinely blank whenever nothing
// matched a keyword, even when there's real leftover text with nowhere
// else to go (e.g. "Sagrario" ahead of a resolved Xalapa/Veracruz/Mexico
// chain). That leftover WAS being computed correctly all along by
// computeLeftoverPlaceName() above - it just fed a DIFFERENT row
// (title:location:place_name, "Baptism Place:") that's hidden by default
// behind the geoicon toggle whenever real geo fields exist, which
// DanCornett was never interacting with. Combines both into the one
// value users actually see by default: the stripped venue (if any)
// first, then any additional leftover residue. Safe to concatenate
// without ever duplicating text - computeLeftoverPlaceName() already
// excludes anything matching geo.place from its own output (geo.place is
// one of its own comparison fields).
function computeCombinedPlaceValue(rawLocation, geo) {
var parts = [geo.place, computeLeftoverPlaceName(rawLocation, geo)].filter(function (v) {
return exists(v) && v !== "";
});
return parts.join(", ");
}
// #248 follow-up (live-reported: "Adath Israel Price Hill, Cincinnati,
// United States" showed "Price Hill" duplicated into Place even though
// it's already part of City). Root cause: extractPlaceNameSegments()
// (parse-location.js) strips a whole comma-segment as venue text BEFORE
// the FamilySearch query ever runs - it has no way to know FamilySearch
// will later independently resolve PART of that same glued segment as a
// real jurisdiction (here, "Price Hill" folded into City alongside its
// enclosing "Cincinnati", per the #234 neighborhood-in-a-city fix).
// segmentMatchesAnyField() above only compares a whole raw segment
// against a whole field value, so it can't catch a field's value glued
// onto extra text with no comma to split on. This instead checks the
// TRAILING words of placeText against each individual comma-part of the
// resolved fields (city/county/state/country can themselves contain an
// internal comma, e.g. "Price Hill, Cincinnati") and strips a trailing
// run of words that exactly matches one, leaving genuine venue text
// ("Adath Israel") intact. Only ever REMOVES a redundant trailing
// fragment, never adds anything - unlike the computeLeftoverPlaceName()
// approach tried and reverted at this same call site before (see
// familySearchPlaceToGeoLocation()'s own comment), which failed because
// it injected extra "leftover residue" text into a field that
// auto-submits. Trailing only - no live-confirmed case yet of the same
// duplication happening as a leading run instead. A placeText that is
// ENTIRELY the redundant fragment correctly strips down to "" - same as
// computeLeftoverPlaceName()'s own "nothing left over" case above, not a
// special case to guard against.
function stripRedundantPlaceSuffix(placeText, geo) {
if (!exists(placeText) || placeText.trim() === "" || !exists(geo)) {
return placeText;
}
var fieldParts = [];
[geo.city, geo.county, geo.state, geo.country].forEach(function (field) {
if (exists(field) && field !== "") {
field.split(",").forEach(function (part) {
part = part.trim();
if (part !== "") {
fieldParts.push(part);
}
});
}
});
fieldParts.sort(function (a, b) {
return b.trim().split(/\s+/).length - a.trim().split(/\s+/).length;
});
var words = placeText.trim().split(/\s+/);
for (var i = 0; i < fieldParts.length; i++) {
var partWords = fieldParts[i].trim().split(/\s+/);
if (words.length >= partWords.length) {
var trailing = words.slice(words.length - partWords.length).join(" ");
if (normalizePlaceSegmentForMatch(trailing) === normalizePlaceSegmentForMatch(fieldParts[i])) {
words = words.slice(0, words.length - partWords.length);
break;
}
}
}
return words.join(" ").trim();
}
function startsWithHTTP(url, match) {
//remove protocol and comapre
url = url.replace("https://", "").replace("http://", "");
match = match.replace("https://", "").replace("http://", "");
return url.startsWith(match);
}
function isGeni(url) {
return (startsWithHTTP(url,"http://www.geni.com/people") || startsWithHTTP(url,"http://www.geni.com/family-tree") || startsWithHTTP(url,"http://www.geni.com/profile"));
}
function isGeniProject(url) {
return startsWithHTTP(url,"http://www.geni.com/projects")
}
function getProject(project_id) {
return project_id.substring(project_id.lastIndexOf('/') + 1).replace("#", "");
}
function getProfile(profile_id) {
//Gets the profile id from the Geni URL
if (profile_id.length > 0) {
var startid = profile_id.toLowerCase();
profile_id = decodeURIComponent(profile_id).trim();
if (profile_id.indexOf("&resolve=") != -1) {
profile_id = profile_id.substring(profile_id.lastIndexOf('#') + 1);
}
if (profile_id.indexOf("profile-") != -1) {
profile_id = profile_id.substring(profile_id.lastIndexOf('/') + 1);
}
if (profile_id.indexOf("#/tab") != -1) {
profile_id = profile_id.substring(0, profile_id.lastIndexOf('#/tab'));
}
if (profile_id.indexOf("/") != -1) {
//Grab the GUID from a URL
profile_id = profile_id.substring(profile_id.lastIndexOf('/') + 1);
}
if (profile_id.indexOf("?through") != -1) {
//In case the copy the profile url by navigating through another 6000000002107278790?through=6000000010985379345
//But skip 6000000029660962822?highlight_id=6000000029660962822#6000000028974729472
profile_id = "profile-g" + profile_id.substring(0, profile_id.lastIndexOf('?'));
}
if (profile_id.indexOf("?from_flash") != -1) {
profile_id = "profile-g" + profile_id.substring(0, profile_id.lastIndexOf('?'));
}
if (profile_id.indexOf("?highlight_id") != -1) {
profile_id = "profile-g" + profile_id.substring(profile_id.lastIndexOf('=') + 1, profile_id.length);
}
if (profile_id.indexOf("#") != -1) {
//In case the copy the profile url by navigating in tree view 6000000001495436722#6000000010985379345
if (profile_id.contains("html5")) {
profile_id = "profile-" + profile_id.substring(0, profile_id.lastIndexOf('#'));
} else {
profile_id = "profile-g" + profile_id.substring(0, profile_id.lastIndexOf('#'));
}
}
var isnum = /^\d+$/.test(profile_id);
if (isnum) {
if (profile_id.length > 16) {
profile_id = "profile-g" + profile_id;
} else if (startid.contains("www.geni.com/people") || startid.contains("www.geni.com/family-tree")) {
profile_id = "profile-g" + profile_id;
} else {
profile_id = "profile-" + profile_id;
}
}
var validate = profile_id.replace("profile-g", "").replace("profile-", "").replace("#","");
if (isNaN(validate)) {
profile_id = "";
}
if (profile_id.indexOf("profile-") != -1 && profile_id !== "profile-g") {
return "?profile=" + profile_id;
} else if (tablink !== "https://www.geni.com/family-tree") {
console.log("Profile ID not detected: " + startid);
console.log("URL: " + tablink);
return "";
}
}
return "";
}
function GeniPerson(obj) {
this.person = obj;
this.get = function (path, subpath) {
var obj = this.person;
if (path === "name_language") {
if (obj["names"] === undefined) {
return "en-US"
} else {
for (lang in obj["names"]) {
if (Object.keys(obj["names"][lang]).length > 0) {
return lang
}
}
}
}
if (path === "names" && subpath !== undefined && obj[path] === undefined && subpath.substring(0,5) === "en-US") {
// names object only exists if there is more than one language on the profile
path = subpath.substring(6,subpath.length)
subpath = undefined
}
if (path == "photo_urls") {
if (checkNested(this.person,"photo_urls", "medium")) {
return this.person["photo_urls"].medium;
} else {
return geniPhoto(this.person.gender);
}
} else if (!obj.hasOwnProperty(path)) {
return "";
} else if (!exists(subpath)) {
if (typeof obj[path] === 'string' || obj[path] instanceof String) {
obj[path] = obj[path].replace(/"/g, """);
} else {
for (var i = 0; i < obj[path].length; i++) {
obj[path][i] = obj[path][i].replace(/"/g, """);
}
}
return obj[path];
} else {
obj = obj[path];
if (subpath === "location_string" && exists(obj.location) && exists(obj.location.formatted_location)) {
subpath = "location.formatted_location";
}
if (subpath === "date.formatted_date" && typeof obj["date"] === 'string') {
subpath = "date";
}
var args = subpath.split(".");
for (var i = 0; i < args.length; i++) {
if (!obj || !obj.hasOwnProperty(args[i])) {
return "";
}
obj = obj[args[i]];
}
return obj;
}
};
this.set = function (path, data) {
this.person[path] = data;
};
this.isLocked = function (path, subpath) {
var obj = this.person;
if (!obj.hasOwnProperty("locked_fields")) {
return false;
}
obj = obj["locked_fields"];
if (!obj.hasOwnProperty(path)) {
return false;
} else if (!exists(subpath)) {
return obj[path];
} else {
obj = obj[path];
var args = subpath.split(".");
for (var i = 0; i < args.length; i++) {
if (!obj || !obj.hasOwnProperty(args[i])) {
return false;
}
obj = obj[args[i]];
}
return obj;
}
};
this.lockIcon = function(path, subpath) {
if (this.isLocked(path, subpath)) {
return "lock.png";
} else {
return "right.png";
}
};
}
function isFemale(title) {
if (!exists(title)) { return false; }
title = title.toLowerCase().replace(" (implied)", "");
return (title === "wife" || title === "ex-wife" || title === "mother" || title === "sister" || title === "daughter" || title === "female" || title === "f");
}
function isMale(title) {
if (!exists(title)) { return false; }
title = title.toLowerCase().replace(" (implied)", "");
return (title === "husband" || title === "ex-husband" || title === "father" || title === "brother" || title === "son" || title === "male" || title === "m");
}
// #284 (live-reported, DanCornett): "half brother"/"half sister"/"half
// sibling" (hyphenated or not) added - a source explicitly labeling the
// relationship this way previously matched none of the exact strings
// here at all, falling through to the unrecognized-relationship branch
// entirely rather than being included as a sibling. buildForm() (see its
// own #284 comment) separately infers halfsibling=true from this same
// "half" wording, so recognizing the label here is what lets that
// member be included as a sibling in the first place.
//
// #304 follow-up (live-reported, DanCornett - a tree copy with "some
// siblings and step siblings" silently failed to create a few of them,
// with a flashed "no first/last name" error): familysearchjson.js infers
// half/step-siblings purely from whether a child shares the focus
// person's own coupleId (FamilySearch's data has no separate literal
// "step" concept here - a different second parent reads as "half" either
// way) and uses that AS the relationship/grouping key itself - the single
// concatenated word "halfsibling", never with a space or hyphen. That
// never matched any of the strings above, so every FamilySearch half- (or
// step-, by this same inference) sibling fell through to the "Unknown"
// relationship bucket, where the top-level checkbox is disabled by design
// pending a manual relationship pick (buildform.js) - which a bulk
// "add everything" action doesn't stop to do, so it reached buildTree()
// with no relationship or name ever resolved, correctly tripping the
// existing #300 guard rather than creating a blank profile.
function isSibling(relationship) {
if (!exists(relationship)) { return false; }
relationship = relationship.toLowerCase().replace(" (implied)", "").replace("half-", "half ");
return (relationship === "siblings" || relationship === "sibling" || relationship === "brother" || relationship === "sister" || relationship === "bro" || relationship === "sis" ||
relationship === "half brother" || relationship === "half sister" || relationship === "half sibling" || relationship === "half siblings" ||
relationship === "halfsibling" || relationship === "halfsiblings");
}
function isChild(relationship) {
if (!exists(relationship)) { return false; }
relationship = relationship.toLowerCase().replace(" (implied)", "");
return (relationship === "children" || relationship === "child" || relationship === "son" || relationship === "daughter" || relationship === "dau");
}
function isParent(relationship) {
if (!exists(relationship)) { return false; }
relationship = relationship.toLowerCase().replace(" (implied)", "");
return (relationship === "parents" || relationship === "father" || relationship === "mother" || relationship === "parent" || relationship === "moth" || relationship === "fath");
}
function isPartner(relationship) {
if (!exists(relationship)) { return false; }
relationship = relationship.toLowerCase().replace(" (implied)", "");
return (relationship === "spouse" || relationship === "wife" || relationship === "husband" || relationship === "partner" || relationship === "ex-husband" || relationship === "ex-wife" || relationship === "ex-partner" || relationship === "ex_husband" || relationship === "ex_wife" || relationship === "ex_partner" || relationship === "spouses");
}
function getGeniData(profile, value, subvalue) {
if (profile === "add") {
if (value === "photo_urls") {
return geniPhoto('unknown');
}
return "";
}
var person = genifamilydata[profile];
if (!exists(person)) {
return "";
}
return person.get(value, subvalue);
}
function getUnionData(union, value) {
if (exists(union[value])) {
return union[value];
} else {
return "";
}
}
function getFocus() {
return genifamily["focus"].id;
}
function getParents() {
var familyset = [];
var focusid = getFocus();
var focus = getGeniData(focusid, "edges");
for (var union in focus) {
if (!focus.hasOwnProperty(union)) continue;
if (isChild(focus[union].rel) && !exists(focus[union].rel_modifier)) {
var edges = uniondata[union]["edges"];
for (var profile in edges) {
if (!edges.hasOwnProperty(profile)) continue;
if (isPartner(edges[profile].rel)) {
var person = genifamilydata[profile];
person.set("relation", getRelationship("parent", person.get("gender")));
person.set("union", getUnionData(uniondata[union], "id"));
person.set("status", getUnionData(uniondata[union], "status"));
if ("marriage" in uniondata[union]) {
person.set("marriage", uniondata[union]["marriage"]);
}
if ("divorce" in uniondata[union]) {
person.set("divorce", uniondata[union]["divorce"]);
}
familyset.push(profile);
}
}
}
}
return familyset;
}
function getParentSets(focus, parents) {
var focusedge = getGeniData(focus, "edges");
var parentset = {};
for (var union in focusedge) {
if (!focusedge.hasOwnProperty(union)) continue;
if (isChild(focusedge[union].rel)) {
for (var i=0; i < parents.length; i++) {
var parentedge = getGeniData(parents[i], "edges");
for (var punion in parentedge) {
if (!parentedge.hasOwnProperty(punion)) continue;
if (punion === union) {
if (!exists(parentset[union])) {
parentset[union] = [];
}
if (parentset[union].indexOf(parents[i]) == -1) {
parentset[union].push(parents[i]);
}
}
}
}
}
}
return parentset;
}
function getChildren(focusid, partner) {
var familyset = [];
var focus = getGeniData(focusid, "edges");
for (var union in focus) {
if (!focus.hasOwnProperty(union)) continue;
if (isPartner(focus[union].rel)) {
if (!exists(uniondata[union])) {
return familyset;
}
var edges = uniondata[union]["edges"];
var loopedges = false;
if (exists(partner) && partner in edges) {
loopedges = true;
} else if (!exists(partner)) {
loopedges = true;
}
if (loopedges) {
for (var profile in edges) {
if (!edges.hasOwnProperty(profile)) continue;
if (isChild(edges[profile].rel) && !exists(edges[profile].rel_modifier)) {
var person = genifamilydata[profile];
person.set("relation", getRelationship("child", person.get("gender")));
person.set("union", getUnionData(uniondata[union], "id"));
person.set("status", "");
familyset.push(profile);
}
}
}
}
}
return familyset;
}
function getSiblings() {
var familyset = [];
var focusid = getFocus();
var focus = getGeniData(focusid, "edges");
for (var union in focus) {
if (!focus.hasOwnProperty(union)) continue;
if (isChild(focus[union].rel) && !exists(focus[union].rel_modifier)) {
var edges = uniondata[union]["edges"];
for (var profile in edges) {
if (!edges.hasOwnProperty(profile)) continue;
if (isChild(edges[profile].rel) && profile !== focusid) {
var person = genifamilydata[profile];
person.set("relation", getRelationship("sibling", person.get("gender")));
person.set("union", getUnionData(uniondata[union], "id"));
person.set("status", "");
familyset.push(profile);
}
}
}
}
return familyset;
}
function getPartners() {
var familyset = [];
var focusid = getFocus();
var focus = getGeniData(focusid, "edges");
for (var union in focus) {
if (!focus.hasOwnProperty(union)) continue;
if (isPartner(focus[union].rel)) {
var edges = uniondata[union]["edges"];
for (var profile in edges) {
if (!edges.hasOwnProperty(profile)) continue;
if (isPartner(edges[profile].rel) && profile !== focusid) {
var person = genifamilydata[profile];
person.set("relation", getRelationship("partner", person.get("gender")));