fix: proper noun detection too aggressive + add ignore button
Proper noun heuristic: - Changed from matching any single capitalized word to requiring TWO OR MORE consecutive capitalized words (e.g. "Acme Corp") - Single capitalized words at sentence starts were causing massive false positives — every sentence starts with a capital letter - Minimum 5 characters total and 2 proper words required Ignore button: - Each PPI warning item now has an X (ignore) button alongside the + (add mapping) button - Ignored values are persisted to storage (ss_ignored_ppi) so they stay dismissed across page reloads - Ignored values are skipped in both pattern detection and proper noun detection - Clicking ignore removes the item from the warning and re-scans https://claude.ai/code/session_01SWSwDfMVij53bCTNSCLMwn
This commit is contained in:
@@ -220,6 +220,24 @@
|
|||||||
.ss-ps-add:hover { background: rgba(74, 222, 128, 0.15); }
|
.ss-ps-add:hover { background: rgba(74, 222, 128, 0.15); }
|
||||||
.ss-ps-add:disabled { border-color: #333; cursor: default; }
|
.ss-ps-add:disabled { border-color: #333; cursor: default; }
|
||||||
|
|
||||||
|
.ss-ps-ignore {
|
||||||
|
background: none;
|
||||||
|
border: 1px solid #6b7280;
|
||||||
|
border-radius: 4px;
|
||||||
|
color: #9ca3af;
|
||||||
|
font-size: 11px;
|
||||||
|
width: 24px;
|
||||||
|
height: 24px;
|
||||||
|
cursor: pointer;
|
||||||
|
display: flex;
|
||||||
|
align-items: center;
|
||||||
|
justify-content: center;
|
||||||
|
flex-shrink: 0;
|
||||||
|
padding: 0;
|
||||||
|
}
|
||||||
|
|
||||||
|
.ss-ps-ignore:hover { background: rgba(156, 163, 175, 0.15); color: #e5e7eb; }
|
||||||
|
|
||||||
/* Floating reveal mode indicator */
|
/* Floating reveal mode indicator */
|
||||||
.ss-reveal-badge {
|
.ss-reveal-badge {
|
||||||
position: fixed;
|
position: fixed;
|
||||||
|
|||||||
+57
-19
@@ -400,50 +400,44 @@
|
|||||||
/**
|
/**
|
||||||
* Detect proper nouns (potential names, company names, project names)
|
* Detect proper nouns (potential names, company names, project names)
|
||||||
* that aren't configured in identity. Uses capitalization heuristics:
|
* that aren't configured in identity. Uses capitalization heuristics:
|
||||||
* - Capitalized words not at the start of a sentence
|
* - ONLY multi-word capitalized sequences (e.g. "Acme Corp", "Project Atlas")
|
||||||
* - Multi-word capitalized sequences (e.g. "Acme Corp", "Project Atlas")
|
* - Single capitalized words are too noisy — every sentence starts with one
|
||||||
* - Filters out common English words and programming terms
|
* - Filters out common English words and programming terms
|
||||||
*/
|
*/
|
||||||
function detectProperNouns(text, configured) {
|
function detectProperNouns(text, configured) {
|
||||||
const findings = [];
|
const findings = [];
|
||||||
// Match capitalized words that aren't at the very start of the text
|
// Only match TWO OR MORE consecutive capitalized words
|
||||||
// and aren't after a period/newline (sentence start)
|
// Single capitalized words cause too many false positives
|
||||||
const re = /(?:^|[.!?\n]\s*)?([A-Z][a-z]{2,}(?:\s+[A-Z][a-z]{2,})*)/g;
|
const re = /\b([A-Z][a-z]{2,}(?:\s+[A-Z][a-z]{2,})+)\b/g;
|
||||||
let m;
|
let m;
|
||||||
|
|
||||||
while ((m = re.exec(text)) !== null) {
|
while ((m = re.exec(text)) !== null) {
|
||||||
const fullMatch = m[1];
|
const fullMatch = m[1];
|
||||||
if (!fullMatch) continue;
|
if (!fullMatch) continue;
|
||||||
|
|
||||||
// Check if this is at the start of a sentence
|
// Split into individual words and filter common ones
|
||||||
const before = text.slice(Math.max(0, m.index - 2), m.index);
|
|
||||||
const isSentenceStart = m.index === 0 || /[.!?\n]\s*$/.test(before);
|
|
||||||
|
|
||||||
// Split into individual words and check each
|
|
||||||
const words = fullMatch.split(/\s+/);
|
const words = fullMatch.split(/\s+/);
|
||||||
const properWords = words.filter(w =>
|
const properWords = words.filter(w =>
|
||||||
w.length >= 3 &&
|
w.length >= 3 &&
|
||||||
!COMMON_CAPITALIZED.has(w.toLowerCase()) &&
|
!COMMON_CAPITALIZED.has(w.toLowerCase()) &&
|
||||||
!configured.has(w.toLowerCase())
|
!configured.has(w.toLowerCase()) &&
|
||||||
|
!ignoredValues.has(w.toLowerCase())
|
||||||
);
|
);
|
||||||
|
|
||||||
if (properWords.length === 0) continue;
|
if (properWords.length < 2) continue; // need at least 2 proper words
|
||||||
|
|
||||||
// Single capitalized word at sentence start = likely not a proper noun
|
|
||||||
if (isSentenceStart && properWords.length === 1 && words.length === 1) continue;
|
|
||||||
|
|
||||||
// Multi-word capitalized sequence is likely a proper noun
|
|
||||||
// Single capitalized word mid-sentence is likely a proper noun
|
|
||||||
const value = properWords.join(' ');
|
const value = properWords.join(' ');
|
||||||
if (value.length >= 3 && !configured.has(value.toLowerCase())) {
|
if (value.length >= 5 && !configured.has(value.toLowerCase()) && !ignoredValues.has(value.toLowerCase())) {
|
||||||
findings.push({
|
findings.push({
|
||||||
name: 'Possible Name/Org',
|
name: 'Possible Name/Org',
|
||||||
value,
|
value,
|
||||||
hint: 'Capitalized word — could be a name, company, or project',
|
hint: 'Capitalized phrase — could be a name, company, or project',
|
||||||
category: 'name',
|
category: 'name',
|
||||||
});
|
});
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
// Deduplicate
|
// Deduplicate
|
||||||
const seen = new Set();
|
const seen = new Set();
|
||||||
@@ -477,6 +471,7 @@
|
|||||||
while ((m = pat.re.exec(text)) !== null) {
|
while ((m = pat.re.exec(text)) !== null) {
|
||||||
const val = m[0];
|
const val = m[0];
|
||||||
if (configured.has(val.toLowerCase())) continue;
|
if (configured.has(val.toLowerCase())) continue;
|
||||||
|
if (ignoredValues.has(val.toLowerCase())) continue;
|
||||||
if (pat.skip && pat.skip.test(val)) continue;
|
if (pat.skip && pat.skip.test(val)) continue;
|
||||||
findings.push({ name: pat.name, value: val, hint: pat.hint, category: pat.cat });
|
findings.push({ name: pat.name, value: val, hint: pat.hint, category: pat.cat });
|
||||||
}
|
}
|
||||||
@@ -630,6 +625,30 @@
|
|||||||
// ============================================================
|
// ============================================================
|
||||||
const sessionSubstitutions = new Map();
|
const sessionSubstitutions = new Map();
|
||||||
|
|
||||||
|
// Values the user has explicitly ignored via the "Ignore" button.
|
||||||
|
// Persisted to storage so they stay dismissed across page reloads.
|
||||||
|
const ignoredValues = new Set();
|
||||||
|
|
||||||
|
// Load ignored values from storage
|
||||||
|
(async () => {
|
||||||
|
const stored = await getStorageData('ss_ignored_ppi');
|
||||||
|
if (Array.isArray(stored)) {
|
||||||
|
for (const v of stored) ignoredValues.add(v.toLowerCase());
|
||||||
|
}
|
||||||
|
})();
|
||||||
|
|
||||||
|
function addIgnoredValue(value) {
|
||||||
|
ignoredValues.add(value.toLowerCase());
|
||||||
|
// Persist
|
||||||
|
getStorageData('ss_ignored_ppi').then(stored => {
|
||||||
|
const list = Array.isArray(stored) ? stored : [];
|
||||||
|
if (!list.includes(value.toLowerCase())) {
|
||||||
|
list.push(value.toLowerCase());
|
||||||
|
setStorageData('ss_ignored_ppi', list);
|
||||||
|
}
|
||||||
|
});
|
||||||
|
}
|
||||||
|
|
||||||
// ============================================================
|
// ============================================================
|
||||||
// Notify content script of substitutions (for badge + logging)
|
// Notify content script of substitutions (for badge + logging)
|
||||||
// ============================================================
|
// ============================================================
|
||||||
@@ -1632,6 +1651,7 @@
|
|||||||
${settings.autoAddDetected !== false
|
${settings.autoAddDetected !== false
|
||||||
? `<button class="ss-ps-add" data-real="${encodeURIComponent(w.value)}" data-fake="${encodeURIComponent(fake)}" data-cat="${w.category}" title="Add mapping: ${displayVal} → ${fake}">+</button>`
|
? `<button class="ss-ps-add" data-real="${encodeURIComponent(w.value)}" data-fake="${encodeURIComponent(fake)}" data-cat="${w.category}" title="Add mapping: ${displayVal} → ${fake}">+</button>`
|
||||||
: ''}
|
: ''}
|
||||||
|
<button class="ss-ps-ignore" data-value="${encodeURIComponent(w.value)}" title="Ignore — stop flagging this value">✕</button>
|
||||||
</div>`;
|
</div>`;
|
||||||
}).join('');
|
}).join('');
|
||||||
|
|
||||||
@@ -1694,6 +1714,24 @@
|
|||||||
btn.disabled = true;
|
btn.disabled = true;
|
||||||
});
|
});
|
||||||
});
|
});
|
||||||
|
|
||||||
|
// Ignore buttons
|
||||||
|
preSendWarningEl.querySelectorAll('.ss-ps-ignore').forEach(btn => {
|
||||||
|
btn.addEventListener('click', () => {
|
||||||
|
const value = decodeURIComponent(btn.dataset.value);
|
||||||
|
addIgnoredValue(value);
|
||||||
|
|
||||||
|
// Remove this item's row
|
||||||
|
const row = btn.closest('.ss-ps-item');
|
||||||
|
if (row) row.remove();
|
||||||
|
|
||||||
|
// Re-scan to update warning
|
||||||
|
if (inputEl) {
|
||||||
|
if (inputScanTimer) clearTimeout(inputScanTimer);
|
||||||
|
inputScanTimer = setTimeout(() => scanInputForPPI(inputEl), 150);
|
||||||
|
}
|
||||||
|
});
|
||||||
|
});
|
||||||
}
|
}
|
||||||
|
|
||||||
// Replace all occurrences of `real` with `fake` in an input or contenteditable element
|
// Replace all occurrences of `real` with `fake` in an input or contenteditable element
|
||||||
|
|||||||
@@ -218,16 +218,15 @@ const AutoDetect = {
|
|||||||
*/
|
*/
|
||||||
_detectProperNouns(text, configured) {
|
_detectProperNouns(text, configured) {
|
||||||
const findings = [];
|
const findings = [];
|
||||||
const re = /(?:^|[.!?\n]\s*)?([A-Z][a-z]{2,}(?:\s+[A-Z][a-z]{2,})*)/g;
|
// Only match TWO OR MORE consecutive capitalized words
|
||||||
|
// Single capitalized words cause too many false positives (sentence starts)
|
||||||
|
const re = /\b([A-Z][a-z]{2,}(?:\s+[A-Z][a-z]{2,})+)\b/g;
|
||||||
let m;
|
let m;
|
||||||
|
|
||||||
while ((m = re.exec(text)) !== null) {
|
while ((m = re.exec(text)) !== null) {
|
||||||
const fullMatch = m[1];
|
const fullMatch = m[1];
|
||||||
if (!fullMatch) continue;
|
if (!fullMatch) continue;
|
||||||
|
|
||||||
const before = text.slice(Math.max(0, m.index - 2), m.index);
|
|
||||||
const isSentenceStart = m.index === 0 || /[.!?\n]\s*$/.test(before);
|
|
||||||
|
|
||||||
const words = fullMatch.split(/\s+/);
|
const words = fullMatch.split(/\s+/);
|
||||||
const properWords = words.filter(w =>
|
const properWords = words.filter(w =>
|
||||||
w.length >= 3 &&
|
w.length >= 3 &&
|
||||||
@@ -235,15 +234,14 @@ const AutoDetect = {
|
|||||||
!configured.has(w.toLowerCase())
|
!configured.has(w.toLowerCase())
|
||||||
);
|
);
|
||||||
|
|
||||||
if (properWords.length === 0) continue;
|
if (properWords.length < 2) continue; // need at least 2 proper words
|
||||||
if (isSentenceStart && properWords.length === 1 && words.length === 1) continue;
|
|
||||||
|
|
||||||
const value = properWords.join(' ');
|
const value = properWords.join(' ');
|
||||||
if (value.length >= 3 && !configured.has(value.toLowerCase())) {
|
if (value.length >= 5 && !configured.has(value.toLowerCase())) {
|
||||||
findings.push({
|
findings.push({
|
||||||
name: 'Possible Name/Org',
|
name: 'Possible Name/Org',
|
||||||
value,
|
value,
|
||||||
hint: 'Capitalized word — could be a name, company, or project',
|
hint: 'Capitalized phrase — could be a name, company, or project',
|
||||||
category: 'name',
|
category: 'name',
|
||||||
});
|
});
|
||||||
}
|
}
|
||||||
|
|||||||
Reference in New Issue
Block a user