fix: proper noun detection too aggressive + add ignore button

Proper noun heuristic:
- Changed from matching any single capitalized word to requiring
  TWO OR MORE consecutive capitalized words (e.g. "Acme Corp")
- Single capitalized words at sentence starts were causing massive
  false positives — every sentence starts with a capital letter
- Minimum 5 characters total and 2 proper words required

Ignore button:
- Each PPI warning item now has an X (ignore) button alongside
  the + (add mapping) button
- Ignored values are persisted to storage (ss_ignored_ppi) so
  they stay dismissed across page reloads
- Ignored values are skipped in both pattern detection and
  proper noun detection
- Clicking ignore removes the item from the warning and re-scans

https://claude.ai/code/session_01SWSwDfMVij53bCTNSCLMwn
This commit is contained in:
Claude
2026-03-26 22:18:11 +00:00
parent ae451f73df
commit c25be011f7
3 changed files with 81 additions and 27 deletions
+18
View File
@@ -220,6 +220,24 @@
.ss-ps-add:hover { background: rgba(74, 222, 128, 0.15); } .ss-ps-add:hover { background: rgba(74, 222, 128, 0.15); }
.ss-ps-add:disabled { border-color: #333; cursor: default; } .ss-ps-add:disabled { border-color: #333; cursor: default; }
.ss-ps-ignore {
background: none;
border: 1px solid #6b7280;
border-radius: 4px;
color: #9ca3af;
font-size: 11px;
width: 24px;
height: 24px;
cursor: pointer;
display: flex;
align-items: center;
justify-content: center;
flex-shrink: 0;
padding: 0;
}
.ss-ps-ignore:hover { background: rgba(156, 163, 175, 0.15); color: #e5e7eb; }
/* Floating reveal mode indicator */ /* Floating reveal mode indicator */
.ss-reveal-badge { .ss-reveal-badge {
position: fixed; position: fixed;
+57 -19
View File
@@ -400,50 +400,44 @@
/** /**
* Detect proper nouns (potential names, company names, project names) * Detect proper nouns (potential names, company names, project names)
* that aren't configured in identity. Uses capitalization heuristics: * that aren't configured in identity. Uses capitalization heuristics:
* - Capitalized words not at the start of a sentence * - ONLY multi-word capitalized sequences (e.g. "Acme Corp", "Project Atlas")
* - Multi-word capitalized sequences (e.g. "Acme Corp", "Project Atlas") * - Single capitalized words are too noisy — every sentence starts with one
* - Filters out common English words and programming terms * - Filters out common English words and programming terms
*/ */
function detectProperNouns(text, configured) { function detectProperNouns(text, configured) {
const findings = []; const findings = [];
// Match capitalized words that aren't at the very start of the text // Only match TWO OR MORE consecutive capitalized words
// and aren't after a period/newline (sentence start) // Single capitalized words cause too many false positives
const re = /(?:^|[.!?\n]\s*)?([A-Z][a-z]{2,}(?:\s+[A-Z][a-z]{2,})*)/g; const re = /\b([A-Z][a-z]{2,}(?:\s+[A-Z][a-z]{2,})+)\b/g;
let m; let m;
while ((m = re.exec(text)) !== null) { while ((m = re.exec(text)) !== null) {
const fullMatch = m[1]; const fullMatch = m[1];
if (!fullMatch) continue; if (!fullMatch) continue;
// Check if this is at the start of a sentence // Split into individual words and filter common ones
const before = text.slice(Math.max(0, m.index - 2), m.index);
const isSentenceStart = m.index === 0 || /[.!?\n]\s*$/.test(before);
// Split into individual words and check each
const words = fullMatch.split(/\s+/); const words = fullMatch.split(/\s+/);
const properWords = words.filter(w => const properWords = words.filter(w =>
w.length >= 3 && w.length >= 3 &&
!COMMON_CAPITALIZED.has(w.toLowerCase()) && !COMMON_CAPITALIZED.has(w.toLowerCase()) &&
!configured.has(w.toLowerCase()) !configured.has(w.toLowerCase()) &&
!ignoredValues.has(w.toLowerCase())
); );
if (properWords.length === 0) continue; if (properWords.length < 2) continue; // need at least 2 proper words
// Single capitalized word at sentence start = likely not a proper noun
if (isSentenceStart && properWords.length === 1 && words.length === 1) continue;
// Multi-word capitalized sequence is likely a proper noun
// Single capitalized word mid-sentence is likely a proper noun
const value = properWords.join(' '); const value = properWords.join(' ');
if (value.length >= 3 && !configured.has(value.toLowerCase())) { if (value.length >= 5 && !configured.has(value.toLowerCase()) && !ignoredValues.has(value.toLowerCase())) {
findings.push({ findings.push({
name: 'Possible Name/Org', name: 'Possible Name/Org',
value, value,
hint: 'Capitalized word — could be a name, company, or project', hint: 'Capitalized phrase — could be a name, company, or project',
category: 'name', category: 'name',
}); });
} }
} }
}
}
// Deduplicate // Deduplicate
const seen = new Set(); const seen = new Set();
@@ -477,6 +471,7 @@
while ((m = pat.re.exec(text)) !== null) { while ((m = pat.re.exec(text)) !== null) {
const val = m[0]; const val = m[0];
if (configured.has(val.toLowerCase())) continue; if (configured.has(val.toLowerCase())) continue;
if (ignoredValues.has(val.toLowerCase())) continue;
if (pat.skip && pat.skip.test(val)) continue; if (pat.skip && pat.skip.test(val)) continue;
findings.push({ name: pat.name, value: val, hint: pat.hint, category: pat.cat }); findings.push({ name: pat.name, value: val, hint: pat.hint, category: pat.cat });
} }
@@ -630,6 +625,30 @@
// ============================================================ // ============================================================
const sessionSubstitutions = new Map(); const sessionSubstitutions = new Map();
// Values the user has explicitly ignored via the "Ignore" button.
// Persisted to storage so they stay dismissed across page reloads.
const ignoredValues = new Set();
// Load ignored values from storage
(async () => {
const stored = await getStorageData('ss_ignored_ppi');
if (Array.isArray(stored)) {
for (const v of stored) ignoredValues.add(v.toLowerCase());
}
})();
function addIgnoredValue(value) {
ignoredValues.add(value.toLowerCase());
// Persist
getStorageData('ss_ignored_ppi').then(stored => {
const list = Array.isArray(stored) ? stored : [];
if (!list.includes(value.toLowerCase())) {
list.push(value.toLowerCase());
setStorageData('ss_ignored_ppi', list);
}
});
}
// ============================================================ // ============================================================
// Notify content script of substitutions (for badge + logging) // Notify content script of substitutions (for badge + logging)
// ============================================================ // ============================================================
@@ -1632,6 +1651,7 @@
${settings.autoAddDetected !== false ${settings.autoAddDetected !== false
? `<button class="ss-ps-add" data-real="${encodeURIComponent(w.value)}" data-fake="${encodeURIComponent(fake)}" data-cat="${w.category}" title="Add mapping: ${displayVal}${fake}">+</button>` ? `<button class="ss-ps-add" data-real="${encodeURIComponent(w.value)}" data-fake="${encodeURIComponent(fake)}" data-cat="${w.category}" title="Add mapping: ${displayVal}${fake}">+</button>`
: ''} : ''}
<button class="ss-ps-ignore" data-value="${encodeURIComponent(w.value)}" title="Ignore — stop flagging this value">&#10005;</button>
</div>`; </div>`;
}).join(''); }).join('');
@@ -1694,6 +1714,24 @@
btn.disabled = true; btn.disabled = true;
}); });
}); });
// Ignore buttons
preSendWarningEl.querySelectorAll('.ss-ps-ignore').forEach(btn => {
btn.addEventListener('click', () => {
const value = decodeURIComponent(btn.dataset.value);
addIgnoredValue(value);
// Remove this item's row
const row = btn.closest('.ss-ps-item');
if (row) row.remove();
// Re-scan to update warning
if (inputEl) {
if (inputScanTimer) clearTimeout(inputScanTimer);
inputScanTimer = setTimeout(() => scanInputForPPI(inputEl), 150);
}
});
});
} }
// Replace all occurrences of `real` with `fake` in an input or contenteditable element // Replace all occurrences of `real` with `fake` in an input or contenteditable element
+6 -8
View File
@@ -218,16 +218,15 @@ const AutoDetect = {
*/ */
_detectProperNouns(text, configured) { _detectProperNouns(text, configured) {
const findings = []; const findings = [];
const re = /(?:^|[.!?\n]\s*)?([A-Z][a-z]{2,}(?:\s+[A-Z][a-z]{2,})*)/g; // Only match TWO OR MORE consecutive capitalized words
// Single capitalized words cause too many false positives (sentence starts)
const re = /\b([A-Z][a-z]{2,}(?:\s+[A-Z][a-z]{2,})+)\b/g;
let m; let m;
while ((m = re.exec(text)) !== null) { while ((m = re.exec(text)) !== null) {
const fullMatch = m[1]; const fullMatch = m[1];
if (!fullMatch) continue; if (!fullMatch) continue;
const before = text.slice(Math.max(0, m.index - 2), m.index);
const isSentenceStart = m.index === 0 || /[.!?\n]\s*$/.test(before);
const words = fullMatch.split(/\s+/); const words = fullMatch.split(/\s+/);
const properWords = words.filter(w => const properWords = words.filter(w =>
w.length >= 3 && w.length >= 3 &&
@@ -235,15 +234,14 @@ const AutoDetect = {
!configured.has(w.toLowerCase()) !configured.has(w.toLowerCase())
); );
if (properWords.length === 0) continue; if (properWords.length < 2) continue; // need at least 2 proper words
if (isSentenceStart && properWords.length === 1 && words.length === 1) continue;
const value = properWords.join(' '); const value = properWords.join(' ');
if (value.length >= 3 && !configured.has(value.toLowerCase())) { if (value.length >= 5 && !configured.has(value.toLowerCase())) {
findings.push({ findings.push({
name: 'Possible Name/Org', name: 'Possible Name/Org',
value, value,
hint: 'Capitalized word — could be a name, company, or project', hint: 'Capitalized phrase — could be a name, company, or project',
category: 'name', category: 'name',
}); });
} }