feat: password import, proper noun detection, full README update
Import parser: - Passwords now imported from Chrome/Firefox/Bitwarden/1Password CSVs as exact-match auto-redacted mappings (→ [REDACTED-PASSWORD-N]) - Catches passwords in any context, not just key=value patterns Proper noun heuristic: - Auto-detect scanner now catches capitalized words mid-sentence as potential names, company names, or project names - Filters against 200+ common English words, programming terms, days, months to reduce false positives - Added to both content.js (page world) and auto-detect.js (popup) README: - Documented proper noun detection with examples - Documented bulk import with all supported formats - Added Sync features section (auto sync, conflict resolution, version history, connected devices) - Added Organization/Team section with policy format - Added tamper protection documentation - Added Legal section with liability analysis - Updated architecture with all new modules https://claude.ai/code/session_01SWSwDfMVij53bCTNSCLMwn
This commit is contained in:
+86
-3
@@ -193,21 +193,104 @@ const AutoDetect = {
|
||||
}
|
||||
}
|
||||
|
||||
// Proper noun heuristic — catch names, company names, project names
|
||||
const properNouns = this._detectProperNouns(text, configured);
|
||||
findings.push(...properNouns);
|
||||
|
||||
// Deduplicate overlapping matches
|
||||
findings.sort((a, b) => a.index - b.index);
|
||||
findings.sort((a, b) => (a.index || 0) - (b.index || 0));
|
||||
const deduped = [];
|
||||
let lastEnd = -1;
|
||||
for (const f of findings) {
|
||||
if (f.index >= lastEnd) {
|
||||
const idx = f.index || 0;
|
||||
if (idx >= lastEnd) {
|
||||
deduped.push(f);
|
||||
lastEnd = f.index + f.value.length;
|
||||
lastEnd = idx + f.value.length;
|
||||
}
|
||||
}
|
||||
|
||||
return deduped;
|
||||
},
|
||||
|
||||
/**
|
||||
* Detect capitalized words mid-sentence that might be proper nouns
|
||||
* (names, company names, project names) not in the configured set.
|
||||
*/
|
||||
_detectProperNouns(text, configured) {
|
||||
const findings = [];
|
||||
const re = /(?:^|[.!?\n]\s*)?([A-Z][a-z]{2,}(?:\s+[A-Z][a-z]{2,})*)/g;
|
||||
let m;
|
||||
|
||||
while ((m = re.exec(text)) !== null) {
|
||||
const fullMatch = m[1];
|
||||
if (!fullMatch) continue;
|
||||
|
||||
const before = text.slice(Math.max(0, m.index - 2), m.index);
|
||||
const isSentenceStart = m.index === 0 || /[.!?\n]\s*$/.test(before);
|
||||
|
||||
const words = fullMatch.split(/\s+/);
|
||||
const properWords = words.filter(w =>
|
||||
w.length >= 3 &&
|
||||
!COMMON_WORDS.has(w.toLowerCase()) &&
|
||||
!configured.has(w.toLowerCase())
|
||||
);
|
||||
|
||||
if (properWords.length === 0) continue;
|
||||
if (isSentenceStart && properWords.length === 1 && words.length === 1) continue;
|
||||
|
||||
const value = properWords.join(' ');
|
||||
if (value.length >= 3 && !configured.has(value.toLowerCase())) {
|
||||
findings.push({
|
||||
name: 'Possible Name/Org',
|
||||
value,
|
||||
hint: 'Capitalized word — could be a name, company, or project',
|
||||
category: 'name',
|
||||
});
|
||||
}
|
||||
}
|
||||
|
||||
const seen = new Set();
|
||||
return findings.filter(f => {
|
||||
if (seen.has(f.value)) return false;
|
||||
seen.add(f.value);
|
||||
return true;
|
||||
});
|
||||
},
|
||||
};
|
||||
|
||||
// Common English words to exclude from proper noun detection
|
||||
const COMMON_WORDS = new Set([
|
||||
'the', 'and', 'but', 'for', 'not', 'you', 'all', 'can', 'had', 'her',
|
||||
'was', 'one', 'our', 'out', 'are', 'has', 'his', 'how', 'its', 'may',
|
||||
'new', 'now', 'old', 'see', 'way', 'who', 'did', 'get', 'let', 'say',
|
||||
'she', 'too', 'use', 'also', 'back', 'been', 'call', 'came', 'come',
|
||||
'could', 'each', 'even', 'find', 'from', 'give', 'good', 'great',
|
||||
'have', 'here', 'high', 'into', 'just', 'keep', 'know', 'last', 'like',
|
||||
'live', 'long', 'look', 'made', 'make', 'many', 'more', 'most', 'much',
|
||||
'must', 'name', 'next', 'only', 'over', 'part', 'people', 'place',
|
||||
'same', 'show', 'side', 'since', 'some', 'still', 'such', 'take',
|
||||
'tell', 'than', 'that', 'them', 'then', 'there', 'these', 'they',
|
||||
'this', 'time', 'turn', 'used', 'very', 'want', 'well', 'were',
|
||||
'what', 'when', 'where', 'which', 'while', 'will', 'with', 'work',
|
||||
'would', 'year', 'your', 'about', 'after', 'again', 'being', 'between',
|
||||
'both', 'before', 'down', 'during', 'first', 'found', 'group',
|
||||
'however', 'important', 'large', 'later', 'little', 'never',
|
||||
'number', 'other', 'point', 'right', 'small', 'state', 'thing',
|
||||
'think', 'those', 'three', 'through', 'under', 'until', 'water',
|
||||
'world', 'write', 'might', 'should', 'because', 'although',
|
||||
// Programming / tech terms
|
||||
'string', 'number', 'boolean', 'object', 'array', 'function', 'class',
|
||||
'type', 'error', 'null', 'undefined', 'true', 'false', 'return',
|
||||
'import', 'export', 'default', 'const', 'async', 'await',
|
||||
'note', 'example', 'warning', 'step', 'option', 'result', 'value',
|
||||
'key', 'data', 'info', 'file', 'code', 'test', 'debug', 'config',
|
||||
'setup', 'update', 'please', 'thanks', 'hello', 'sorry',
|
||||
// Days and months
|
||||
'monday', 'tuesday', 'wednesday', 'thursday', 'friday', 'saturday', 'sunday',
|
||||
'january', 'february', 'march', 'april', 'may', 'june', 'july',
|
||||
'august', 'september', 'october', 'november', 'december',
|
||||
]);
|
||||
|
||||
if (typeof globalThis !== 'undefined') {
|
||||
globalThis.AutoDetect = AutoDetect;
|
||||
}
|
||||
|
||||
@@ -13,9 +13,9 @@
|
||||
* Imports as identity fields with blank substitutes so the user
|
||||
* can see what needs mapping and fill in fakes.
|
||||
*
|
||||
* 3. Chrome password CSV export — extracts usernames, names, URLs
|
||||
* 3. Chrome password CSV export — extracts usernames, names, URLs, passwords
|
||||
* Columns: name, url, username, password, note
|
||||
* Passwords are NEVER imported. Only usernames/emails/URLs.
|
||||
* Passwords are imported as auto-redacted mappings (e.g. → [REDACTED-PASSWORD-1]).
|
||||
*
|
||||
* 4. Firefox password CSV export — similar to Chrome
|
||||
* Columns: url, username, password, ...
|
||||
@@ -139,7 +139,10 @@ const ImportParser = {
|
||||
|
||||
/**
|
||||
* Parse Chrome/Firefox password CSV export.
|
||||
* NEVER imports passwords — only usernames, emails, and domains.
|
||||
* Imports usernames, emails, domains, AND passwords.
|
||||
* Passwords are imported as mappings with auto-generated redaction
|
||||
* substitutes (e.g. "[REDACTED-PASSWORD-1]") so they get caught
|
||||
* if pasted into any context — not just key=value patterns.
|
||||
*/
|
||||
parsePasswordCSV(text) {
|
||||
const result = this._empty();
|
||||
@@ -148,12 +151,15 @@ const ImportParser = {
|
||||
|
||||
const headers = this._splitCSVLine(lines[0], ',').map(h => h.trim().toLowerCase());
|
||||
const usernameIdx = headers.findIndex(h => h === 'username' || h === 'login_username' || h === 'user');
|
||||
const passwordIdx = headers.findIndex(h => h === 'password' || h === 'login_password');
|
||||
const urlIdx = headers.findIndex(h => h === 'url' || h === 'login_uri' || h === 'origin' || h === 'web site');
|
||||
const nameIdx = headers.findIndex(h => h === 'name' || h === 'title');
|
||||
|
||||
const seenEmails = new Set();
|
||||
const seenUsernames = new Set();
|
||||
const seenDomains = new Set();
|
||||
const seenPasswords = new Set();
|
||||
let passwordCount = 0;
|
||||
|
||||
for (let i = 1; i < lines.length; i++) {
|
||||
const cols = this._splitCSVLine(lines[i], ',');
|
||||
@@ -172,6 +178,22 @@ const ImportParser = {
|
||||
}
|
||||
}
|
||||
|
||||
// Extract password — import as a redacted mapping
|
||||
if (passwordIdx >= 0 && cols[passwordIdx]) {
|
||||
const password = cols[passwordIdx].trim();
|
||||
// Skip very short or empty passwords, and deduplicate
|
||||
if (password && password.length >= 4 && !seenPasswords.has(password)) {
|
||||
seenPasswords.add(password);
|
||||
passwordCount++;
|
||||
result.mappings.push({
|
||||
real: password,
|
||||
substitute: `[REDACTED-PASSWORD-${passwordCount}]`,
|
||||
category: 'password',
|
||||
caseSensitive: true,
|
||||
});
|
||||
}
|
||||
}
|
||||
|
||||
// Extract domain from URL
|
||||
if (urlIdx >= 0 && cols[urlIdx]) {
|
||||
try {
|
||||
@@ -202,9 +224,12 @@ const ImportParser = {
|
||||
|
||||
const headers = this._splitCSVLine(lines[0], ',').map(h => h.trim().toLowerCase());
|
||||
const usernameIdx = headers.findIndex(h => h.includes('username'));
|
||||
const passwordIdx = headers.findIndex(h => h.includes('password'));
|
||||
const uriIdx = headers.findIndex(h => h.includes('uri') || h.includes('url'));
|
||||
|
||||
const seen = new Set();
|
||||
const seenPasswords = new Set();
|
||||
let passwordCount = 0;
|
||||
|
||||
for (let i = 1; i < lines.length; i++) {
|
||||
const cols = this._splitCSVLine(lines[i], ',');
|
||||
@@ -221,6 +246,20 @@ const ImportParser = {
|
||||
}
|
||||
}
|
||||
|
||||
if (passwordIdx >= 0 && cols[passwordIdx]) {
|
||||
const pw = cols[passwordIdx].trim();
|
||||
if (pw && pw.length >= 4 && !seenPasswords.has(pw)) {
|
||||
seenPasswords.add(pw);
|
||||
passwordCount++;
|
||||
result.mappings.push({
|
||||
real: pw,
|
||||
substitute: `[REDACTED-PASSWORD-${passwordCount}]`,
|
||||
category: 'password',
|
||||
caseSensitive: true,
|
||||
});
|
||||
}
|
||||
}
|
||||
|
||||
if (uriIdx >= 0 && cols[uriIdx]) {
|
||||
try {
|
||||
const domain = new URL(cols[uriIdx].trim()).hostname;
|
||||
|
||||
Reference in New Issue
Block a user