import fs from 'fs' const all = JSON.parse(fs.readFileSync(process.argv[2], 'utf8')) const norm = s => s.toLowerCase().replace(/[^a-z0-9 ]/g, ' ').replace(/\s+/g, ' ').trim() const INCLUDE = /\baccount(?:ant|ing)\b|\bcontroller\b|\bbookkeep/i const seen = new Set(), jobs = [] for (const j of all.filter(x => INCLUDE.test(x.title))) { const k = j.co + '|' + norm(j.title) + '|' + j.content.length if (!seen.has(k)) { seen.add(k); jobs.push(j) } } // A case-insensitive \bExcel\b also counted the verb ("everyone can excel", "you'd // excel in this role") — about one match in nine on the 2026-09-25 corpus. // Capitalised Excel is the product unless followed by at/in/as; lower-case excel // counts only after a word that makes it a tool ("user of excel", "advanced excel"). const EXCEL_PRODUCT = /\bExcel\b(?!\s+(?:at|in|as)\b)/ const EXCEL_CONTEXT = /\b(?:MS|Microsoft|advanced|user of|proficien(?:t|cy) (?:in|with)|skills in|knowledge of|expert in|using)\s+excel\b/i const unesc = s => s.replace(/</g,'<').replace(/>/g,'>').replace(/&/g,'&') .replace(/"/g,'"').replace(/'/g,"'").replace(/ /g,' ') const texts = jobs.map(j => unesc(j.content).replace(/<[^>]+>/g,' ').replace(/\s+/g,' ')) const T = [ ['Standards & rules','GAAP',/\bGAAP\b/i],['Standards & rules','IFRS',/\bIFRS\b/i], ['Standards & rules','SOX / Sarbanes-Oxley',/\bSOX\b|\bSarbanes/i], ['Standards & rules','ASC 606',/\bASC\s*606\b/i], ['Standards & rules','revenue recognition',/\brevenue recognition\b/i], ['Standards & rules','internal controls',/\binternal controls?\b/i], ['Credentials','CPA',/\bCPA\b/i],['Credentials','Big 4 experience',/\bBig\s*4\b|\bBig Four\b/i], ['Credentials','CMA',/\bCMA\b/],['Credentials','public accounting',/\bpublic accounting\b/i], ['Systems','NetSuite',/\bNetSuite\b/i],['Systems','SAP',/\bSAP\b/],['Systems','Oracle',/\bOracle\b/i], ['Systems','Workday',/\bWorkday\b/i],['Systems','QuickBooks',/\bQuickBooks\b/i], ['Systems','Excel',{ test: (t) => EXCEL_PRODUCT.test(t) || EXCEL_CONTEXT.test(t) }],['Systems','SQL',/\bSQL\b/i],['Systems','Blackline',/\bBlackline\b/i], ['Systems','Coupa',/\bCoupa\b/i],['Systems','Looker / Tableau',/\bLooker\b|\bTableau\b/i], ['Process','audit',/\baudit/i],['Process','reconciliation',/\breconcil/i], ['Process','journal entries',/\bjournal entr/i],['Process','month-end close',/\bmonth-?end close\b|\bmonthly close\b/i], ['Process','financial statements',/\bfinancial statements?\b/i],['Process','accruals',/\baccrual/i], ['Process','variance analysis',/\bvariance analysis\b/i],['Process','forecasting',/\bforecast/i], ['Process','process improvement / automation',/\bprocess improvement\b|\bautomation\b/i], ['Process','cross-functional',/\bcross-?functional\b/i], ] const n = texts.length const moe = p => +(1.96 * Math.sqrt((p/100)*(1-p/100)/n) * 100).toFixed(1) const rows = T.map(([cat, term, re]) => { const c = texts.filter(t => re.test(t)).length const pct = +(100*c/n).toFixed(1) return { cat, term, n: c, pct, moe: moe(pct) } }).sort((a,b) => b.n - a.n) const out = { role: 'Accountant', n, companies: [...new Set(jobs.map(j=>j.co))].sort(), collected: new Date(Date.now() - new Date().getTimezoneOffset() * 60000).toISOString().slice(0,10), medianWords: texts.map(t=>t.split(' ').length).sort((a,b)=>a-b)[Math.floor(n/2)], rows, } fs.writeFileSync(process.argv[3], JSON.stringify(out, null, 1)) console.log(`n=${n} from ${out.companies.length} companies, median ${out.medianWords} words`) rows.slice(0,12).forEach((r,i)=>console.log(` ${String(i+1).padStart(2)}. ${r.term.padEnd(28)} ${String(r.pct).padStart(5)}% ±${r.moe}`))