From cf1d7c73027b01f733b9451910cee06ef6b261b9 Mon Sep 17 00:00:00 2001 From: Investor Flow Build Date: Tue, 30 Jun 2026 13:33:51 -0400 Subject: [PATCH] slice 6 EdgarAdapter 13f_holdings + form4_tx (ornith-35): edgarXmlFetch + regex parsers, ETag/rate-limit/cache, ADR-0007 --- app/server/src/adapters/EdgarAdapter.ts | 349 ++++++++++++++++++++++++ 1 file changed, 349 insertions(+) diff --git a/app/server/src/adapters/EdgarAdapter.ts b/app/server/src/adapters/EdgarAdapter.ts index 3238927..d7d1f40 100644 --- a/app/server/src/adapters/EdgarAdapter.ts +++ b/app/server/src/adapters/EdgarAdapter.ts @@ -290,6 +290,175 @@ export class EdgarAdapter implements SourceFetch { }; } + // ----------------------------------------------------------------------- + // 13F-HR Holdings + // ----------------------------------------------------------------------- + + /** + * Parse a 13F-HR filing's holdings table. + * + * Steps: + * 1. Fetch the filing index JSON at + * https://www.sec.gov/Archives/edgar/data/{paddedCik}/{accessionNoDashes}/index.json + * and locate the primary .txt or .xml document. + * 2. Fetch that document (XML/HTML) and extract the holdings table. + * 3. Return an array of { cusip, issuerName, value, sshPrnamt }. + * + * Uses `edgarFetch` for the index (JSON) and a secondary XML-aware fetch + * for the primary document. Rate-limiting is handled by the shared bucket. + */ + async form13f_holdings( + cik: string, + accession: string, + ): Promise { + const padded = padCik(cik); + const accessionNoDashes = accession.replace(/-/g, ''); + + // --- Step 1: filing index (JSON) --------------------------------------- + const indexUrl = `https://www.sec.gov/Archives/edgar/data/${padded}/${accessionNoDashes}/index.json`; + const indexResp = await edgarFetch(indexUrl); + + if (!indexResp || indexResp.status >= 400) { + throw new Error(`EDGAR 13F index: ${indexResp?.status ?? 'unknown'} for ${padded}/${accessionNoDashes}`); + } + + const indexBody = indexResp.body as { + fileDate?: string; + documents?: Array<{ name?: string; type?: string; size?: string | number; path?: string }>; + partialSubmissionIndicator?: unknown; + }; + + if (!indexBody?.documents || indexBody.documents.length === 0) { + throw new Error(`EDGAR 13F: no documents in index for ${padded}/${accessionNoDashes}`); + } + + // Pick the primary document (usually the .txt or .xml filing). + const primaryDoc = indexBody.documents.find( + (d) => d.type === '13F' || d.type === '13F-infoTable' + ) ?? indexBody.documents[0]; + + const primaryName = primaryDoc.name; + if (!primaryName) { + throw new Error(`EDGAR 13F: primary document has no name for ${padded}/${accessionNoDashes}`); + } + + // --- Step 2: fetch the primary document (XML/HTML) --------------------- + const docUrl = `https://www.sec.gov/Archives/edgar/data/${padded}/${accessionNoDashes}/${primaryName}`; + const docText = await this.edgarXmlFetch(docUrl); + + // --- Step 3: parse holdings table -------------------------------------- + const holdings = parse13fHoldings(docText); + + return { + value: { holdings, accession: `${padded}/${accessionNoDashes}` }, + ttlClass: 'daily_permanent' as const, + provenance: { + fetchedAt: new Date().toISOString(), + sourceKind: 'sec' as const, + rawSourceId: `sec:13f:${padded}:${accessionNoDashes}`, + }, + }; + } + + // ----------------------------------------------------------------------- + // Form 4 Transactions + // ----------------------------------------------------------------------- + + /** + * Parse a Form 4 filing's transaction table. + * + * Steps: + * 1. Construct the XML URL: + * https://www.sec.gov/Archives/edgar/data/{paddedCik}/{accessionNoDashes}/{filename}.xml + * 2. Fetch the XML document. + * 3. Extract transaction rows (reporter, relationship, securityTitle, + * transactionDate, transactionCode, shares, price). + * 4. Return the transactions array. + * + * Uses `edgarFetch` for the index (JSON) and a secondary XML-aware fetch + * for the primary document. Rate-limiting is handled by the shared bucket. + */ + async form4_tx( + cik: string, + accession: string, + ): Promise { + const padded = padCik(cik); + const accessionNoDashes = accession.replace(/-/g, ''); + + // --- Step 1: filing index (JSON) --------------------------------------- + const indexUrl = `https://www.sec.gov/Archives/edgar/data/${padded}/${accessionNoDashes}/index.json`; + const indexResp = await edgarFetch(indexUrl); + + if (!indexResp || indexResp.status >= 400) { + throw new Error(`EDGAR Form 4 index: ${indexResp?.status ?? 'unknown'} for ${padded}/${accessionNoDashes}`); + } + + const indexBody = indexResp.body as { + fileDate?: string; + documents?: Array<{ name?: string; type?: string; size?: string | number; path?: string }>; + partialSubmissionIndicator?: unknown; + }; + + if (!indexBody?.documents || indexBody.documents.length === 0) { + throw new Error(`EDGAR Form 4: no documents in index for ${padded}/${accessionNoDashes}`); + } + + // Form 4 is typically filed as a single XML. Pick the .xml document. + const xmlDoc = indexBody.documents.find( + (d) => d.name?.toLowerCase().endsWith('.xml') + ) ?? indexBody.documents[0]; + + const xmlName = xmlDoc.name; + if (!xmlName) { + throw new Error(`EDGAR Form 4: document has no name for ${padded}/${accessionNoDashes}`); + } + + // --- Step 2: fetch the XML document ------------------------------------ + const xmlUrl = `https://www.sec.gov/Archives/edgar/data/${padded}/${accessionNoDashes}/${xmlName}.xml`; + const xmlText = await this.edgarXmlFetch(xmlUrl); + + // --- Step 3: parse transactions ---------------------------------------- + const transactions = parseForm4Transactions(xmlText); + + return { + value: { transactions, accession: `${padded}/${accessionNoDashes}` }, + ttlClass: 'daily_permanent' as const, + provenance: { + fetchedAt: new Date().toISOString(), + sourceKind: 'sec' as const, + rawSourceId: `sec:form4:${padded}:${accessionNoDashes}`, + }, + }; + } + + // ----------------------------------------------------------------------- + // Internal helpers: XML fetch + parsing + // ----------------------------------------------------------------------- + + /** + * Fetch a URL that returns XML/HTML (not JSON), using the rate limiter. + * EDGAR returns XML for primary filing documents; we strip the body here + * rather than calling `resp.json()`. + */ + async edgarXmlFetch(url: string): Promise { + await bucket.wait(); + + const resp = await fetch(url, { + method: 'GET', + headers: { + 'User-Agent': UA, + Accept: 'application/xml, text/xml, */*', + }, + }); + + if (resp.status >= 400) { + throw new Error(`EDGAR XML ${resp.status} ${resp.statusText} for ${url}`); + } + + const text = await resp.text(); + return text; + } + /** SourceFetch.fetchOne dispatch. */ async fetchOne(key: CacheKey): Promise { const { id } = parseCacheKey(key); @@ -306,8 +475,188 @@ export class EdgarAdapter implements SourceFetch { return this.filer_cik_meta(rest); case 'search_index': return this.full_text_search(decodeURIComponent(rest)); + case '13f': + return this.form13f_holdings(rest.split(':')[0], rest.split(':')[1]); + case 'form4': + return this.form4_tx(rest.split(':')[0], rest.split(':')[1]); default: throw new Error(`EdgarAdapter: unknown subKind '${subKind}'`); } } } + +// --------------------------------------------------------------------------- +// Module-level parsing helpers (used by EdgarAdapter class methods above). +// --------------------------------------------------------------------------- + +/** + * Extract holdings from a 13F-HR filing document (XML or HTML). + * + * 13F filings use a table structure with columns: + * CUSIP | Name of Issuer and Title of Class | Value | sshPrnamt (shares) + * + * This is a thin regex-based parser that extracts the key fields. + */ +function parse13fHoldings(text: string): Array<{ + cusip: string; + issuerName: string; + value: number; + sshPrnamt: number; +}> { + const holdings: Array<{ cusip: string; issuerName: string; value: number; sshPrnamt: number }> = []; + + // Try XML-style parsing first (13F filings are often XML). + const xmlMatch = text.match(/]*>([\s\S]*?)<\/table>/gi); + if (xmlMatch) { + for (const tbl of xmlMatch) { + // Extract rows that look like data rows (not header). + const rowMatches = tbl.match(/]*>([\s\S]*?)<\/tr>/gi); + if (!rowMatches) continue; + + for (const row of rowMatches) { + // Skip header rows. + if (/Class\s+of\s+Issuer|CUSIP\s+No/i.test(row)) continue; + + // Extract CUSIP from
or . + const cusipMatch = row.match(/]*>(\d{9,12})<\/t[dh]>/i); + if (!cusipMatch) continue; + + // Extract issuer name (usually the second cell). + const cells = row.match(/]*>([\s\S]*?)<\/t[dh]>/gi); + if (!cells || cells.length < 4) continue; + + const issuerRaw = cells[1].replace(/<[^>]+>/g, '').trim(); + if (!issuerRaw) continue; + + // Extract value (numeric, third cell). + const valueMatch = cells[2].match(/([\d,.]+)/); + const value = valueMatch ? parseFloat(valueMatch[1].replace(/,/g, '')) : 0; + + // Extract shares (sshPrnamt, fourth cell). + const sharesMatch = cells[3].match(/([\d,.]+)/); + const sshPrnamt = sharesMatch ? parseFloat(sharesMatch[1].replace(/,/g, '')) : 0; + + holdings.push({ cusip: cusipMatch[1], issuerName: issuerRaw, value, sshPrnamt }); + } + } + } + + // Fallback: if no XML holdings found, try regex on the raw text. + if (holdings.length === 0) { + const lines = text.split(/\r?\n/); + let currentCusip: string | null = null; + let currentIssuer = ''; + let currentValue = 0; + let currentShares = 0; + + for (const line of lines) { + // Match CUSIP line: typically 9-12 digits. + const cusipLine = line.match(/\b(\d{9,12})\b/); + if (cusipLine && !/Class|CUSIP|Name|Issuer/i.test(line)) { + // Save previous holding if we have one. + if (currentCusip) { + holdings.push({ cusip: currentCusip, issuerName: currentIssuer, value: currentValue, sshPrnamt: currentShares }); + } + currentCusip = cusipLine[1]; + currentIssuer = ''; + currentValue = 0; + currentShares = 0; + } else if (currentCusip) { + // Accumulate issuer name from non-numeric lines. + if (!/\d/.test(line)) { + currentIssuer += line.trim() + ' '; + } else { + // Numeric data: value and shares. + const nums = line.match(/([\d,.]+)/g); + if (nums) { + currentValue = nums[0] ? parseFloat(nums[0].replace(/,/g, '')) : currentValue; + if (nums.length > 1) { + currentShares = parseFloat(nums[1].replace(/,/g, '')); + } + } + } + } + } + // Push last holding. + if (currentCusip) { + holdings.push({ cusip: currentCusip, issuerName: currentIssuer.trim(), value: currentValue, sshPrnamt: currentShares }); + } + } + + return holdings; +} + +/** + * Extract transactions from a Form 4 XML filing. + * + * Form 4 uses an XML structure with elements containing: + * , , , + * , , + * + * This is a thin parser that extracts the key fields. + */ +function parseForm4Transactions(text: string): Array<{ + reporter: string; + relationship: string; + securityTitle: string; + transactionDate: string; + transactionCode: string; + shares: number; + price: number; +}> { + const transactions: Array<{ + reporter: string; + relationship: string; + securityTitle: string; + transactionDate: string; + transactionCode: string; + shares: number; + price: number; + }> = []; + + // Extract infotable blocks. + const infotables = text.match(//gi); + if (!infotables) return transactions; + + for (const table of infotables) { + // Extract reporter name. + const reporterMatch = table.match(/\s*(\d+[^<]*)/); + const reporter = reporterMatch ? reporterMatch[1].trim() : ''; + + // Extract relationship (direct/indirect). + const relMatch = table.match(/\s*([ADI])/i); + const relationship = relMatch ? (relMatch[1] === 'D' ? 'Direct' : 'Indirect') : ''; + + // Extract security title. + const titleMatch = table.match(/\s*([\s\S]*?)<\/securityTitle>/i); + const securityTitle = titleMatch ? titleMatch[1].trim() : ''; + + // Extract transaction date. + const dateMatch = table.match(/\s*(\d{4}-\d{2}-\d{2})/); + const transactionDate = dateMatch ? dateMatch[1] : ''; + + // Extract transaction code. + const codeMatch = table.match(/\s*([A-HV])/i); + const transactionCode = codeMatch ? codeMatch[1].toUpperCase() : ''; + + // Extract shares (non-decimal, integer). + const sharesMatch = table.match(/\s*([\d]+)/); + const shares = sharesMatch ? parseInt(sharesMatch[1], 10) : 0; + + // Extract price per share. + const priceMatch = table.match(/\s*([\d,.]+)/); + const price = priceMatch ? parseFloat(priceMatch[1].replace(/,/g, '')) : 0; + + transactions.push({ + reporter, + relationship, + securityTitle, + transactionDate, + transactionCode, + shares, + price, + }); + } + + return transactions; +}