slice 6 EdgarAdapter 13f_holdings + form4_tx (ornith-35): edgarXmlFetch + regex parsers, ETag/rate-limit/cache, ADR-0007
This commit is contained in:
@@ -290,6 +290,175 @@ export class EdgarAdapter implements SourceFetch {
|
||||
};
|
||||
}
|
||||
|
||||
// -----------------------------------------------------------------------
|
||||
// 13F-HR Holdings
|
||||
// -----------------------------------------------------------------------
|
||||
|
||||
/**
|
||||
* Parse a 13F-HR filing's holdings table.
|
||||
*
|
||||
* Steps:
|
||||
* 1. Fetch the filing index JSON at
|
||||
* https://www.sec.gov/Archives/edgar/data/{paddedCik}/{accessionNoDashes}/index.json
|
||||
* and locate the primary .txt or .xml document.
|
||||
* 2. Fetch that document (XML/HTML) and extract the holdings table.
|
||||
* 3. Return an array of { cusip, issuerName, value, sshPrnamt }.
|
||||
*
|
||||
* Uses `edgarFetch` for the index (JSON) and a secondary XML-aware fetch
|
||||
* for the primary document. Rate-limiting is handled by the shared bucket.
|
||||
*/
|
||||
async form13f_holdings(
|
||||
cik: string,
|
||||
accession: string,
|
||||
): Promise<FetchResult> {
|
||||
const padded = padCik(cik);
|
||||
const accessionNoDashes = accession.replace(/-/g, '');
|
||||
|
||||
// --- Step 1: filing index (JSON) ---------------------------------------
|
||||
const indexUrl = `https://www.sec.gov/Archives/edgar/data/${padded}/${accessionNoDashes}/index.json`;
|
||||
const indexResp = await edgarFetch(indexUrl);
|
||||
|
||||
if (!indexResp || indexResp.status >= 400) {
|
||||
throw new Error(`EDGAR 13F index: ${indexResp?.status ?? 'unknown'} for ${padded}/${accessionNoDashes}`);
|
||||
}
|
||||
|
||||
const indexBody = indexResp.body as {
|
||||
fileDate?: string;
|
||||
documents?: Array<{ name?: string; type?: string; size?: string | number; path?: string }>;
|
||||
partialSubmissionIndicator?: unknown;
|
||||
};
|
||||
|
||||
if (!indexBody?.documents || indexBody.documents.length === 0) {
|
||||
throw new Error(`EDGAR 13F: no documents in index for ${padded}/${accessionNoDashes}`);
|
||||
}
|
||||
|
||||
// Pick the primary document (usually the .txt or .xml filing).
|
||||
const primaryDoc = indexBody.documents.find(
|
||||
(d) => d.type === '13F' || d.type === '13F-infoTable'
|
||||
) ?? indexBody.documents[0];
|
||||
|
||||
const primaryName = primaryDoc.name;
|
||||
if (!primaryName) {
|
||||
throw new Error(`EDGAR 13F: primary document has no name for ${padded}/${accessionNoDashes}`);
|
||||
}
|
||||
|
||||
// --- Step 2: fetch the primary document (XML/HTML) ---------------------
|
||||
const docUrl = `https://www.sec.gov/Archives/edgar/data/${padded}/${accessionNoDashes}/${primaryName}`;
|
||||
const docText = await this.edgarXmlFetch(docUrl);
|
||||
|
||||
// --- Step 3: parse holdings table --------------------------------------
|
||||
const holdings = parse13fHoldings(docText);
|
||||
|
||||
return {
|
||||
value: { holdings, accession: `${padded}/${accessionNoDashes}` },
|
||||
ttlClass: 'daily_permanent' as const,
|
||||
provenance: {
|
||||
fetchedAt: new Date().toISOString(),
|
||||
sourceKind: 'sec' as const,
|
||||
rawSourceId: `sec:13f:${padded}:${accessionNoDashes}`,
|
||||
},
|
||||
};
|
||||
}
|
||||
|
||||
// -----------------------------------------------------------------------
|
||||
// Form 4 Transactions
|
||||
// -----------------------------------------------------------------------
|
||||
|
||||
/**
|
||||
* Parse a Form 4 filing's transaction table.
|
||||
*
|
||||
* Steps:
|
||||
* 1. Construct the XML URL:
|
||||
* https://www.sec.gov/Archives/edgar/data/{paddedCik}/{accessionNoDashes}/{filename}.xml
|
||||
* 2. Fetch the XML document.
|
||||
* 3. Extract transaction rows (reporter, relationship, securityTitle,
|
||||
* transactionDate, transactionCode, shares, price).
|
||||
* 4. Return the transactions array.
|
||||
*
|
||||
* Uses `edgarFetch` for the index (JSON) and a secondary XML-aware fetch
|
||||
* for the primary document. Rate-limiting is handled by the shared bucket.
|
||||
*/
|
||||
async form4_tx(
|
||||
cik: string,
|
||||
accession: string,
|
||||
): Promise<FetchResult> {
|
||||
const padded = padCik(cik);
|
||||
const accessionNoDashes = accession.replace(/-/g, '');
|
||||
|
||||
// --- Step 1: filing index (JSON) ---------------------------------------
|
||||
const indexUrl = `https://www.sec.gov/Archives/edgar/data/${padded}/${accessionNoDashes}/index.json`;
|
||||
const indexResp = await edgarFetch(indexUrl);
|
||||
|
||||
if (!indexResp || indexResp.status >= 400) {
|
||||
throw new Error(`EDGAR Form 4 index: ${indexResp?.status ?? 'unknown'} for ${padded}/${accessionNoDashes}`);
|
||||
}
|
||||
|
||||
const indexBody = indexResp.body as {
|
||||
fileDate?: string;
|
||||
documents?: Array<{ name?: string; type?: string; size?: string | number; path?: string }>;
|
||||
partialSubmissionIndicator?: unknown;
|
||||
};
|
||||
|
||||
if (!indexBody?.documents || indexBody.documents.length === 0) {
|
||||
throw new Error(`EDGAR Form 4: no documents in index for ${padded}/${accessionNoDashes}`);
|
||||
}
|
||||
|
||||
// Form 4 is typically filed as a single XML. Pick the .xml document.
|
||||
const xmlDoc = indexBody.documents.find(
|
||||
(d) => d.name?.toLowerCase().endsWith('.xml')
|
||||
) ?? indexBody.documents[0];
|
||||
|
||||
const xmlName = xmlDoc.name;
|
||||
if (!xmlName) {
|
||||
throw new Error(`EDGAR Form 4: document has no name for ${padded}/${accessionNoDashes}`);
|
||||
}
|
||||
|
||||
// --- Step 2: fetch the XML document ------------------------------------
|
||||
const xmlUrl = `https://www.sec.gov/Archives/edgar/data/${padded}/${accessionNoDashes}/${xmlName}.xml`;
|
||||
const xmlText = await this.edgarXmlFetch(xmlUrl);
|
||||
|
||||
// --- Step 3: parse transactions ----------------------------------------
|
||||
const transactions = parseForm4Transactions(xmlText);
|
||||
|
||||
return {
|
||||
value: { transactions, accession: `${padded}/${accessionNoDashes}` },
|
||||
ttlClass: 'daily_permanent' as const,
|
||||
provenance: {
|
||||
fetchedAt: new Date().toISOString(),
|
||||
sourceKind: 'sec' as const,
|
||||
rawSourceId: `sec:form4:${padded}:${accessionNoDashes}`,
|
||||
},
|
||||
};
|
||||
}
|
||||
|
||||
// -----------------------------------------------------------------------
|
||||
// Internal helpers: XML fetch + parsing
|
||||
// -----------------------------------------------------------------------
|
||||
|
||||
/**
|
||||
* Fetch a URL that returns XML/HTML (not JSON), using the rate limiter.
|
||||
* EDGAR returns XML for primary filing documents; we strip the body here
|
||||
* rather than calling `resp.json()`.
|
||||
*/
|
||||
async edgarXmlFetch(url: string): Promise<string> {
|
||||
await bucket.wait();
|
||||
|
||||
const resp = await fetch(url, {
|
||||
method: 'GET',
|
||||
headers: {
|
||||
'User-Agent': UA,
|
||||
Accept: 'application/xml, text/xml, */*',
|
||||
},
|
||||
});
|
||||
|
||||
if (resp.status >= 400) {
|
||||
throw new Error(`EDGAR XML ${resp.status} ${resp.statusText} for ${url}`);
|
||||
}
|
||||
|
||||
const text = await resp.text();
|
||||
return text;
|
||||
}
|
||||
|
||||
/** SourceFetch.fetchOne dispatch. */
|
||||
async fetchOne(key: CacheKey): Promise<FetchResult> {
|
||||
const { id } = parseCacheKey(key);
|
||||
@@ -306,8 +475,188 @@ export class EdgarAdapter implements SourceFetch {
|
||||
return this.filer_cik_meta(rest);
|
||||
case 'search_index':
|
||||
return this.full_text_search(decodeURIComponent(rest));
|
||||
case '13f':
|
||||
return this.form13f_holdings(rest.split(':')[0], rest.split(':')[1]);
|
||||
case 'form4':
|
||||
return this.form4_tx(rest.split(':')[0], rest.split(':')[1]);
|
||||
default:
|
||||
throw new Error(`EdgarAdapter: unknown subKind '${subKind}'`);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Module-level parsing helpers (used by EdgarAdapter class methods above).
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
/**
|
||||
* Extract holdings from a 13F-HR filing document (XML or HTML).
|
||||
*
|
||||
* 13F filings use a table structure with columns:
|
||||
* CUSIP | Name of Issuer and Title of Class | Value | sshPrnamt (shares)
|
||||
*
|
||||
* This is a thin regex-based parser that extracts the key fields.
|
||||
*/
|
||||
function parse13fHoldings(text: string): Array<{
|
||||
cusip: string;
|
||||
issuerName: string;
|
||||
value: number;
|
||||
sshPrnamt: number;
|
||||
}> {
|
||||
const holdings: Array<{ cusip: string; issuerName: string; value: number; sshPrnamt: number }> = [];
|
||||
|
||||
// Try XML-style <table> parsing first (13F filings are often XML).
|
||||
const xmlMatch = text.match(/<table[^>]*>([\s\S]*?)<\/table>/gi);
|
||||
if (xmlMatch) {
|
||||
for (const tbl of xmlMatch) {
|
||||
// Extract rows that look like data rows (not header).
|
||||
const rowMatches = tbl.match(/<tr[^>]*>([\s\S]*?)<\/tr>/gi);
|
||||
if (!rowMatches) continue;
|
||||
|
||||
for (const row of rowMatches) {
|
||||
// Skip header rows.
|
||||
if (/Class\s+of\s+Issuer|CUSIP\s+No/i.test(row)) continue;
|
||||
|
||||
// Extract CUSIP from <td> or <th>.
|
||||
const cusipMatch = row.match(/<t[dh][^>]*>(\d{9,12})<\/t[dh]>/i);
|
||||
if (!cusipMatch) continue;
|
||||
|
||||
// Extract issuer name (usually the second cell).
|
||||
const cells = row.match(/<t[dh][^>]*>([\s\S]*?)<\/t[dh]>/gi);
|
||||
if (!cells || cells.length < 4) continue;
|
||||
|
||||
const issuerRaw = cells[1].replace(/<[^>]+>/g, '').trim();
|
||||
if (!issuerRaw) continue;
|
||||
|
||||
// Extract value (numeric, third cell).
|
||||
const valueMatch = cells[2].match(/([\d,.]+)/);
|
||||
const value = valueMatch ? parseFloat(valueMatch[1].replace(/,/g, '')) : 0;
|
||||
|
||||
// Extract shares (sshPrnamt, fourth cell).
|
||||
const sharesMatch = cells[3].match(/([\d,.]+)/);
|
||||
const sshPrnamt = sharesMatch ? parseFloat(sharesMatch[1].replace(/,/g, '')) : 0;
|
||||
|
||||
holdings.push({ cusip: cusipMatch[1], issuerName: issuerRaw, value, sshPrnamt });
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// Fallback: if no XML holdings found, try regex on the raw text.
|
||||
if (holdings.length === 0) {
|
||||
const lines = text.split(/\r?\n/);
|
||||
let currentCusip: string | null = null;
|
||||
let currentIssuer = '';
|
||||
let currentValue = 0;
|
||||
let currentShares = 0;
|
||||
|
||||
for (const line of lines) {
|
||||
// Match CUSIP line: typically 9-12 digits.
|
||||
const cusipLine = line.match(/\b(\d{9,12})\b/);
|
||||
if (cusipLine && !/Class|CUSIP|Name|Issuer/i.test(line)) {
|
||||
// Save previous holding if we have one.
|
||||
if (currentCusip) {
|
||||
holdings.push({ cusip: currentCusip, issuerName: currentIssuer, value: currentValue, sshPrnamt: currentShares });
|
||||
}
|
||||
currentCusip = cusipLine[1];
|
||||
currentIssuer = '';
|
||||
currentValue = 0;
|
||||
currentShares = 0;
|
||||
} else if (currentCusip) {
|
||||
// Accumulate issuer name from non-numeric lines.
|
||||
if (!/\d/.test(line)) {
|
||||
currentIssuer += line.trim() + ' ';
|
||||
} else {
|
||||
// Numeric data: value and shares.
|
||||
const nums = line.match(/([\d,.]+)/g);
|
||||
if (nums) {
|
||||
currentValue = nums[0] ? parseFloat(nums[0].replace(/,/g, '')) : currentValue;
|
||||
if (nums.length > 1) {
|
||||
currentShares = parseFloat(nums[1].replace(/,/g, ''));
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
// Push last holding.
|
||||
if (currentCusip) {
|
||||
holdings.push({ cusip: currentCusip, issuerName: currentIssuer.trim(), value: currentValue, sshPrnamt: currentShares });
|
||||
}
|
||||
}
|
||||
|
||||
return holdings;
|
||||
}
|
||||
|
||||
/**
|
||||
* Extract transactions from a Form 4 XML filing.
|
||||
*
|
||||
* Form 4 uses an XML structure with <infotable> elements containing:
|
||||
* <reportOwner>, <securityTitle>, <transactionDate>,
|
||||
* <transactionCode>, <shares>, <pricePerShare>
|
||||
*
|
||||
* This is a thin parser that extracts the key fields.
|
||||
*/
|
||||
function parseForm4Transactions(text: string): Array<{
|
||||
reporter: string;
|
||||
relationship: string;
|
||||
securityTitle: string;
|
||||
transactionDate: string;
|
||||
transactionCode: string;
|
||||
shares: number;
|
||||
price: number;
|
||||
}> {
|
||||
const transactions: Array<{
|
||||
reporter: string;
|
||||
relationship: string;
|
||||
securityTitle: string;
|
||||
transactionDate: string;
|
||||
transactionCode: string;
|
||||
shares: number;
|
||||
price: number;
|
||||
}> = [];
|
||||
|
||||
// Extract infotable blocks.
|
||||
const infotables = text.match(/<infotable[\s\S]*?<\/infotable>/gi);
|
||||
if (!infotables) return transactions;
|
||||
|
||||
for (const table of infotables) {
|
||||
// Extract reporter name.
|
||||
const reporterMatch = table.match(/<reporterCik>\s*(\d+[^<]*)/);
|
||||
const reporter = reporterMatch ? reporterMatch[1].trim() : '';
|
||||
|
||||
// Extract relationship (direct/indirect).
|
||||
const relMatch = table.match(/<directOrIndirectOwner>\s*([ADI])/i);
|
||||
const relationship = relMatch ? (relMatch[1] === 'D' ? 'Direct' : 'Indirect') : '';
|
||||
|
||||
// Extract security title.
|
||||
const titleMatch = table.match(/<securityTitle>\s*([\s\S]*?)<\/securityTitle>/i);
|
||||
const securityTitle = titleMatch ? titleMatch[1].trim() : '';
|
||||
|
||||
// Extract transaction date.
|
||||
const dateMatch = table.match(/<transactionDate>\s*(\d{4}-\d{2}-\d{2})/);
|
||||
const transactionDate = dateMatch ? dateMatch[1] : '';
|
||||
|
||||
// Extract transaction code.
|
||||
const codeMatch = table.match(/<transactionCode>\s*([A-HV])/i);
|
||||
const transactionCode = codeMatch ? codeMatch[1].toUpperCase() : '';
|
||||
|
||||
// Extract shares (non-decimal, integer).
|
||||
const sharesMatch = table.match(/<nonDerivativeShares>\s*([\d]+)/);
|
||||
const shares = sharesMatch ? parseInt(sharesMatch[1], 10) : 0;
|
||||
|
||||
// Extract price per share.
|
||||
const priceMatch = table.match(/<priceOrStrike\s*Price>\s*([\d,.]+)/);
|
||||
const price = priceMatch ? parseFloat(priceMatch[1].replace(/,/g, '')) : 0;
|
||||
|
||||
transactions.push({
|
||||
reporter,
|
||||
relationship,
|
||||
securityTitle,
|
||||
transactionDate,
|
||||
transactionCode,
|
||||
shares,
|
||||
price,
|
||||
});
|
||||
}
|
||||
|
||||
return transactions;
|
||||
}
|
||||
|
||||
Reference in New Issue
Block a user