feat(queue): adapter queue overhaul — pause/resume, error capture, scheduling
- Pause/resume persisted in queue_state; drain loop checks isPaused() and skips - Error capture: every failed/backoff attempt logged to queue_errors with the full stack trace; admin queue page expands a failed job to stream it - Retry controls: retryJob(key), retrySource(kind), clearDone(olderThanMs) - Per-source scheduling: queue_schedules table + 30s enqueueDueSchedules loop (seed defaults sec-fetch 24h, yfinance 5min); admin UI lists/adds/deletes - Startup recovery: interrupted in_flight jobs reset to pending on boot - fix(edgar): archive URLs use the filer CIK (accession-number prefix), not the company CIK — resolves Cloudflare 429 that left SEC backfills sparse/empty - Migration: add queue_errors, queue_schedules, queue_state tables plus error/ scheduled_for columns on adapter_queue (idempotent ALTER on startup) Verified: full sec-fetch backfill now succeeds for all watched symbols (NVDA 14,675 institution filings, CIFR 274 insider txns, TSLA 6,011, etc.). 503 backend tests pass. Co-Authored-By: Claude <noreply@anthropic.com>
This commit is contained in:
co-authored by
Claude
parent
55f07e6b42
commit
ca385c1960
@@ -17,22 +17,23 @@ const OPERATOR_EMAIL = process.env.SEC_OPERATOR_EMAIL ?? 'research@example.com';
|
||||
const UA = `Investor Flow (${OPERATOR_EMAIL})`;
|
||||
|
||||
/**
|
||||
* Token-bucket rate limiter: 8 tokens, refilled at 8/sec, min 125ms between
|
||||
* calls. Guarantees we stay under EDGAR's 10 req/s rule with headroom.
|
||||
* Token-bucket rate limiter: 6 tokens, refilled at 6/sec, min ~167ms between
|
||||
* calls. Stays within EDGAR's 10 req/s rule with headroom.
|
||||
*/
|
||||
class TokenBucket {
|
||||
private tokens = 8;
|
||||
private tokens = 6;
|
||||
private lastDrain = Date.now();
|
||||
private static readonly MAX_TOKENS = 6;
|
||||
private static readonly REFILL_RATE = 6; // tokens/sec
|
||||
|
||||
async wait(): Promise<void> {
|
||||
const now = Date.now();
|
||||
const elapsed = (now - this.lastDrain) / 1000;
|
||||
// Refill tokens up to maxRate (8).
|
||||
this.tokens = Math.min(8, this.tokens + elapsed * 8);
|
||||
this.tokens = Math.min(TokenBucket.MAX_TOKENS, this.tokens + elapsed * TokenBucket.REFILL_RATE);
|
||||
this.lastDrain = now;
|
||||
|
||||
if (this.tokens < 1) {
|
||||
const waitMs = Math.ceil(((1 - this.tokens) / 8) * 1000);
|
||||
const waitMs = Math.ceil(((1 - this.tokens) / TokenBucket.REFILL_RATE) * 1000);
|
||||
await new Promise((r) => setTimeout(r, waitMs));
|
||||
this.tokens = 0;
|
||||
this.lastDrain = Date.now();
|
||||
@@ -47,34 +48,46 @@ const bucket = new TokenBucket();
|
||||
/**
|
||||
* Fetch a URL with EDGAR-compliant headers, rate limiting, and ETag caching.
|
||||
* Returns null on 304 (caller should return cached row).
|
||||
* Retries with exponential backoff on 429 (rate limited).
|
||||
*/
|
||||
async function edgarFetch(
|
||||
url: string,
|
||||
extraHeaders?: Record<string, string>,
|
||||
retries = 3,
|
||||
): Promise<{ status: number; headers: { etag?: string | null; lastModified?: string | null }; body: unknown } | null> {
|
||||
await bucket.wait();
|
||||
for (let attempt = 1; attempt <= retries; attempt++) {
|
||||
await bucket.wait();
|
||||
|
||||
const headers: Record<string, string> = {
|
||||
'User-Agent': UA,
|
||||
Accept: 'application/json',
|
||||
...extraHeaders,
|
||||
};
|
||||
const headers: Record<string, string> = {
|
||||
'User-Agent': UA,
|
||||
Accept: 'application/json',
|
||||
...extraHeaders,
|
||||
};
|
||||
|
||||
const resp = await fetch(url, { method: 'GET', headers });
|
||||
const resp = await fetch(url, { method: 'GET', headers });
|
||||
|
||||
const etag = resp.headers.get('etag');
|
||||
const lastModified = resp.headers.get('last-modified');
|
||||
const etag = resp.headers.get('etag');
|
||||
const lastModified = resp.headers.get('last-modified');
|
||||
|
||||
if (resp.status === 304) {
|
||||
return { status: 304, headers: { etag, lastModified }, body: null };
|
||||
if (resp.status === 304) {
|
||||
return { status: 304, headers: { etag, lastModified }, body: null };
|
||||
}
|
||||
|
||||
if (resp.status === 429 && attempt < retries) {
|
||||
const backoff = Math.pow(2, attempt) * 1000;
|
||||
await new Promise((r) => setTimeout(r, backoff));
|
||||
continue;
|
||||
}
|
||||
|
||||
if (resp.status >= 400) {
|
||||
throw new Error(`EDGAR ${resp.status} ${resp.statusText} for ${url}`);
|
||||
}
|
||||
|
||||
const body = (await resp.json()) as unknown;
|
||||
return { status: resp.status, headers: { etag, lastModified }, body };
|
||||
}
|
||||
|
||||
if (resp.status >= 400) {
|
||||
throw new Error(`EDGAR ${resp.status} ${resp.statusText} for ${url}`);
|
||||
}
|
||||
|
||||
const body = (await resp.json()) as unknown;
|
||||
return { status: resp.status, headers: { etag, lastModified }, body };
|
||||
throw new Error(`EDGAR max retries (${retries}) exceeded for ${url}`);
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
@@ -160,12 +173,34 @@ export class EdgarAdapter implements SourceFetch {
|
||||
throw new Error(`EDGAR 304 but no cached data for ${key}`);
|
||||
}
|
||||
|
||||
const rawData = resp?.body as { name?: string; filings?: { recent?: Array<{ form?: string; dateReporter?: string; accessionNumber?: string; accessionNormalization?: string; reportDate?: string; reportFile?: string; primaryDocument?: string }> } } | undefined;
|
||||
const recentFilings = rawData?.filings?.recent;
|
||||
if (!recentFilings) {
|
||||
const rawData = resp?.body as { name?: string; filings?: { recent?: unknown } } | undefined;
|
||||
const recentObj = rawData?.filings?.recent;
|
||||
if (!recentObj) {
|
||||
throw new Error(`EDGAR filings_index: no recent filings for CIK${padded}`);
|
||||
}
|
||||
|
||||
// SEC EDGAR returns recent filings as either:
|
||||
// (a) an array of objects: [{form, dateReporter, ...}, ...] (current format)
|
||||
// (b) parallel arrays: {form: [...], dateReporter: [...], ...} (legacy)
|
||||
let recentFilings: Array<Record<string, unknown>>;
|
||||
if (Array.isArray(recentObj)) {
|
||||
recentFilings = recentObj as Array<Record<string, unknown>>;
|
||||
} else if (typeof recentObj === 'object') {
|
||||
// Parallel arrays — convert to array of objects.
|
||||
const keys = Object.keys(recentObj);
|
||||
const len = (recentObj[keys[0]] as unknown[])?.length ?? 0;
|
||||
recentFilings = [];
|
||||
for (let i = 0; i < len; i++) {
|
||||
const row: Record<string, unknown> = {};
|
||||
for (const k of keys) {
|
||||
row[k] = (recentObj as Record<string, unknown[]>)[k]?.[i];
|
||||
}
|
||||
recentFilings.push(row);
|
||||
}
|
||||
} else {
|
||||
throw new Error(`EDGAR filings_index: unexpected recent filings format for CIK${padded}`);
|
||||
}
|
||||
|
||||
// Filter by form type.
|
||||
let filings = recentFilings;
|
||||
if (opts?.formTypes && opts.formTypes.length > 0) {
|
||||
@@ -178,7 +213,7 @@ export class EdgarAdapter implements SourceFetch {
|
||||
const from = opts.dateRange.from ? new Date(opts.dateRange.from).getTime() : null;
|
||||
const to = opts.dateRange.to ? new Date(opts.dateRange.to).getTime() : null;
|
||||
filings = filings.filter((f) => {
|
||||
const ts = new Date(f.dateReporter ?? f.accessionNormalization ?? '').getTime();
|
||||
const ts = new Date(f.reportDate ?? f.filingDate ?? f.dateReporter ?? '').getTime();
|
||||
if (Number.isNaN(ts)) return false;
|
||||
if (from !== null && ts < from) return false;
|
||||
if (to !== null && ts > to) return false;
|
||||
@@ -318,7 +353,8 @@ export class EdgarAdapter implements SourceFetch {
|
||||
cik: string,
|
||||
accession: string,
|
||||
): Promise<FetchResult> {
|
||||
const padded = padCik(cik);
|
||||
const filerCik = accession.split('-')[0];
|
||||
const padded = padCik(filerCik);
|
||||
const accessionNoDashes = accession.replace(/-/g, '');
|
||||
|
||||
// --- Step 1: filing index (JSON) ---------------------------------------
|
||||
@@ -332,29 +368,49 @@ export class EdgarAdapter implements SourceFetch {
|
||||
const indexBody = indexResp.body as {
|
||||
fileDate?: string;
|
||||
documents?: Array<{ name?: string; type?: string; size?: string | number; path?: string }>;
|
||||
directory?: { item?: Array<{ name?: string; type?: string; size?: string | number }> };
|
||||
partialSubmissionIndicator?: unknown;
|
||||
};
|
||||
|
||||
if (!indexBody?.documents || indexBody.documents.length === 0) {
|
||||
// EDGAR API may return documents as `documents` (old) or `directory.item` (new)
|
||||
const docList = indexBody?.documents ?? indexBody?.directory?.item ?? [];
|
||||
|
||||
if (docList.length === 0) {
|
||||
throw new Error(`EDGAR 13F: no documents in index for ${padded}/${accessionNoDashes}`);
|
||||
}
|
||||
|
||||
// Pick the primary document (usually the .txt or .xml filing).
|
||||
const primaryDoc = indexBody.documents.find(
|
||||
(d) => d.type === '13F' || d.type === '13F-infoTable'
|
||||
) ?? indexBody.documents[0];
|
||||
// Pick candidates: XML documents (excluding index files), sorted by size desc
|
||||
const xmlCandidates = docList
|
||||
.filter((d) => d.name?.toLowerCase().endsWith('.xml') && !d.name?.includes('index'))
|
||||
.sort((a, b) => {
|
||||
const sa = typeof a.size === 'string' ? parseInt(a.size, 10) || 0 : (a.size as number) ?? 0;
|
||||
const sb = typeof b.size === 'string' ? parseInt(b.size, 10) || 0 : (b.size as number) ?? 0;
|
||||
return sb - sa;
|
||||
});
|
||||
|
||||
const primaryName = primaryDoc.name;
|
||||
if (!primaryName) {
|
||||
throw new Error(`EDGAR 13F: primary document has no name for ${padded}/${accessionNoDashes}`);
|
||||
// Try each candidate until we find one with holdings; fall back to any non-HTML doc
|
||||
let docText = '';
|
||||
let parsed: Array<{ cusip: string; issuerName: string; value: number; sshPrnamt: number }> = [];
|
||||
for (const cand of xmlCandidates) {
|
||||
const docUrl = `https://www.sec.gov/Archives/edgar/data/${padded}/${accessionNoDashes}/${cand.name}`;
|
||||
docText = await this.edgarXmlFetch(docUrl);
|
||||
parsed = parse13fHoldings(docText);
|
||||
if (parsed.length > 0) break;
|
||||
}
|
||||
|
||||
// --- Step 2: fetch the primary document (XML/HTML) ---------------------
|
||||
const docUrl = `https://www.sec.gov/Archives/edgar/data/${padded}/${accessionNoDashes}/${primaryName}`;
|
||||
const docText = await this.edgarXmlFetch(docUrl);
|
||||
// Fallback: try first non-HTML doc if XML candidates had no holdings
|
||||
if (parsed.length === 0) {
|
||||
const fallback = docList.find(
|
||||
(d) => !d.name?.toLowerCase().endsWith('.html') && !d.name?.includes('index')
|
||||
) ?? docList[0];
|
||||
if (fallback && fallback.name) {
|
||||
const docUrl = `https://www.sec.gov/Archives/edgar/data/${padded}/${accessionNoDashes}/${fallback.name}`;
|
||||
docText = await this.edgarXmlFetch(docUrl);
|
||||
parsed = parse13fHoldings(docText);
|
||||
}
|
||||
}
|
||||
|
||||
// --- Step 3: parse holdings table --------------------------------------
|
||||
const holdings = parse13fHoldings(docText);
|
||||
const holdings = parsed;
|
||||
|
||||
return {
|
||||
value: { holdings, accession: `${padded}/${accessionNoDashes}` },
|
||||
@@ -389,7 +445,8 @@ export class EdgarAdapter implements SourceFetch {
|
||||
cik: string,
|
||||
accession: string,
|
||||
): Promise<FetchResult> {
|
||||
const padded = padCik(cik);
|
||||
const filerCik = accession.split('-')[0];
|
||||
const padded = padCik(filerCik);
|
||||
const accessionNoDashes = accession.replace(/-/g, '');
|
||||
|
||||
// --- Step 1: filing index (JSON) ---------------------------------------
|
||||
@@ -403,17 +460,21 @@ export class EdgarAdapter implements SourceFetch {
|
||||
const indexBody = indexResp.body as {
|
||||
fileDate?: string;
|
||||
documents?: Array<{ name?: string; type?: string; size?: string | number; path?: string }>;
|
||||
directory?: { item?: Array<{ name?: string; type?: string; size?: string | number }> };
|
||||
partialSubmissionIndicator?: unknown;
|
||||
};
|
||||
|
||||
if (!indexBody?.documents || indexBody.documents.length === 0) {
|
||||
// EDGAR API may return documents as `documents` (old) or `directory.item` (new)
|
||||
const docList = indexBody?.documents ?? indexBody?.directory?.item ?? [];
|
||||
|
||||
if (docList.length === 0) {
|
||||
throw new Error(`EDGAR Form 4: no documents in index for ${padded}/${accessionNoDashes}`);
|
||||
}
|
||||
|
||||
// Form 4 is typically filed as a single XML. Pick the .xml document.
|
||||
const xmlDoc = indexBody.documents.find(
|
||||
const xmlDoc = docList.find(
|
||||
(d) => d.name?.toLowerCase().endsWith('.xml')
|
||||
) ?? indexBody.documents[0];
|
||||
) ?? docList[0];
|
||||
|
||||
const xmlName = xmlDoc.name;
|
||||
if (!xmlName) {
|
||||
@@ -446,24 +507,35 @@ export class EdgarAdapter implements SourceFetch {
|
||||
* Fetch a URL that returns XML/HTML (not JSON), using the rate limiter.
|
||||
* EDGAR returns XML for primary filing documents; we strip the body here
|
||||
* rather than calling `resp.json()`.
|
||||
* Retries with exponential backoff on 429 (rate limited).
|
||||
*/
|
||||
async edgarXmlFetch(url: string): Promise<string> {
|
||||
await bucket.wait();
|
||||
async edgarXmlFetch(url: string, retries = 3): Promise<string> {
|
||||
for (let attempt = 1; attempt <= retries; attempt++) {
|
||||
await bucket.wait();
|
||||
|
||||
const resp = await fetch(url, {
|
||||
method: 'GET',
|
||||
headers: {
|
||||
'User-Agent': UA,
|
||||
Accept: 'application/xml, text/xml, */*',
|
||||
},
|
||||
});
|
||||
const resp = await fetch(url, {
|
||||
method: 'GET',
|
||||
headers: {
|
||||
'User-Agent': UA,
|
||||
Accept: 'application/xml, text/xml, */*',
|
||||
},
|
||||
});
|
||||
|
||||
if (resp.status >= 400) {
|
||||
throw new Error(`EDGAR XML ${resp.status} ${resp.statusText} for ${url}`);
|
||||
if (resp.status === 429 && attempt < retries) {
|
||||
const backoff = Math.pow(2, attempt) * 1000;
|
||||
await new Promise((r) => setTimeout(r, backoff));
|
||||
continue;
|
||||
}
|
||||
|
||||
if (resp.status >= 400) {
|
||||
throw new Error(`EDGAR XML ${resp.status} ${resp.statusText} for ${url}`);
|
||||
}
|
||||
|
||||
const text = await resp.text();
|
||||
return text;
|
||||
}
|
||||
|
||||
const text = await resp.text();
|
||||
return text;
|
||||
throw new Error(`EDGAR XML max retries (${retries}) exceeded for ${url}`);
|
||||
}
|
||||
|
||||
/** SourceFetch.fetchOne dispatch. */
|
||||
@@ -512,7 +584,30 @@ function parse13fHoldings(text: string): Array<{
|
||||
}> {
|
||||
const holdings: Array<{ cusip: string; issuerName: string; value: number; sshPrnamt: number }> = [];
|
||||
|
||||
// Try XML-style <table> parsing first (13F filings are often XML).
|
||||
// Try 13F XML format: <infoTable> (possibly namespaced) with <nameOfIssuer>, <cusip>, <value>, <sshPrnamt>
|
||||
// SEC 13F XML uses namespaces like <ns1:infoTable><ns1:nameOfIssuer>...
|
||||
const infoTableMatches = text.match(/<(?:\w+:)?infoTable[^>]*>([\s\S]*?)<\/(?:\w+:)?infoTable>/gi);
|
||||
if (infoTableMatches && infoTableMatches.length > 0) {
|
||||
for (const block of infoTableMatches) {
|
||||
const nameMatch = block.match(/<(?:\w+:)?nameOfIssuer>\s*([\s\S]*?)\s*<\/(?:\w+:)?nameOfIssuer>/i);
|
||||
const cusipMatch = block.match(/<(?:\w+:)?cusip>\s*([\s\S]*?)\s*<\/(?:\w+:)?cusip>/i);
|
||||
const valueMatch = block.match(/<(?:\w+:)?value>\s*([\s\S]*?)\s*<\/(?:\w+:)?value>/i);
|
||||
const sharesMatch = block.match(/<(?:\w+:)?sshPrnamt>\s*([\s\S]*?)\s*<\/(?:\w+:)?sshPrnamt>/i);
|
||||
|
||||
if (!cusipMatch) continue;
|
||||
|
||||
const issuerName = nameMatch ? nameMatch[1].replace(/<[^>]+>/g, '').trim() : '';
|
||||
const cusip = cusipMatch ? cusipMatch[1].replace(/<[^>]+>/g, '').trim() : '';
|
||||
const value = valueMatch ? parseFloat(valueMatch[1].replace(/<[^>]+>/g, '').replace(/,/g, '')) || 0 : 0;
|
||||
const sshPrnamt = sharesMatch ? parseFloat(sharesMatch[1].replace(/<[^>]+>/g, '').replace(/,/g, '')) || 0 : 0;
|
||||
|
||||
holdings.push({ cusip, issuerName, value, sshPrnamt });
|
||||
}
|
||||
|
||||
if (holdings.length > 0) return holdings;
|
||||
}
|
||||
|
||||
// Try HTML-style <table> parsing (older 13F filings may use HTML tables).
|
||||
const xmlMatch = text.match(/<table[^>]*>([\s\S]*?)<\/table>/gi);
|
||||
if (xmlMatch) {
|
||||
for (const tbl of xmlMatch) {
|
||||
@@ -596,11 +691,16 @@ function parse13fHoldings(text: string): Array<{
|
||||
/**
|
||||
* Extract transactions from a Form 4 XML filing.
|
||||
*
|
||||
* Form 4 uses an XML structure with <infotable> elements containing:
|
||||
* <reportOwner>, <securityTitle>, <transactionDate>,
|
||||
* <transactionCode>, <shares>, <pricePerShare>
|
||||
* Modern EDGAR Form 4 uses <ownershipDocument> with:
|
||||
* <reportingOwner><reportingOwnerId><rptOwnerName>...</rptOwnerName>...
|
||||
* <nonDerivativeTable><nonDerivativeTransaction>...
|
||||
* <securityTitle><value>...</value></securityTitle>
|
||||
* <transactionDate><value>YYYY-MM-DD</value></transactionDate>
|
||||
* <transactionCoding><transactionCode>...</transactionCode>...
|
||||
* <transactionAmounts><transactionShares><value>...</value>...
|
||||
* <transactionPricePerShare><value>...</value>...
|
||||
*
|
||||
* This is a thin parser that extracts the key fields.
|
||||
* The old <infotable> format is also handled as a fallback.
|
||||
*/
|
||||
function parseForm4Transactions(text: string): Array<{
|
||||
reporter: string;
|
||||
@@ -621,12 +721,57 @@ function parseForm4Transactions(text: string): Array<{
|
||||
price: number;
|
||||
}> = [];
|
||||
|
||||
// Extract infotable blocks.
|
||||
// Try modern format: <nonDerivativeTable> with <nonDerivativeTransaction> blocks
|
||||
const hasNonDerivative = /<nonDerivativeTable>/i.test(text);
|
||||
if (hasNonDerivative) {
|
||||
// Get reporting owner info from document root (shared across all transactions)
|
||||
const ownerNameMatch = text.match(/<rptOwnerName>\s*([\s\S]*?)<\/rptOwnerName>/i);
|
||||
const reporter = ownerNameMatch ? ownerNameMatch[1].replace(/<[^>]+>/g, '').trim() : '';
|
||||
|
||||
const relBlockMatch = text.match(/<reportingOwnerRelationship>([\s\S]*?)<\/reportingOwnerRelationship>/i);
|
||||
let relationship = '';
|
||||
if (relBlockMatch) {
|
||||
const relBlock = relBlockMatch[1];
|
||||
const parts: string[] = [];
|
||||
if (/<isDirector>\s*1/i.test(relBlock)) parts.push('Director');
|
||||
if (/<isOfficer>\s*1/i.test(relBlock)) {
|
||||
const titleMatch = relBlock.match(/<officerTitle>\s*([\s\S]*?)<\/officerTitle>/i);
|
||||
parts.push(titleMatch ? titleMatch[1].replace(/<[^>]+>/g, '').trim() : 'Officer');
|
||||
}
|
||||
if (/<isTenPercentOwner>\s*1/i.test(relBlock)) parts.push('10% Owner');
|
||||
if (/<isOther>\s*1/i.test(relBlock)) parts.push('Other');
|
||||
relationship = parts.join(', ');
|
||||
}
|
||||
|
||||
const txBlocks = text.match(/<nonDerivativeTransaction[\s\S]*?<\/nonDerivativeTransaction>/gi);
|
||||
if (txBlocks) {
|
||||
for (const block of txBlocks) {
|
||||
const titleMatch = block.match(/<securityTitle[\s\S]*?<value>\s*([\s\S]*?)\s*<\/value>/i);
|
||||
const securityTitle = titleMatch ? titleMatch[1].replace(/<[^>]+>/g, '').trim() : '';
|
||||
|
||||
const dateMatch = block.match(/<transactionDate[\s\S]*?<value>\s*(\d{4}-\d{2}-\d{2})/);
|
||||
const transactionDate = dateMatch ? dateMatch[1] : '';
|
||||
|
||||
const codeMatch = block.match(/<transactionCode>\s*([A-HV])/i);
|
||||
const transactionCode = codeMatch ? codeMatch[1].toUpperCase() : '';
|
||||
|
||||
const sharesMatch = block.match(/<transactionShares[\s\S]*?<value>\s*([\d]+)/i);
|
||||
const shares = sharesMatch ? parseInt(sharesMatch[1], 10) : 0;
|
||||
|
||||
const priceMatch = block.match(/<transactionPricePerShare[\s\S]*?<value>\s*([\d,.]+)/i);
|
||||
const price = priceMatch ? parseFloat(priceMatch[1].replace(/,/g, '')) : 0;
|
||||
|
||||
transactions.push({ reporter, relationship, securityTitle, transactionDate, transactionCode, shares, price });
|
||||
}
|
||||
}
|
||||
return transactions;
|
||||
}
|
||||
|
||||
// Fallback: old <infotable> format.
|
||||
const infotables = text.match(/<infotable[\s\S]*?<\/infotable>/gi);
|
||||
if (!infotables) return transactions;
|
||||
|
||||
for (const table of infotables) {
|
||||
// Extract reporter CIK and name from <rptOwner>.
|
||||
const rptOwnerMatch = table.match(/<rptOwner>([\s\S]*?)<\/rptOwner>/i);
|
||||
let reporter = '';
|
||||
if (rptOwnerMatch) {
|
||||
@@ -639,7 +784,6 @@ function parseForm4Transactions(text: string): Array<{
|
||||
}
|
||||
}
|
||||
|
||||
// Extract relationship from <rptOwnerRelationship>.
|
||||
const relBlockMatch = table.match(/<rptOwnerRelationship>([\s\S]*?)<\/rptOwnerRelationship>/i);
|
||||
let relationship = '';
|
||||
if (relBlockMatch) {
|
||||
@@ -655,35 +799,22 @@ function parseForm4Transactions(text: string): Array<{
|
||||
relationship = parts.join(', ');
|
||||
}
|
||||
|
||||
// Extract security title.
|
||||
const titleMatch = table.match(/<securityTitle>\s*([\s\S]*?)<\/securityTitle>/i);
|
||||
const securityTitle = titleMatch ? titleMatch[1].trim() : '';
|
||||
|
||||
// Extract transaction date.
|
||||
const dateMatch = table.match(/<transactionDate>\s*(\d{4}-\d{2}-\d{2})/);
|
||||
const transactionDate = dateMatch ? dateMatch[1] : '';
|
||||
|
||||
// Extract transaction code.
|
||||
const codeMatch = table.match(/<transactionCode>\s*([A-HV])/i);
|
||||
const transactionCode = codeMatch ? codeMatch[1].toUpperCase() : '';
|
||||
|
||||
// Extract shares (non-decimal, integer).
|
||||
const sharesMatch = table.match(/<nonDerivativeShares>\s*([\d]+)/);
|
||||
const shares = sharesMatch ? parseInt(sharesMatch[1], 10) : 0;
|
||||
|
||||
// Extract price per share.
|
||||
const priceMatch = table.match(/<priceOrStrike\s*Price>\s*([\d,.]+)/);
|
||||
const price = priceMatch ? parseFloat(priceMatch[1].replace(/,/g, '')) : 0;
|
||||
|
||||
transactions.push({
|
||||
reporter,
|
||||
relationship,
|
||||
securityTitle,
|
||||
transactionDate,
|
||||
transactionCode,
|
||||
shares,
|
||||
price,
|
||||
});
|
||||
transactions.push({ reporter, relationship, securityTitle, transactionDate, transactionCode, shares, price });
|
||||
}
|
||||
|
||||
return transactions;
|
||||
|
||||
Reference in New Issue
Block a user