{
  "schemaVersion": "1.0",
  "statusAsOf": "2026-07-20",
  "canonicalUrl": "https://www.david-medeiros.com/archive-status/",
  "machineReadableUrl": "https://www.david-medeiros.com/datasets/archive-status.json",
  "methodology": "Counts are taken from the current static build data, verified release records, and named source manifests. Completion claims are limited to what those sources prove. Search indexing, AI use, manual accessibility review, legal conclusions, and unreviewed OCR are reported separately.",
  "verifiedReleaseSnapshot": {
    "publishedFiles": 5872,
    "htmlPages": 3452,
    "indexableHtmlPages": 3246,
    "intentionalNoindexOrAliasPages": 206,
    "sitemapFiles": 7,
    "uniqueSitemapUrls": 5220,
    "pdfFiles": 425
  },
  "currentSourceCounts": {
    "mediaRecords": 1236,
    "imageFiles": 1230,
    "localVideoFiles": 6,
    "automatedOcrAttemptRecords": 672,
    "imagesWithNonemptyAutomatedOcr": 662,
    "imagesWithoutNonemptyAutomatedOcr": 568,
    "reviewedOcrTranscripts": 0,
    "youtubeRecords": 357,
    "youtubeTranscriptsPublishedOnSite": 0,
    "muckrockVerifiedRequests": 170,
    "muckrockClaimedRequests": 200,
    "muckrockUnresolvedClaimedRequests": 30,
    "peopleCandidates": 75,
    "peopleWithPublishedReferences": 29,
    "peopleWithDiscoverySignals": 8,
    "publishedPeopleReferences": 645,
    "unreviewedPeopleDiscoverySignals": 95
  },
  "workstreams": [
    {
      "id": "muckrock-reconstruction",
      "label": "MuckRock request reconstruction",
      "status": "partial",
      "completedCount": 170,
      "targetCount": 200,
      "unit": "requests",
      "boundary": "A verified request page does not prove that every communication and attachment in its complete paper trail has been recovered.",
      "href": "/muckrock-requests/"
    },
    {
      "id": "image-ocr",
      "label": "Image OCR coverage",
      "status": "partial",
      "completedCount": 662,
      "targetCount": 1230,
      "unit": "images with non-empty automated OCR",
      "boundary": "Automated OCR is an unreviewed discovery aid, not a verified transcript. Readers must compare it with the source image.",
      "href": "/media/"
    },
    {
      "id": "youtube-transcripts",
      "label": "YouTube transcript publication",
      "status": "incomplete",
      "completedCount": 0,
      "targetCount": 357,
      "unit": "reviewed transcripts published on this site",
      "boundary": "Video titles and embeds are published, but they are not substitutes for synchronized captions or reviewed transcripts.",
      "href": "/youtube-videos/"
    },
    {
      "id": "people-index",
      "label": "People Index cross-references",
      "status": "partial",
      "completedCount": 29,
      "targetCount": 75,
      "unit": "conservative candidates with published-record references",
      "boundary": "The candidate registry is not an exhaustive extraction of every person in every PDF, image, video, FOIA thread, or article. A reference is not a finding of wrongdoing.",
      "href": "/people/"
    },
    {
      "id": "pdf-review",
      "label": "PDF editorial and accessibility review",
      "status": "partial",
      "completedCount": null,
      "targetCount": 425,
      "unit": "published PDFs",
      "boundary": "All 425 PDFs are public and discoverable in the verified release snapshot. Complete OCR, metadata, tagging, reading-order, and manual accessibility review is not yet proven.",
      "href": "/pdf-document-index/"
    },
    {
      "id": "external-indexing",
      "label": "External search and AI indexing",
      "status": "external",
      "completedCount": null,
      "targetCount": null,
      "unit": "third-party systems",
      "boundary": "The archive can publish crawlable pages, sitemaps, manifests, and structured data. It cannot guarantee crawling, indexing, ranking, retention, citation, or interpretation by a third party.",
      "href": "/verification/"
    }
  ],
  "sourceNotes": [
    {
      "label": "Media catalog",
      "path": "/media-manifest.json",
      "status": "current source manifest"
    },
    {
      "label": "MuckRock request archive",
      "path": "/datasets/muckrock-requests.json",
      "status": "current verified reconstruction"
    },
    {
      "label": "People cross-reference index",
      "path": "/datasets/people-index.json",
      "status": "current conservative candidate registry"
    },
    {
      "label": "Sitemap index",
      "path": "/sitemap-index.xml",
      "status": "current crawler discovery index"
    }
  ],
  "explicitLimitations": [
    "No percentage is reported when the project lacks a defensible completed-item count.",
    "Automated OCR is kept separate from reviewed transcription.",
    "Published and sitemapped does not mean indexed by a search engine.",
    "Automated accessibility checks do not replace manual assistive-technology testing.",
    "A SHA-256 digest verifies bytes, not authorship, context, truth, or legal effect.",
    "The referenced G: source drive was unavailable during the latest MuckRock reconciliation."
  ]
}