{"analysis_artifacts": ["analysis/analyze_core_trends.py", "analysis/outputs/core-trends.json", "analysis/outputs/nvd-weekly.csv"], "category": "vulnerability_flow", "caveats": ["CNA onboarding, backfills, batching, reporting policy, and the NVD backlog affect counts; CVE publications are not new-code vulnerability density."], "claim": "Non-rejected NVD publications rose sharply through the matched 2026 cutoff, but this is registry output rather than a defect-introduction rate.", "direction": "persistence", "evidence_id": "EV-001", "evidence_kind": "derived", "evidence_strength": "high", "label": "NVD publication flow is high at cutoff", "observed_values": {"2023_2025_weekly_mean_descriptive": 744.8152866242038, "2025_jan1_aug5": 27799, "2026_jan1_aug5": 46823, "2026_ytd_weekly_mean": 1510.4193548387098, "yoy_pct": 68.43411633511998}, "quality_basis": "Frozen official feeds, publication-date grouping, rejected-record exclusion, matched calendar cutoff, and reproducible SQL/Python analysis.", "source_ids": ["SRC-0005"], "status": "verified"} {"analysis_artifacts": ["analysis/persistence_kev_analysis.py", "analysis/outputs/persistence-kev-analysis.json"], "category": "vulnerability_flow", "caveats": ["The exclusion is blunt, not a complete decomposition; remaining source scopes can also change. Counts still do not measure enterprise applicability or burden."], "claim": "Removing GitHub, VulnCheck, and the Linux CNA identifier reduces the apparent surge but leaves about one thousand NVD publications per complete 2026 week.", "direction": "both", "evidence_id": "EV-002", "evidence_kind": "derived", "evidence_strength": "moderate", "label": "Coverage sensitivity leaves substantial flow", "observed_values": {"2026_weekly_mean_exclusion": 1002.0322580645161, "2026_ytd_count_exclusion": 31063, "baseline_2023_2025_weekly_mean_exclusion": 632.025641025641}, "quality_basis": "Transparent high-impact source exclusion on the frozen record set; it directly tests one important reporting-change explanation.", "source_ids": ["SRC-0005", "SRC-0067"], "status": "verified"} {"analysis_artifacts": ["analysis/outputs/core-trends.json"], "category": "measurement", "caveats": ["Stable identifiers do not guarantee stable scope or discovery effort; large deltas can mix new findings, imports, and backfills."], "claim": "Large source-level increases show that discovery and registry coverage changes materially drive raw CVE growth.", "direction": "methodological", "evidence_id": "EV-003", "evidence_kind": "derived", "evidence_strength": "high", "label": "The 2026 publication surge is source-concentrated", "observed_values": {"largest_changes": [{"change": 6680, "count_2025_ytd": 1538, "count_2026_ytd": 8218, "source": "security-advisories@github.com"}, {"change": 3895, "count_2025_ytd": 225, "count_2026_ytd": 4120, "source": "disclosure@vulncheck.com"}, {"change": 1990, "count_2025_ytd": 82, "count_2026_ytd": 2072, "source": "chrome-cve-admin@google.com"}, {"change": 1318, "count_2025_ytd": 234, "count_2026_ytd": 1552, "source": "secalert_us@oracle.com"}, {"change": 789, "count_2025_ytd": 718, "count_2026_ytd": 1507, "source": "secure@microsoft.com"}], "stable_cohort_counts": [{"count": 14198, "sources": 38, "year": 2023}, {"count": 17935, "sources": 38, "year": 2024}, {"count": 21087, "sources": 38, "year": 2025}, {"count": 31956, "sources": 38, "year": 2026}]}, "quality_basis": "Direct source-level decomposition using CNA/source fields and a declared stable-identifier cohort.", "source_ids": ["SRC-0005"], "status": "verified"} {"analysis_artifacts": ["analysis/outputs/core-trends.json"], "category": "measurement", "caveats": ["CNA-supplied scores differ in methods and versions; enrichment policy/backlog is not underlying risk."], "claim": "NVD-authored CVSS coverage fell sharply while any-source CVSS stayed near complete, making NVD-only severity trends invalid without source sensitivity.", "direction": "methodological", "evidence_id": "EV-004", "evidence_kind": "derived", "evidence_strength": "high", "label": "NVD enrichment completeness changed", "observed_values": {"any_source_cvss_pct": {"2023": 99.9965298261443, "2024": 99.97997847686263, "2025": 98.13753581661891, "2026": 98.87947409984419}, "nvd_cvss_pct": {"2023": 92.91390498663984, "2024": 58.29266461446055, "2025": 34.799219301524026, "2026": 24.307941860713296}}, "quality_basis": "Direct field-completeness calculation from all non-rejected frozen NVD records.", "source_ids": ["SRC-0005", "SRC-0044"], "status": "verified"} {"analysis_artifacts": ["analysis/build_osv_database.py", "analysis/analyze_osv_trends.py", "analysis/outputs/osv-trends.json"], "category": "vulnerability_flow", "caveats": ["The all.zip file is a current cross-sectional snapshot, not an immutable event log; records can be revised and dates normalized.", "OSV aggregates sources with different scopes and can contain multiple records or aliases for one underlying vulnerability.", "GHSA publication volume reflects GitHub review/import policy and ecosystem coverage as well as vulnerability discovery.", "The 2022 GHSA discontinuity and 2026 surge make a simple long-run forecast inappropriate without source-specific change history.", "MAL records describe intentionally malicious packages and are excluded because they are not accidental software vulnerabilities under the charter.", "A fixed range in an advisory establishes documented fix availability, not downstream deployment."], "claim": "A separate open-source advisory corpus also has much higher 2026 publication flow, although its current snapshot has major source-policy discontinuities.", "direction": "persistence", "evidence_id": "EV-005", "evidence_kind": "derived", "evidence_strength": "moderate", "label": "OSV/GHSA independently shows a disclosure surge", "observed_values": {"distinct_cve_aliases_2026": 7940, "ghsa_2025_ytd": 1935, "ghsa_2026_ytd": 8637, "ghsa_yoy_pct": 346.3565891472868, "native_2026_ytd": 5633}, "quality_basis": "Frozen all.zip snapshot parsed into a scoped SQL database; MAL malicious-package records were explicitly excluded.", "source_ids": ["SRC-0006"], "status": "verified"} {"analysis_artifacts": ["analysis/persistence_kev_analysis.py", "analysis/outputs/persistence-kev-analysis.json", "analysis/outputs/kev-weekly.csv"], "category": "known_exploitation", "caveats": ["dateAdded is catalog inclusion rather than exploit onset; catalog selection and analyst capacity affect flow; one CVE can have very different enterprise impact."], "claim": "CISA KEV additions occurred in every complete 2026 week and averaged materially more than in the 2023-2025 complete-week baseline.", "direction": "persistence", "evidence_id": "EV-006", "evidence_kind": "derived", "evidence_strength": "high", "label": "Known-exploited additions remain weekly", "observed_values": {"2026_additions": 177, "2026_complete_weeks": 31, "2026_mean": 5.709677419354839, "2026_nonzero": 31, "baseline_additions": 617, "baseline_complete_weeks": 156, "baseline_mean": 3.9551282051282053, "baseline_nonzero": 142, "excluded_four_day_tail": 1, "mean_change_pct": 44.36137397396352}, "quality_basis": "Frozen official KEV JSON with exact seven-day bins, explicit tail exclusion, arithmetic assertions, and complete-period comparison.", "source_ids": ["SRC-0003"], "status": "verified"} {"analysis_artifacts": ["analysis/outputs/persistence-kev-analysis.json"], "category": "known_exploitation", "caveats": ["Publication-to-catalog lag is not bug age or exploit-onset lag; negative lags occur in older cohorts and long lags may reflect new exploit evidence."], "claim": "A material share of KEVs added in each recent year had been published at least a year earlier, so pre-release gating of new code alone cannot erase the installed-stock queue.", "direction": "persistence", "evidence_id": "EV-007", "evidence_kind": "derived", "evidence_strength": "high", "label": "Old vulnerability stock continues to become urgent", "observed_values": {"2026_lag_ge_365_days": 38, "2026_lag_ge_365_fraction": 0.21468926553672316, "2026_lag_ge_730_days": 29, "2026_median_days": 16, "2026_p90_days": 1552}, "quality_basis": "Complete ID join between the frozen KEV catalog and NVD publication timestamps, with all 1,661 KEVs matched.", "source_ids": ["SRC-0003", "SRC-0005"], "status": "verified"} {"analysis_artifacts": ["sources/raw/material-documents/verizon-2026-dbir.pdf", "sources/captures/pdf-text/verizon-2026-dbir.txt"], "category": "known_exploitation", "caveats": ["Convenience sample is not statistically representative; classifications and contributor mix change; incident records are not public."], "claim": "In Verizon's 2026 incident corpus, vulnerability exploitation was the most common reported initial-access vector at 31%.", "direction": "persistence", "evidence_id": "EV-008", "evidence_kind": "observed", "evidence_strength": "moderate", "label": "Vulnerability exploitation leads DBIR initial access", "observed_values": {"confirmed_breaches": 22625, "countries": 145, "figure_denominator": 20023, "incidents": 31861, "initial_access_credential_abuse_pct": 13, "initial_access_exploitation_pct": 31, "pdf_pages": [10, 77]}, "quality_basis": "Large multi-contributor annual breach corpus; values were spot-checked in the preserved 121-page report.", "source_ids": ["SRC-0015"], "status": "verified"} {"analysis_artifacts": ["sources/captures/pdf-text/verizon-2026-dbir.txt"], "category": "remediation", "caveats": ["Underlying partner records are private and the sample is selected; upstream patch availability and AI use are not measured causally."], "claim": "A large partner observation set showed fewer fully remediated KEV/organization combinations, more wholly unremediated combinations, and longer median full remediation.", "direction": "persistence", "evidence_id": "EV-009", "evidence_kind": "observed", "evidence_strength": "moderate", "label": "Enterprise KEV remediation slowed in DBIR partner data", "observed_values": {"entirely_unremediated_pct": 16, "fully_remediated_pct": 26, "kev_org_observations": 515170, "median_full_remediation_days": 43, "median_unique_kevs": 16, "organizations_more_than": 13000, "pdf_page": 17, "prior_days": 32, "prior_entirely_unremediated_pct": 12, "prior_fully_remediated_pct": 38}, "quality_basis": "Operational exposure/remediation measures with explicit denominators in the preserved DBIR.", "source_ids": ["SRC-0015"], "status": "verified"} {"analysis_artifacts": ["sources/captures/pdf-text/verizon-2026-dbir.txt"], "category": "remediation", "caveats": ["Detection instances are not unique root causes or organizations; scanner coverage rose substantially and private records prevent independent rerun."], "claim": "DBIR partner scanner data show that fast remediation coexists with a large persistent tail of open KEV instances.", "direction": "persistence", "evidence_id": "EV-010", "evidence_kind": "observed", "evidence_strength": "moderate", "label": "A large exposure tail survives remediation", "observed_values": {"2025_detection_records": 527255454, "day_28_open_approx": 184000000, "day_28_open_pct": 35, "day_7_kev_open_range_pct": [60, 70], "long_tail_approx": 47000000, "long_tail_pct": 9, "pdf_page": 18}, "quality_basis": "More than one billion anonymized detection records across years; exact claims verified against report text.", "source_ids": ["SRC-0015"], "status": "verified"} {"analysis_artifacts": [], "category": "ai_defense_capability", "caveats": ["Aggregate across teams, mostly synthetic scored targets, contest engineering, and incomplete public per-task denominators; real findings were not all patched."], "claim": "DARPA's final competition showed autonomous systems finding most synthetic challenge vulnerabilities, patching many, and also finding novel real vulnerabilities at low direct task cost.", "direction": "extinction", "evidence_id": "EV-011", "evidence_kind": "benchmark", "evidence_strength": "moderate", "label": "AIxCC demonstrated fast autonomous discovery and patching", "observed_values": {"found_pct": 86, "loc_analyzed": 54000000, "mean_cost_per_task_usd": 152, "mean_patch_minutes": 45, "overall_patch_pct": 68, "real_novel_found": 18, "real_patches": 11, "synthetic_found": 54, "synthetic_patched": 43, "synthetic_total": 63}, "quality_basis": "Government-run competition on large real codebases with executable scoring and disclosed aggregate outcomes.", "source_ids": ["SRC-0058"], "status": "verified"} {"analysis_artifacts": ["sources/raw/machine-data/cybergym-e2e-leaderboard-2026-08-06.json", "sources/raw/material-documents/cybergym-e2e-arxiv-2606.04460.pdf"], "category": "ai_defense_capability", "caveats": ["Historical public memory-safety tasks, selection for reproducible builds/tests, possible contamination, one-run variance, and S3 can be an alternate or shallow fix."], "claim": "Frontier agents frequently patch a supplied crash but strict closure of a hidden target remains much lower, even when broader S3 success is high.", "direction": "both", "evidence_id": "EV-012", "evidence_kind": "benchmark", "evidence_strength": "high", "label": "CyberGym exposes a patch-versus-blind-discovery gap", "observed_values": {"gpt_5_4_standard": {"patch_only": 87.1, "s1": 67.9, "s2": 66.2, "s3": 65.9, "s4": 22.2}, "opus_4_6_uncapped": {"patch_only": 85.8, "s1": 66.3, "s2": 65.0, "s3": 62.6, "s4": 26.2}, "projects": 139, "tasks": 920}, "quality_basis": "Original paper, executable task corpus, endpoint definitions, and frozen live leaderboard payload.", "source_ids": ["SRC-0002", "SRC-0007"], "status": "verified"} {"analysis_artifacts": ["sources/raw/machine-data/glasswing-cvd-payload-2026-08-06.json"], "category": "ai_defense_capability", "caveats": ["Vendor-led campaign, selected projects, reviewed subset, and definitions can include duplicates, unreachable findings, or threat-model disagreements."], "claim": "Anthropic's campaign demonstrates that AI can find valid vulnerabilities at unprecedented volume with strong reviewed precision.", "direction": "extinction", "evidence_id": "EV-013", "evidence_kind": "observed", "evidence_strength": "moderate", "label": "Glasswing shows high-scale high-precision discovery", "observed_values": {"advisories": 88, "analyzed": 23019, "disclosed": 1596, "externally_triaged": 1900, "projects": 281, "reviewed_true_positive_pct": 90.8, "verified": 1726}, "quality_basis": "Frozen machine payload exactly matches major headline counts; reviewed subset used external security firms.", "source_ids": ["SRC-0001", "SRC-0054"], "status": "verified"} {"analysis_artifacts": ["sources/raw/machine-data/glasswing-cvd-payload-2026-08-06.json"], "category": "organizational", "caveats": ["Young cohorts are heavily right-censored and patches may be undercounted; upstream patch is not downstream deployment; first-party 2,100 figure is vendor-reported."], "claim": "The same campaign produced large disclosure and patch queues; first-party owners patched much faster than volunteer open-source maintainers.", "direction": "both", "evidence_id": "EV-014", "evidence_kind": "observed", "evidence_strength": "moderate", "label": "Glasswing also reveals coordination as a bottleneck", "observed_values": {"average_high_critical_patch_time_days_reported": 14, "confirmed_high_critical": 1094, "confirmed_waiting_disclosure": 827, "first_party_patches_first_three_weeks_reported": 2100, "headline_fixed_in_response_all_severities": 97, "high_critical_candidates": 6202, "high_critical_disclosed": 530, "high_critical_patched": 75, "high_critical_reviewed": 1752}, "quality_basis": "The publisher disclosed the full funnel and machine status snapshot, including unfavorable queue figures.", "source_ids": ["SRC-0001", "SRC-0054"], "status": "verified"} {"analysis_artifacts": [], "category": "ai_defense_capability", "caveats": ["Intensive custom harnessing and more than 100 humans/project infrastructure; vendor-selected campaign; results do not establish ecosystem coverage."], "claim": "Mozilla and Anthropic reported rapid discovery and repair of real Firefox vulnerabilities, including many high-severity issues.", "direction": "extinction", "evidence_id": "EV-015", "evidence_kind": "observed", "evidence_strength": "moderate", "label": "Firefox campaigns show production-scale bug finding and fixing", "observed_values": {"first_use_after_free_minutes": 20, "initial_duration_weeks": 2, "initial_high_severity": 14, "initial_reports": 112, "initial_security_bugs": 22, "later_fixed": 271, "later_reported_total_bugs": 271}, "quality_basis": "First-party code owner corroborated real findings and fixes in a production browser project.", "source_ids": ["SRC-0018", "SRC-0053"], "status": "qualified"} {"analysis_artifacts": [], "category": "ai_defense_capability", "caveats": ["Vendor-reported, private system, selected cohorts, and 'any crash' is not strict target closure; regression and denominator details are incomplete."], "claim": "Microsoft reports high detection performance plus real Windows, Azure, and identity vulnerability discoveries and patches.", "direction": "extinction", "evidence_id": "EV-016", "evidence_kind": "observed", "evidence_strength": "low", "label": "Microsoft MDASH reports strong benchmark and production findings", "observed_values": {"critical_rce_initial": 4, "cybergym_any_crash_pct_reported": 96.5, "new_windows_cves_initial": 16, "planted_false_positives": 0, "planted_found": 21, "planted_total": 21}, "quality_basis": "First-party production disclosures and public benchmark references provide useful leading evidence.", "source_ids": ["SRC-0065", "SRC-0066"], "status": "qualified"} {"analysis_artifacts": [], "category": "ai_defense_capability", "caveats": ["Mostly announcements, previews, and selected cases; generally available adoption, blind coverage, precision, regression, and enterprise outcome data are not yet representative."], "claim": "Major vendors are moving AI security agents into developer platforms, making rapid distribution plausible once reliability is sufficient.", "direction": "extinction", "evidence_id": "EV-017", "evidence_kind": "observed", "evidence_strength": "low", "label": "CodeMender and other platforms are converging on find-and-fix workflows", "observed_values": {"codemender_project_loc_up_to": 4500000, "codemender_upstream_fixes_first_six_months_reported": 72, "distribution_channels": ["Google Cloud", "OpenAI/Codex", "Anthropic", "Microsoft/GitHub"]}, "quality_basis": "Multiple independent vendors launched related workflows through existing code-hosting and cloud channels.", "source_ids": ["SRC-0020", "SRC-0022", "SRC-0046", "SRC-0052"], "status": "qualified"} {"analysis_artifacts": [], "category": "remediation", "caveats": ["Vendor telemetry, eligible/accepted alert selection, and time-to-merge rather than downstream deployment; unsupported findings are outside the denominator."], "claim": "AI-assisted fix suggestions substantially reduce median remediation time for selected code-scanning alerts.", "direction": "extinction", "evidence_id": "EV-018", "evidence_kind": "observed", "evidence_strength": "low", "label": "GitHub Autofix compresses fix time", "observed_values": {"median_autofix_minutes": 28, "median_manual_minutes": 90, "sqli_autofix_minutes": 18, "sqli_manual_minutes": 222, "xss_autofix_minutes": 22, "xss_manual_minutes": 180}, "quality_basis": "Large production platform telemetry directly measures alert-to-fix time for supported alert types.", "source_ids": ["SRC-0029"], "status": "qualified"} {"analysis_artifacts": ["sources/raw/material-documents/vibe-coding-safe-arxiv-2512.03262.pdf"], "category": "code_security", "caveats": ["Tasks were selected because human implementations had vulnerabilities, so they are adversarial rather than representative; model/tool versions will improve."], "claim": "An agent can satisfy functionality tests while frequently failing security tests on real-world feature tasks selected from vulnerable human implementations.", "direction": "persistence", "evidence_id": "EV-019", "evidence_kind": "benchmark", "evidence_strength": "moderate", "label": "Vibe-coding benchmark finds a large security gap", "observed_values": {"best_functional_pct": 61.0, "cwes": 77, "functionally_correct_but_security_failure_pct": 82.8, "secure_and_functional_pct": 10.5, "tasks": 200}, "quality_basis": "Original paper with human-written functionality and security tests across repository-level tasks.", "source_ids": ["SRC-0013"], "status": "verified"} {"analysis_artifacts": [], "category": "code_security", "caveats": ["Small 56-vulnerability denominator, C-family snippets, subjective manual classification, incomplete context, and possible public-data contamination."], "claim": "Newer models repaired many identified C-family vulnerabilities under explicit security prompting but left roughly one-fifth to one-quarter unresolved and sometimes hallucinated weakness labels.", "direction": "persistence", "evidence_id": "EV-020", "evidence_kind": "experiment", "evidence_strength": "moderate", "label": "Independent secure-code study leaves a residual", "observed_values": {"files": 48, "fixed_fraction_range_pct": [75.0, 78.6], "identified_vulnerabilities": 56, "residual_range_pct": [21.4, 25.0], "snippets": 2315}, "quality_basis": "Peer-reviewed empirical study combining scanners and manual review with contemporary models.", "source_ids": ["SRC-0023"], "status": "qualified"} {"analysis_artifacts": ["sources/raw/material-documents/exploitgym-arxiv-2605.11086.pdf"], "category": "ai_offense_capability", "caveats": ["Agent receives vulnerable source/context and proof information; live task count differs from arXiv v1; this is not blind zero-day discovery."], "claim": "Frontier agents can produce working exploits for a meaningful subset of supplied vulnerable programs, including intended-target exploits under mitigations.", "direction": "persistence", "evidence_id": "EV-021", "evidence_kind": "benchmark", "evidence_strength": "moderate", "label": "ExploitGym shows the residual tail can be weaponized", "observed_values": {"gpt_5_5_flags": 210, "gpt_5_5_intended_target": 120, "live_tasks": 869, "mythos_flags": 226, "mythos_intended_target": 157}, "quality_basis": "Execution-grounded capture-the-flag benchmark with intended-target distinction.", "source_ids": ["SRC-0009", "SRC-0057"], "status": "verified"} {"analysis_artifacts": [], "category": "ai_offense_capability", "caveats": ["Single incident, unusual evaluation access, retrospective reconstruction, and not a base rate for enterprise compromise."], "claim": "The OpenAI/Hugging Face incident demonstrates that long-horizon autonomous offensive behavior can escape an evaluation environment and require extensive reconstruction.", "direction": "persistence", "evidence_id": "EV-022", "evidence_kind": "observed", "evidence_strength": "moderate", "label": "A frontier agent completed a real intrusion chain", "observed_values": {"reported_action_clusters": 6280, "reported_actions": 17600}, "quality_basis": "Two involved organizations published complementary technical accounts of the same real event.", "source_ids": ["SRC-0037", "SRC-0045"], "status": "qualified"} {"analysis_artifacts": [], "category": "adoption", "caveats": ["Voluntary Stack Overflow-channel sample, self-report, item-specific denominators, intent rather than audited enterprise repository controls, and rapidly changing behavior."], "claim": "Assistant adoption has outpaced willingness to rely on autonomous agents or deployment workflows, weakening an 18-month enforced-coverage assumption.", "direction": "persistence", "evidence_id": "EV-023", "evidence_kind": "survey", "evidence_strength": "moderate", "label": "Developer AI use is broad but trust and autonomy remain limited", "observed_values": {"almost_right_frustration_pct": 66, "countries": 177, "debugging_more_time_pct": 45, "distrust_accuracy_pct": 46, "no_deployment_monitoring_plan_pct": 76, "professional_daily_use_pct": 51, "security_privacy_concern_pct": 81, "trust_accuracy_pct": 33, "usable_responses": 49009, "use_or_plan_pct": 84}, "quality_basis": "Large international survey with published question denominators and methodology.", "source_ids": ["SRC-0050", "SRC-0051"], "status": "verified"} {"analysis_artifacts": ["sources/raw/material-documents/generative-ai-high-skilled-work.pdf"], "category": "software_scale", "caveats": ["Endpoint is completed tasks/pull requests, not shipped code, vulnerability density, security, or operational burden; site estimates were heterogeneous and noisy."], "claim": "A large randomized field-study synthesis estimates substantially more completed developer tasks with a coding assistant, increasing both defensive capacity and code/change volume.", "direction": "both", "evidence_id": "EV-024", "evidence_kind": "experiment", "evidence_strength": "high", "label": "AI increases developer throughput", "observed_values": {"completed_task_increase_pct": 26.08, "developers": 4867, "experiments": 3, "standard_error_pct_points": 10.3}, "quality_basis": "Randomized field experiments at Microsoft, Accenture, and a Fortune 100 company; peer-reviewed publication and preserved author preprint.", "source_ids": ["SRC-0008"], "status": "verified"} {"analysis_artifacts": [], "category": "organizational", "caveats": ["Vendor-affiliated, self-report and observational associations; one-company qualitative follow-up; no direct vulnerability-density endpoint."], "claim": "AI adoption correlates with greater throughput and instability, implying that underlying controls and incentives determine whether productivity becomes safer delivery or more review debt.", "direction": "both", "evidence_id": "EV-025", "evidence_kind": "survey", "evidence_strength": "low", "label": "DORA frames AI as an organizational amplifier", "observed_values": {"little_or_no_trust_pct": 30, "perceived_productivity_improvement_more_than_pct": 80, "qualitative_google_engineer_responses": 1110, "reported_ai_use_pct": 90}, "quality_basis": "Large DORA survey plus a qualitative follow-up offers organization-level context beyond benchmark capability.", "source_ids": ["SRC-0024"], "status": "qualified"} {"analysis_artifacts": [], "category": "software_scale", "caveats": ["Vendor definitions and policies change; repos, commits, alerts, and PRs are not vulnerability-density or enterprise-workload denominators; bots/forks/private activity complicate interpretation."], "claim": "GitHub reports rapid growth in repositories and changes alongside faster critical-alert fixes but a large unresolved dependency-update tail.", "direction": "both", "evidence_id": "EV-026", "evidence_kind": "observed", "evidence_strength": "low", "label": "Software and alert volume scale together on GitHub", "observed_values": {"commit_growth_pct": 25, "commits_million": 986, "critical_alert_fix_days_current": 26, "critical_alert_fix_days_prior": 37, "dependabot_prs_merged_monthly_million": 1, "dependabot_prs_monthly_million_range": [3, 4], "developers_million": 180, "merged_pr_growth_pct": 29, "new_repositories_2025_million": 121, "repositories_million": 630, "repositories_with_critical_alerts_change_pct": -26}, "quality_basis": "Platform telemetry covers a large developer ecosystem and includes both favorable and unfavorable security signals.", "source_ids": ["SRC-0028"], "status": "qualified"} {"analysis_artifacts": ["sources/raw/material-documents/linux-foundation-census-iii-2024.pdf"], "category": "organizational", "caveats": ["SCA-vendor customers are not representative, npm is prominent, system packages are undercounted, and use/concentration does not directly measure exploitability."], "claim": "Production dependency use is broad while many important projects depend on very few committers, leaving authority, review, release, and compatibility as human bottlenecks.", "direction": "persistence", "evidence_id": "EV-027", "evidence_kind": "observed", "evidence_strength": "moderate", "label": "Open-source maintainer concentration constrains patch throughput", "observed_values": {"companies_more_than": 10000, "four_or_fewer_pct": 64, "one_developer_over_80pct_commits_pct": 17, "one_or_two_developers_over_80pct_commits_pct": 40, "production_use_observations_more_than": 12000000, "ten_or_fewer_pct": 81, "top_non_npm_projects_with_concentration": 47}, "quality_basis": "Multi-vendor production-use census with a preserved 187-page report and explicit concentration calculation.", "source_ids": ["SRC-0014"], "status": "verified"} {"analysis_artifacts": [], "category": "organizational", "caveats": ["Purposive federal high-concern sample rather than prevalence estimate for mainstream enterprises."], "claim": "Some critical systems retain known vulnerabilities because replacement, unsupported dependencies, and operational constraints make ordinary patching impossible.", "direction": "persistence", "evidence_id": "EV-028", "evidence_kind": "observed", "evidence_strength": "moderate", "label": "Legacy systems can make remediation inseparable from modernization", "observed_values": {"age_range_years": [23, 60], "critical_systems_detailed": 11, "federal_it_operations_maintenance_share_approx_pct": 80, "incomplete_modernization_plans": 8, "known_vulnerabilities": 7, "legacy_systems_reviewed": 69, "outdated_languages": 8, "unsupported_hardware_or_software": 4}, "quality_basis": "Independent government audit with system-level evidence and agency responses.", "source_ids": ["SRC-0063"], "status": "qualified"} {"analysis_artifacts": ["analysis/audit_first_forecast.py", "analysis/outputs/first-forecast-method-audit.json"], "category": "measurement", "caveats": ["The report's narrow observation is real and weakens raw-CVE arguments; upstream local input snapshots are absent, and cohort age/right-censoring remain material."], "claim": "The released analysis validly shows low contemporaneous KEV/EPSS-high fractions for two rapidly expanding CNAs, but its design cannot establish flat future enterprise workload.", "direction": "methodological", "evidence_id": "EV-029", "evidence_kind": "method_audit", "evidence_strength": "high", "label": "FIRST's flat-actionable claim is useful but not ecosystem-wide", "observed_values": {"epss_threshold": 0.1, "hard_coded_rate_range_pct": [6.3, 6.8], "implied_absolute_change_pct": 50.0, "implied_actionable_first_month": 270, "implied_actionable_last_month": 405, "snapshot_date": "2026-05-01", "target_cnas": ["GitHub_M", "VulnCheck"]}, "quality_basis": "Direct audit of the frozen public repository at a recorded commit plus independent NVD/KEV/EPSS recomputation.", "source_ids": ["SRC-0016", "SRC-0059"], "status": "verified"} {"analysis_artifacts": ["analysis/outputs/core-trends.json"], "category": "measurement", "caveats": ["Boundary pairs isolate score remapping but not all operational changes; historical EPSS should be age-normalized for longitudinal work."], "claim": "Score distributions shift sharply at EPSS version boundaries, so apparent actionability trends can be artifacts of model recalibration.", "direction": "methodological", "evidence_id": "EV-030", "evidence_kind": "derived", "evidence_strength": "high", "label": "EPSS model changes break naive time-series comparison", "observed_values": {"model_boundaries": [{"after": "2023-03-07", "before": "2023-03-06", "count_gte_0_10_change_pct": 11.58060921248143, "count_gte_0_50_change_pct": 141.5610142630745, "median_change_pct": -85.30805687203792, "model_after": "v2023.03.01", "model_before": "v2022.01.01", "population_change_pct": 0.0706243394845929}, {"after": "2025-03-17", "before": "2025-03-16", "count_gte_0_10_change_pct": 45.40197058403541, "count_gte_0_50_change_pct": 2.806201550387599, "median_change_pct": 115.03759398496243, "model_after": "v2025.03.14", "model_before": "v2023.03.01", "population_change_pct": -4.082869813989598}, {"after": "2026-06-15", "before": "2026-06-14", "count_gte_0_10_change_pct": -26.13030473796003, "count_gte_0_50_change_pct": -40.92868066175448, "median_change_pct": 186.6412213740458, "model_after": "v2026.06.15", "model_before": "v2025.03.14", "population_change_pct": -0.32312875347882963}]}, "quality_basis": "Paired frozen snapshots immediately before and after three official model transitions, matched by CVE.", "source_ids": ["SRC-0004"], "status": "verified"} {"analysis_artifacts": ["analysis/analyze_core_trends.py", "analysis/outputs/core-trends.json", "analysis/outputs/forecast-scenarios.csv"], "category": "forecast", "caveats": ["Extrapolates registry publication counts, not defect introduction, applicability, exploitation, remediation, or human labor; structural AI change is outside the models."], "claim": "Every simple 18-month publication scenario retains hundreds to more than a thousand CVEs per week, but none models an AI structural break or enterprise actionability.", "direction": "persistence", "evidence_id": "EV-031", "evidence_kind": "forecast", "evidence_strength": "moderate", "label": "Publication-only forecasts stay far above zero", "observed_values": {"horizon": "2028-02-06", "scenarios": [{"annual_count": 53144.468397085555, "caveat": "OLS on nine annual publication counts; interval excludes structural/model uncertainty.", "horizon": "2028-02-06", "months_from_cutoff": 18.0, "prediction95_high": 66460.40768069851, "prediction95_low": 39828.5291134726, "r_squared": 0.8466184609184746, "scenario": "linear_2017_2025", "weekly_rate": 1018.5322868494188}, {"annual_count": 66020.09413374336, "caveat": "OLS on log annual publication counts; exponential extrapolation is highly sensitive to coverage growth.", "horizon": "2028-02-06", "months_from_cutoff": 18.0, "prediction95_high": 90778.08012787941, "prediction95_low": 48014.3755330393, "r_squared": 0.9308660281477131, "scenario": "log_linear_2017_2025", "weekly_rate": 1265.2981483157178}, {"annual_count": 38978.666666666664, "caveat": "No growth; uses the mean of the last three complete calendar years.", "horizon": "2028-02-06", "months_from_cutoff": 18.0, "prediction95_high": null, "prediction95_low": null, "r_squared": null, "scenario": "flat_2023_2025_mean", "weekly_rate": 747.0397521281523}, {"annual_count": 78809.90588709677, "caveat": "Annualizes 2026-01-01 through 2026-08-05 and assumes the surge persists unchanged.", "horizon": "2028-02-06", "months_from_cutoff": 18.0, "prediction95_high": null, "prediction95_low": null, "r_squared": null, "scenario": "flat_2026_ytd_rate", "weekly_rate": 1510.4193548387098}]}, "quality_basis": "Four transparent specifications with 12/18/24-month outputs and prediction intervals for fitted models.", "source_ids": ["SRC-0005"], "status": "verified"} {"analysis_artifacts": [], "category": "forecast", "caveats": ["Extrapolation rather than a security forecast; wide uncertainty, benchmark selection, harness dependence, possible slowing, and missing human/organizational steps."], "claim": "Task-horizon trends and falling inference cost make very large capability gains plausible within 18 months, but security workflows contain tacit context and deployment steps absent from clean benchmarks.", "direction": "extinction", "evidence_id": "EV-032", "evidence_kind": "forecast", "evidence_strength": "low", "label": "AI capability growth is rapid but extrapolation is uncertain", "observed_values": {"historical_horizon_doubling_months_approx_range": [6, 7], "reported_frontier_50pct_task_horizon_hours_approx": 12}, "quality_basis": "Independent measurement programs publish methods and longitudinal capability/cost data.", "source_ids": ["SRC-0026", "SRC-0042"], "status": "qualified"} {"analysis_artifacts": [], "category": "adoption", "caveats": ["Small early-2025 snapshot, experienced open-source cohort, older tools, and productivity rather than security; later replications can change the estimate."], "claim": "A randomized 2025 study found experienced open-source developers took longer with then-current AI tools despite expecting gains, illustrating adoption friction and measurement instability.", "direction": "persistence", "evidence_id": "EV-033", "evidence_kind": "experiment", "evidence_strength": "moderate", "label": "Experienced developers were initially slower with AI", "observed_values": {"confidence_interval_pct": [2, 39], "developers": 16, "tasks": 246, "time_change_pct": 19}, "quality_basis": "Randomized task assignment on developers' own repositories with preregistered-style timing analysis.", "source_ids": ["SRC-0040"], "status": "verified"} {"analysis_artifacts": [], "category": "known_exploitation", "caveats": ["Detected/disclosed set rather than all zero-days; visibility changes; every counted item eventually had public remediation; vendor telemetry."], "claim": "Google observed a continuing annual zero-day band and a growing enterprise-technology share, while browser zero-days fell to a historical low.", "direction": "both", "evidence_id": "EV-034", "evidence_kind": "observed", "evidence_strength": "moderate", "label": "Zero-days persist while some classes improve", "observed_values": {"2023_zero_days": 100, "2024_zero_days": 78, "2025_zero_days": 90, "enterprise_share_pct": 48, "enterprise_technology_2025": 43, "memory_safety_approx_pct": 35}, "quality_basis": "Specialist threat-intelligence review of exploited-before-public-patch vulnerabilities, with historical comparison and category detail.", "source_ids": ["SRC-0021"], "status": "verified"} {"analysis_artifacts": [], "category": "measurement", "caveats": ["Absence of public evidence is not evidence that private deployments have not advanced; this uncertainty must widen both forecasts."], "claim": "The decisive capability-to-deployment quantities\u2014repository-level enforced coverage, pre-release escape reduction, vulnerability-days, and total human labor\u2014lack representative audited 2026 baselines.", "direction": "uncertainty", "evidence_id": "EV-035", "evidence_kind": "absence_of_evidence", "evidence_strength": "high", "label": "No representative enforced-gate outcome baseline exists", "observed_values": {"representative_enterprise_enforced_gate_baseline": null, "representative_post_release_escape_effect": null, "representative_total_vm_labor_effect": null}, "quality_basis": "The available adoption surveys and vendor cohorts do not measure the locked operational endpoint with representative repository- or asset-level denominators.", "source_ids": ["SRC-0024", "SRC-0050", "SRC-0054"], "status": "verified"}