{
  "dataset": {
    "approved_record_count": 11,
    "author": "Mike Ye",
    "canonical_url": "https://strategicsignal.ai/ai-data-transactions/",
    "methodology_version": "1.0.0",
    "name": "Strategic Signal — Corporate AI Data Transactions",
    "notes": "11 approved observations; 5 legacy observations remain provisional and excluded from approved benchmarks. Filter publication_review.status=human_reviewed for approved longitudinal measures.",
    "provisional_record_count": 5,
    "published_on": "2026-09-15",
    "publisher": "Strategic Signal",
    "record_count": 16,
    "review_standard": "AI-discovered; human approval required. Legacy observations remain provisional until approved.",
    "updated_on": "2026-09-30",
    "version": "1.0.15"
  },
  "transactions": [
    {
      "access_structure": "Astex, Bristol Myers Squibb and Takeda joined the existing AbbVie and Johnson & Johnson initiative. Local computation permits model learning without pooling the underlying records.",
      "analysis_boundary": "March 27, 2025 founding is context, not a second transaction here. September 14, 2026 performance evidence is a later update. Uplift is consortium-reported, not independently replicated; no price is disclosed.",
      "announcement_date": "2025-10-01",
      "asset_disposition": "Embedded / Continuing",
      "buyer_ai_developer": [
        "AISB Federated OpenFold3 Initiative",
        "AlQuraishi Lab",
        "Apheris"
      ],
      "canonical_path": "/ai-data-transactions/deals/ss-aidt-2025-0001/",
      "confidence": "high",
      "corpus_description": "Proprietary experimentally determined protein–small molecule structures supplied for federated fine-tuning. This observation records the October 2025 consortium expansion.",
      "corpus_renewal": "Renewable Corpus",
      "data_category": [
        "Scientific / pharmaceutical",
        "Protein–small molecule structures"
      ],
      "date_added": "2026-09-15",
      "date_last_reviewed": "2026-09-16",
      "decision_outcome_richness": "Analysis: experimental structures directly constrain molecular predictions; they are not a complete history of clinical decisions or drug-development outcomes.",
      "disclosed_consideration": {
        "amount": null,
        "currency": null,
        "description": "No data-specific consideration disclosed.",
        "status": "not_disclosed"
      },
      "economics_scope": "not_disclosed",
      "estimated_economics": {
        "amount": null,
        "currency": null,
        "method": null,
        "status": "not_estimated"
      },
      "event_history": [
        {
          "count_as_new_transaction": false,
          "date": "2025-03-27",
          "event": "Founding initiative"
        },
        {
          "count_as_new_transaction": true,
          "date": "2025-10-01",
          "event": "Three new contributors join the existing initiative"
        },
        {
          "count_as_new_transaction": false,
          "date": "2026-09-14",
          "event": "Consortium-reported performance update"
        }
      ],
      "historical_depth": "not_disclosed",
      "industry": "Pharmaceuticals & biotechnology",
      "likely_model_use": "Fine-tune OpenFold3 for protein–ligand prediction.",
      "likely_strategic_value": "Analysis: multi-owner scientific archives can supply differentiated training examples while retaining source-data control.",
      "m_and_a_implications": "Analysis: rights to learn can be valuable without an asset sale. Reported predictive improvement is not a cash valuation or evidence of clinical benefit.",
      "model_uplift": {
        "baseline": "OpenFold3 Preview 2",
        "evaluation_n": 1056,
        "evidence_status": "consortium_reported_not_independently_replicated",
        "limitations": "Private held-out evaluation; consortium-funded work; clinical utility and independent replication not established.",
        "metrics": [
          {
            "baseline_pct": 35.6,
            "change_percentage_points": 16.5,
            "name": "fraction PL-lDDT >= 0.8",
            "updated_pct": 52.1
          },
          {
            "baseline_pct": 28.9,
            "change_percentage_points": 17.9,
            "name": "fraction ligand bisyRMSD <= 2 angstrom",
            "updated_pct": 46.8
          }
        ],
        "reported_on": "2026-09-14",
        "training_structures": 20167
      },
      "primary_transaction_structure": "Federated / Controlled Learning Rights",
      "refreshability": "continuing_network; contribution schedule not disclosed",
      "restrictions": {
        "de_identification": "not_disclosed",
        "employee_customer_data": "not_applicable",
        "geographic_sovereignty": "not_disclosed",
        "governed_access": "Local computation and contributor-defined access policies.",
        "privacy_pii": "Confidential scientific data; no patient-data entitlement inferred.",
        "trade_secret": "Raw structures stay under contributor control."
      },
      "rights": {
        "derived_model": "yes_model_improvement; current network describes member ownership, not full contract terms",
        "exclusivity": "not_disclosed",
        "inference_retrieval": "unknown",
        "ownership_transfer": "no_disclosed_transfer",
        "post_training": "yes_explicit_fine_tuning",
        "retention": "raw_data_remains_with_contributors; model retention terms not_disclosed",
        "sublicensing": "not_disclosed",
        "training": "yes_explicit"
      },
      "scores": {
        "asset": {
          "coverage_pct": 84,
          "factor_evidence": {
            "decision_outcome_richness": {
              "basis": "analysis",
              "rationale": "A structure records an experimental result, but does not establish downstream commercial or clinical outcomes.",
              "source_ids": [
                "E1"
              ]
            },
            "decision_value_density": {
              "basis": "analysis",
              "rationale": "Better molecular prioritization can influence expensive research choices; a qualitative domain judgment.",
              "source_ids": [
                "E1"
              ]
            },
            "domain_value": {
              "basis": "analysis",
              "rationale": "Drug-discovery research makes predictive accuracy economically consequential.",
              "source_ids": [
                "E1"
              ]
            },
            "model_learning_usefulness": {
              "basis": "analysis",
              "rationale": "The learning objective is closely matched to the described scientific observations.",
              "source_ids": [
                "E1"
              ]
            },
            "non_replicability": {
              "basis": "analysis",
              "rationale": "Reproducing experimental observations requires laboratory effort; cost is not quantified.",
              "source_ids": [
                "E1"
              ]
            },
            "proprietary_advantage": {
              "basis": "analysis",
              "rationale": "Combining otherwise separate private archives offers access unavailable from public corpora alone.",
              "source_ids": [
                "E1"
              ]
            },
            "real_world_grounding": {
              "basis": "analysis",
              "rationale": "The described observations originate in physical experiments rather than generated scenarios.",
              "source_ids": [
                "E1"
              ]
            },
            "rights_usability": {
              "basis": "analysis",
              "rationale": "Local learning is explicitly enabled, while undisclosed downstream terms limit certainty.",
              "source_ids": [
                "E1"
              ]
            },
            "uniqueness_scarcity": {
              "basis": "analysis",
              "rationale": "Nonpublic experimental observations are hard to substitute with generic text.",
              "source_ids": [
                "E1"
              ]
            }
          },
          "factor_ratings": {
            "decision_outcome_richness": 3.5,
            "decision_value_density": 4.5,
            "domain_value": 4.5,
            "historical_depth": null,
            "model_learning_usefulness": 4.5,
            "non_replicability": 4,
            "proprietary_advantage": 4.5,
            "real_world_grounding": 5,
            "refreshability": null,
            "rights_usability": 4,
            "uniqueness_scarcity": 4.5
          },
          "rationale": "Strong experimental grounding and learning fit. Structural results are not full decision histories. Unknown archive age and contribution cadence remain unscored.",
          "score": 87
        },
        "methodology_version": "1.0.0",
        "transaction_signal": {
          "coverage_pct": 85,
          "factor_evidence": {
            "clean_price_discovery": {
              "basis": "analysis",
              "rationale": "No observable price or auction is provided by this announcement.",
              "source_ids": [
                "E1"
              ]
            },
            "data_consideration_separability": {
              "basis": "analysis",
              "rationale": "No separately quantified data consideration is available.",
              "source_ids": [
                "E1"
              ]
            },
            "disclosed_economics": {
              "basis": "analysis",
              "rationale": "The reviewed announcement supplies no monetary amount usable for valuation.",
              "source_ids": [
                "E1"
              ]
            },
            "explicit_model_use": {
              "basis": "analysis",
              "rationale": "Fine-tuning is the stated purpose, not an inference from product deployment.",
              "source_ids": [
                "E1"
              ]
            },
            "precedent_value": {
              "basis": "analysis",
              "rationale": "A multi-owner structure offers a repeatable pattern for otherwise inaccessible scientific corpora.",
              "source_ids": [
                "E1"
              ]
            },
            "rights_clarity": {
              "basis": "analysis",
              "rationale": "The permitted learning method is clear; the full contractual allocation remains unknown.",
              "source_ids": [
                "E1"
              ]
            },
            "strategic_buyer_quality": {
              "basis": "analysis",
              "rationale": "Established research participants support execution relevance, not evidence of a market-clearing price.",
              "source_ids": [
                "E1"
              ]
            }
          },
          "factor_ratings": {
            "clean_price_discovery": 0,
            "competing_bids": null,
            "data_consideration_separability": 0,
            "disclosed_economics": 0,
            "explicit_model_use": 5,
            "independent_model_uplift": null,
            "precedent_value": 4.5,
            "rights_clarity": 4,
            "strategic_buyer_quality": 4
          },
          "rationale": "Clear learning arrangement, weak price discovery. Consortium-reported uplift is recorded separately and earns no independent-replication points.",
          "score": 40
        }
      },
      "seller_data_owner": [
        "AbbVie",
        "Johnson & Johnson",
        "Bristol Myers Squibb",
        "Takeda",
        "Astex Pharmaceuticals"
      ],
      "slug": "openfold-pharma-federated-training-consortium",
      "source_code_software_asset": "no",
      "sources": [
        {
          "publisher": "Apheris",
          "title": "AISB initiative expansion — October 1, 2025",
          "type": "primary",
          "url": "https://www.apheris.com/resources/aisb-network-expands-federated-openfold3-initiative-with-three-new-pharma-contributors"
        },
        {
          "publisher": "Apheris",
          "title": "Founding initiative — March 27, 2025",
          "type": "primary",
          "url": "https://www.apheris.com/resources/alquraishi-lab-s-openfold3-to-be-fine-tuned-with-pharma-industry-data-in-a-secure-ai-collaboration"
        },
        {
          "publisher": "Apheris",
          "title": "Consortium performance report — September 14, 2026",
          "type": "primary",
          "url": "https://www.apheris.com/resources/federated-training-dramatically-improves-the-accuracy-of-protein-ligand-co-folding-on-private-pharma-structures"
        },
        {
          "publisher": "Apheris",
          "title": "AISB network governance — current description",
          "type": "primary",
          "url": "https://www.apheris.com/networks/aisb"
        }
      ],
      "transaction_id": "SS-AIDT-2025-0001",
      "transaction_status": "active",
      "publication_review": {
        "status": "human_reviewed",
        "approval_channel": "owner_conversation",
        "reviewer": "Mike Ye",
        "reviewed_at": "2026-09-16T21:38:20.361Z",
        "decision_id": "conversation-20260916-81BF2CAD028B",
        "packet_sha256": "4e1f58d72f53b3ad5140e1c8b01fbb7347a8af99dc3f26fc29a5d28f52422173",
        "record_sha256": "0e4eef8e3f50a13a1e520e21f1545515918d9e29722134c414ceab9f6c02dc16"
      }
    },
    {
      "access_structure": "Reported transfer of a finite archive; detailed contract terms and buyer identity were not disclosed.",
      "analysis_boundary": "The corpus and consideration description are attributed to the former CEO through reporting. Buyer, contract rights, and privacy controls remain unknown.",
      "announcement_date": "2026-04-16",
      "asset_disposition": "Stranded / Separable",
      "buyer_ai_developer": [
        "Undisclosed AI buyer"
      ],
      "canonical_path": "/ai-data-transactions/deals/ss-aidt-2026-0001/",
      "confidence": "medium",
      "corpus_description": "Thirteen years of the shuttered company's Slack messages, internal email, and Jira tickets, reported sold as AI training data.",
      "corpus_renewal": "Finite Corpus",
      "data_category": [
        "Internal communications",
        "Product / engineering workflow",
        "Institutional memory"
      ],
      "date_added": "2026-09-15",
      "date_last_reviewed": "2026-09-15",
      "decision_outcome_richness": "Moderate-to-high: communications and tickets can connect plans, execution, and product outcomes, but outcome labeling is not disclosed.",
      "disclosed_consideration": {
        "amount": null,
        "currency": "USD",
        "description": "Seller described proceeds as hundreds of thousands of dollars; no exact figure was disclosed.",
        "status": "reported_range_only"
      },
      "economics_scope": "reported_dataset_specific",
      "estimated_economics": {
        "amount": null,
        "currency": null,
        "method": null,
        "status": "not_estimated"
      },
      "historical_depth": "13 years (reported)",
      "industry": "Media technology",
      "likely_model_use": "Training AI systems to navigate realistic workplace communication, engineering, and project-management tasks.",
      "likely_strategic_value": "A coherent company history provides temporal and organizational context that synthetic office tasks lack.",
      "m_and_a_implications": "Shows that a failed company's collaboration exhaust may remain separately monetizable after operating value disappears.",
      "primary_transaction_structure": "Data Asset Acquisition",
      "publication_review": {
        "audited_on": "2026-09-16",
        "human_approval_recorded": false,
        "notice": "Sale reporting is supported by indexed Forbes excerpts, but full text and an independent corroborating source remain to be captured. Detailed contract rights are not verified.",
        "status": "legacy_review_required"
      },
      "refreshability": "none_company_shuttered",
      "restrictions": {
        "de_identification": "not_disclosed",
        "employee_customer_data": "not_disclosed",
        "geographic_sovereignty": "not_disclosed",
        "governed_access": "not_disclosed",
        "privacy_pii": "not_disclosed",
        "trade_secret": "not_disclosed"
      },
      "rights": {
        "derived_model": "not_disclosed",
        "exclusivity": "not_disclosed",
        "inference_retrieval": "not_disclosed",
        "ownership_transfer": "unknown",
        "post_training": "not_disclosed",
        "retention": "not_disclosed",
        "sublicensing": "not_disclosed",
        "training": "yes_reported"
      },
      "scores": {
        "asset": {
          "coverage_pct": 0,
          "factor_ratings": {
            "decision_outcome_richness": null,
            "decision_value_density": null,
            "domain_value": null,
            "historical_depth": null,
            "model_learning_usefulness": null,
            "non_replicability": null,
            "proprietary_advantage": null,
            "real_world_grounding": null,
            "refreshability": null,
            "rights_usability": null,
            "uniqueness_scarcity": null
          },
          "rationale": "Legacy score withdrawn pending evidence-linked scoring and human approval. The superseded v1.0.0 release retains the original values.",
          "score": null
        },
        "methodology_version": "1.0.0",
        "transaction_signal": {
          "coverage_pct": 0,
          "factor_ratings": {
            "clean_price_discovery": null,
            "competing_bids": null,
            "data_consideration_separability": null,
            "disclosed_economics": null,
            "explicit_model_use": null,
            "independent_model_uplift": null,
            "precedent_value": null,
            "rights_clarity": null,
            "strategic_buyer_quality": null
          },
          "rationale": "Legacy score withdrawn pending evidence-linked scoring and human approval. The superseded v1.0.0 release retains the original values.",
          "score": null
        }
      },
      "seller_data_owner": [
        "Cielo24"
      ],
      "slug": "cielo24-institutional-memory-sale",
      "source_code_software_asset": "unknown",
      "sources": [
        {
          "publisher": "Forbes",
          "title": "AI's New Training Data: Your Old Work Slacks And Emails",
          "type": "secondary",
          "url": "https://www.forbes.com/sites/annatong/2026/04/16/ais-new-training-data-your-old-work-slacks-and-emails/"
        },
        {
          "publisher": "Fast Company",
          "title": "Shuttered startups are selling old Slack chats and emails to AI companies",
          "type": "secondary",
          "url": "https://www.fastcompany.com/91528808/shuttered-startups-are-selling-old-slack-chats-and-emails-to-ai-companies"
        }
      ],
      "transaction_id": "SS-AIDT-2026-0001",
      "transaction_status": "reported_completed"
    },
    {
      "access_structure": "Bankruptcy asset auction. Google was selected at $10 million; Mercor was alternate at $7.5 million. micro1 later noticed a $12.5 million competing bid. No sale order had been entered as of the review date.",
      "analysis_boundary": "Bid amounts, selected/alternate bidders, asset scope, and pending status are sourced facts. Model-use and valuation implications beyond disclosed purpose are Strategic Signal analysis.",
      "announcement_date": "2026-08-14",
      "asset_disposition": "Stranded / Separable",
      "buyer_ai_developer": [
        "Google LLC (selected bidder; challenged)"
      ],
      "canonical_path": "/ai-data-transactions/deals/ss-aidt-2026-0002/",
      "confidence": "high",
      "corpus_description": "A deidentified archive spanning roughly 34 years, reported to include about 100 million emails, 500 million Teams messages, 30 million lines of source code, operational data, documents, and internally developed software. Customer databases are outside scope.",
      "corpus_renewal": "Finite Corpus",
      "data_category": [
        "Internal communications",
        "Source code and software",
        "Operational records",
        "Financial and workforce records"
      ],
      "date_added": "2026-09-15",
      "date_last_reviewed": "2026-09-15",
      "decision_outcome_richness": "High: communications, code, operations, scheduling, finance, and maintenance-adjacent workflows create a dense cross-functional operating history.",
      "disclosed_consideration": {
        "amount": 10000000,
        "currency": "USD",
        "description": "Google selected bid; court approval pending. Later micro1 notice proposed $12.5 million.",
        "status": "disclosed_bid_not_closed"
      },
      "economics_scope": "dataset_and_related_internal_software_specific",
      "estimated_economics": {
        "amount": null,
        "currency": null,
        "method": null,
        "status": "not_estimated"
      },
      "historical_depth": "Approximately 34 years",
      "industry": "Aviation",
      "likely_model_use": "Train and improve AI systems and productivity products on real enterprise work, software, communication, and operational sequences.",
      "likely_strategic_value": "Large, temporally coherent, multi-modal corporate history with explicit training use and measurable auction demand.",
      "m_and_a_implications": "Provides unusually clean evidence that a defunct company's data and institutional memory can be auctioned separately from its operating business, while exposing privacy and chain-of-title diligence as value-critical.",
      "primary_transaction_structure": "Data Asset Acquisition",
      "publication_review": {
        "audited_on": "2026-09-16",
        "human_approval_recorded": false,
        "notice": "Stretto reports an auction and pending approval through September 11. Treat this as secondary reporting. Underlying filing and current status remain to be verified.",
        "status": "legacy_review_required"
      },
      "refreshability": "none_airline_ceased_operations",
      "restrictions": {
        "de_identification": "required_before_transfer",
        "employee_customer_data": "Customer datasets excluded; treatment of employee and labor data remains contested.",
        "geographic_sovereignty": "not_disclosed_for_google_bid",
        "governed_access": "Independent third-party deidentification contemplated before delivery; final restrictions remain subject to court approval.",
        "privacy_pii": "Customer databases excluded; incidental consumer data to be deidentified. Employee-data objections remain unresolved.",
        "trade_secret": "Contract-counterparty and labor objections remain on file."
      },
      "rights": {
        "derived_model": "yes_intended_use_if_sale_closes",
        "exclusivity": "asset_sale_subject_to_sale_order",
        "inference_retrieval": "not_disclosed",
        "ownership_transfer": "yes_if_sale_closes",
        "post_training": "yes_general_model_improvement",
        "retention": "not_final_pending_sale_order",
        "sublicensing": "not_final_pending_sale_order",
        "training": "yes_explicit"
      },
      "scores": {
        "asset": {
          "coverage_pct": 0,
          "factor_ratings": {
            "decision_outcome_richness": null,
            "decision_value_density": null,
            "domain_value": null,
            "historical_depth": null,
            "model_learning_usefulness": null,
            "non_replicability": null,
            "proprietary_advantage": null,
            "real_world_grounding": null,
            "refreshability": null,
            "rights_usability": null,
            "uniqueness_scarcity": null
          },
          "rationale": "Legacy score withdrawn pending evidence-linked scoring and human approval. The superseded v1.0.0 release retains the original values.",
          "score": null
        },
        "methodology_version": "1.0.0",
        "transaction_signal": {
          "coverage_pct": 0,
          "factor_ratings": {
            "clean_price_discovery": null,
            "competing_bids": null,
            "data_consideration_separability": null,
            "disclosed_economics": null,
            "explicit_model_use": null,
            "independent_model_uplift": null,
            "precedent_value": null,
            "rights_clarity": null,
            "strategic_buyer_quality": null
          },
          "rationale": "Legacy score withdrawn pending evidence-linked scoring and human approval. The superseded v1.0.0 release retains the original values.",
          "score": null
        }
      },
      "seller_data_owner": [
        "Spirit Aviation Holdings / Spirit Airlines"
      ],
      "slug": "spirit-airlines-data-auction-google",
      "source_code_software_asset": "yes",
      "sources": [
        {
          "publisher": "Research Suite by Stretto",
          "title": "Spirit deidentified-data auction results and docket chronology",
          "type": "secondary",
          "url": "https://chapter11cases.com/blogs/news/the-spirit-airlines-deidentified-data-sale-auction-results-objections-and-the-september-30-hearing"
        },
        {
          "publisher": "Reuters",
          "title": "Google to buy Spirit Airlines business data for $10 million",
          "type": "secondary",
          "url": "https://www.reuters.com/legal/litigation/google-buy-spirit-airlines-business-data-10-million-2026-08-17/"
        },
        {
          "publisher": "WIRED",
          "title": "Spirit Airlines Wants to Sell Its Data to Google",
          "type": "secondary",
          "url": "https://www.wired.com/story/spirit-airlines-wants-to-sell-its-data-to-google-former-flight-attendants-are-freaked-out"
        }
      ],
      "transaction_id": "SS-AIDT-2026-0002",
      "transaction_status": "pending_court_approval"
    },
    {
      "access_structure": "Collaborative physical-AI development; detailed data-access boundaries and model-rights allocation were not disclosed.",
      "analysis_boundary": "The announced combination of operational data and robot foundation models is fact. Training, retention, exclusivity, and derived-model rights are unknown.",
      "announcement_date": "2026-09-02",
      "asset_disposition": "Embedded / Continuing",
      "buyer_ai_developer": [
        "FieldAI"
      ],
      "canonical_path": "/ai-data-transactions/deals/ss-aidt-2026-0003/",
      "confidence": "medium_high",
      "corpus_description": "Caterpillar operational data, engineering capability, and industry expertise combined with FieldAI robot foundation models for complex jobsites and manufacturing environments.",
      "corpus_renewal": "Renewable Corpus",
      "data_category": [
        "Industrial operational data",
        "Sensor and robotics data",
        "Engineering knowledge"
      ],
      "date_added": "2026-09-15",
      "date_last_reviewed": "2026-09-15",
      "decision_outcome_richness": "High: autonomous inspections and operations can generate state–action–outcome data in safety-critical environments.",
      "disclosed_consideration": {
        "amount": null,
        "currency": null,
        "description": "No consideration disclosed.",
        "status": "not_disclosed"
      },
      "economics_scope": "not_disclosed",
      "estimated_economics": {
        "amount": null,
        "currency": null,
        "method": null,
        "status": "not_estimated"
      },
      "historical_depth": "not_disclosed",
      "industry": "Industrial equipment & construction",
      "likely_model_use": "Improve robotic autonomy, inspection, situational awareness, and digital-twin systems for industrial environments.",
      "likely_strategic_value": "Connects foundation-model capability to hard-to-reproduce industrial environments and continuous field feedback.",
      "m_and_a_implications": "Industrial incumbents may structure data-bearing co-development instead of selling operating histories, preserving control while giving AI partners domain access.",
      "primary_transaction_structure": "Strategic Model Co-Development",
      "publication_review": {
        "audited_on": "2026-09-16",
        "human_approval_recorded": false,
        "notice": "Caterpillar explicitly mentions operational data and FieldAI foundation models; separately material learning rights are not disclosed. Qualification remains unresolved.",
        "status": "legacy_review_required"
      },
      "refreshability": "live_and_recurring_operational_data_likely",
      "restrictions": {
        "de_identification": "not_disclosed",
        "employee_customer_data": "not_disclosed",
        "geographic_sovereignty": "not_disclosed",
        "governed_access": "not_disclosed",
        "privacy_pii": "not_disclosed",
        "trade_secret": "Engineering and operational data are described but controls are not disclosed."
      },
      "rights": {
        "derived_model": "unknown",
        "exclusivity": "not_disclosed",
        "inference_retrieval": "yes_operational_application",
        "ownership_transfer": "no_disclosed_transfer",
        "post_training": "unknown",
        "retention": "not_disclosed",
        "sublicensing": "not_disclosed",
        "training": "unknown"
      },
      "scores": {
        "asset": {
          "coverage_pct": 0,
          "factor_ratings": {
            "decision_outcome_richness": null,
            "decision_value_density": null,
            "domain_value": null,
            "historical_depth": null,
            "model_learning_usefulness": null,
            "non_replicability": null,
            "proprietary_advantage": null,
            "real_world_grounding": null,
            "refreshability": null,
            "rights_usability": null,
            "uniqueness_scarcity": null
          },
          "rationale": "Legacy score withdrawn pending evidence-linked scoring and human approval. The superseded v1.0.0 release retains the original values.",
          "score": null
        },
        "methodology_version": "1.0.0",
        "transaction_signal": {
          "coverage_pct": 0,
          "factor_ratings": {
            "clean_price_discovery": null,
            "competing_bids": null,
            "data_consideration_separability": null,
            "disclosed_economics": null,
            "explicit_model_use": null,
            "independent_model_uplift": null,
            "precedent_value": null,
            "rights_clarity": null,
            "strategic_buyer_quality": null
          },
          "rationale": "Legacy score withdrawn pending evidence-linked scoring and human approval. The superseded v1.0.0 release retains the original values.",
          "score": null
        }
      },
      "seller_data_owner": [
        "Caterpillar"
      ],
      "slug": "caterpillar-fieldai-industrial-autonomy",
      "source_code_software_asset": "unknown",
      "sources": [
        {
          "publisher": "Caterpillar",
          "title": "Caterpillar and FieldAI Advance AI-Powered Industrial Innovation",
          "type": "primary",
          "url": "https://www.caterpillar.com/en/news/corporate-press-releases/h/caterpillar-and-fieldai-advance-ai-powered-industrial-innovation.html"
        },
        {
          "publisher": "PR Newswire",
          "title": "Caterpillar and FieldAI Advance AI-Powered Industrial Innovation",
          "type": "primary_syndicated",
          "url": "https://www.prnewswire.com/news-releases/caterpillar-and-fieldai-advance-ai-powered-industrial-innovation-302866862.html"
        }
      ],
      "transaction_id": "SS-AIDT-2026-0003",
      "transaction_status": "announced_active"
    },
    {
      "access_structure": "A 33-organization consortium is developing cybersecurity foundation models using member operational and security data. Custody and member-level licensing terms are not disclosed.",
      "analysis_boundary": "The 830 TB planned corpus, contributors and training purpose are company disclosures. Scores are qualitative Strategic Signal analysis, not measured model performance. Federated topology and contract terms are unknown.",
      "announcement_date": "2026-09-03",
      "asset_disposition": "Embedded / Continuing",
      "buyer_ai_developer": [
        "NAVER Cloud consortium",
        "NAVER Cloud",
        "LG AI Research"
      ],
      "canonical_path": "/ai-data-transactions/deals/ss-aidt-2026-0004/",
      "confidence": "high",
      "corpus_description": "Approximately 830 TB of high-quality real-world operational and security data from consortium members across power, finance, telecommunications, science, semiconductors, defense, and aerospace.",
      "corpus_renewal": "Renewable Corpus",
      "data_category": [
        "Cybersecurity records",
        "Critical-infrastructure operations",
        "Industrial systems data"
      ],
      "date_added": "2026-09-15",
      "date_last_reviewed": "2026-09-16",
      "decision_outcome_richness": "Strategic Signal analysis: security operations and planned field trials may connect events to responses; intervention/outcome labels are not disclosed.",
      "disclosed_consideration": {
        "amount": null,
        "currency": null,
        "description": "Government GPU support is disclosed, alongside consortium computing resources. Dataset-specific cash consideration is not disclosed.",
        "status": "in_kind_and_government_compute"
      },
      "economics_scope": "bundled_program_support",
      "estimated_economics": {
        "amount": null,
        "currency": null,
        "method": null,
        "status": "not_estimated"
      },
      "historical_depth": "not_disclosed",
      "industry": "Cybersecurity & critical infrastructure",
      "likely_model_use": "Train defensive and offensive cybersecurity foundation models and validate them in seven critical-industry domains.",
      "likely_strategic_value": "Pools otherwise unobtainable cross-industry operational security data at sovereign scale.",
      "m_and_a_implications": "Consortium structures can aggregate embedded national-infrastructure data without a conventional acquisition, making governance and model rights central valuation terms.",
      "primary_transaction_structure": "Strategic Model Co-Development",
      "publication_review": {
        "approval_channel": "owner_conversation",
        "decision_id": "conversation-20260916-2D6377E44A57",
        "packet_sha256": "0d3511e1d7c9a9ce6cd876440fe1b414e967551ce11a7404834f16cf82849ba7",
        "record_sha256": "9e66511ee6acde56b1620a2e82c7786e746dd09abcef983c751844558ad77bff",
        "reviewed_at": "2026-09-16T21:17:28.097Z",
        "reviewer": "Mike Ye",
        "status": "human_reviewed"
      },
      "refreshability": "consortium_operational_sources_likely_renewable",
      "restrictions": {
        "de_identification": "not_disclosed",
        "employee_customer_data": "not_disclosed",
        "geographic_sovereignty": "Korean program; contractual geographic restrictions not_disclosed",
        "governed_access": "Consortium/government program; detailed member-level controls are not disclosed.",
        "privacy_pii": "not_disclosed",
        "trade_secret": "Operational security and critical-infrastructure data; controls not disclosed."
      },
      "rights": {
        "derived_model": "planned_open_source_models_commercial_use_terms_not_disclosed",
        "exclusivity": "not_disclosed",
        "inference_retrieval": "yes_field_demonstrations",
        "ownership_transfer": "no_disclosed_transfer",
        "post_training": "unknown",
        "retention": "not_disclosed",
        "sublicensing": "not_disclosed",
        "training": "yes_explicit"
      },
      "scores": {
        "asset": {
          "coverage_pct": 88,
          "factor_evidence": {
            "decision_outcome_richness": {
              "basis": "analysis",
              "rationale": "Analysis: operational context suggests useful decision relationships; missing outcome-label evidence limits the rating.",
              "source_ids": [
                "E1"
              ]
            },
            "decision_value_density": {
              "basis": "analysis",
              "rationale": "Analysis: mistakes in this domain can have major economic consequences. This is domain judgment, not a disclosed corpus valuation.",
              "source_ids": [
                "E1"
              ]
            },
            "domain_value": {
              "basis": "analysis",
              "rationale": "Analysis: the disclosed domain supports economically consequential enterprise activities; this is not a model-uplift result.",
              "source_ids": [
                "E1"
              ]
            },
            "model_learning_usefulness": {
              "basis": "analysis",
              "rationale": "Analysis: the announcement directly links this corpus to model development; improvement magnitude is unmeasured.",
              "source_ids": [
                "E1"
              ]
            },
            "non_replicability": {
              "basis": "analysis",
              "rationale": "Analysis: reproducing owner-specific operating history would require comparable activity and time.",
              "source_ids": [
                "E1"
              ]
            },
            "proprietary_advantage": {
              "basis": "analysis",
              "rationale": "Analysis: member-owned operational or scientific data offers context beyond ordinary public web corpora.",
              "source_ids": [
                "E1"
              ]
            },
            "real_world_grounding": {
              "basis": "analysis",
              "rationale": "Analysis: the described corporate operational or scientific records are grounded in real activity rather than synthetic-only tasks.",
              "source_ids": [
                "E1"
              ]
            },
            "refreshability": {
              "basis": "analysis",
              "rationale": "Analysis: planned field trials and model refinement indicate recurring operational input; no perpetual refresh entitlement is claimed.",
              "source_ids": [
                "E1"
              ]
            },
            "uniqueness_scarcity": {
              "basis": "analysis",
              "rationale": "Analysis: access to this owner-contributed domain corpus is difficult to reproduce; exclusivity is not established.",
              "source_ids": [
                "E1"
              ]
            }
          },
          "factor_ratings": {
            "decision_outcome_richness": 3,
            "decision_value_density": 4,
            "domain_value": 4.5,
            "historical_depth": null,
            "model_learning_usefulness": 4.5,
            "non_replicability": 4,
            "proprietary_advantage": 4.5,
            "real_world_grounding": 4.5,
            "refreshability": 4,
            "rights_usability": null,
            "uniqueness_scarcity": 4
          },
          "rationale": "Qualitative Strategic Signal assessment based on the disclosed corpus and arrangement. Unknown factors remain null; no measured model uplift is claimed.",
          "score": 81
        },
        "methodology_version": "1.0.0",
        "transaction_signal": {
          "coverage_pct": 85,
          "factor_evidence": {
            "clean_price_discovery": {
              "basis": "analysis",
              "rationale": "Analysis: no separately priced corporate dataset is identified; the announcement offers little clean data-price discovery.",
              "source_ids": [
                "E1"
              ]
            },
            "data_consideration_separability": {
              "basis": "analysis",
              "rationale": "No separately quantified consideration for data is disclosed. Zero measures disclosure weakness, not zero data value.",
              "source_ids": [
                "E1"
              ]
            },
            "disclosed_economics": {
              "basis": "analysis",
              "rationale": "No dataset-specific cash amount is disclosed. This rating measures public price disclosure, not whether payment exists.",
              "source_ids": [
                "E1"
              ]
            },
            "explicit_model_use": {
              "basis": "analysis",
              "rationale": "The company explicitly connects proprietary corporate data to developing the named AI models.",
              "source_ids": [
                "E1"
              ]
            },
            "precedent_value": {
              "basis": "analysis",
              "rationale": "Analysis: this is a reusable model-co-development structure, but repeatable standalone pricing is not established.",
              "source_ids": [
                "E1"
              ]
            },
            "rights_clarity": {
              "basis": "analysis",
              "rationale": "Analysis: the announced model-development purpose is clear, while detailed transfer, retention and licensing terms remain unknown.",
              "source_ids": [
                "E1"
              ]
            },
            "strategic_buyer_quality": {
              "basis": "analysis",
              "rationale": "Analysis: an identified model developer with a specific domain program makes the observation strategically relevant.",
              "source_ids": [
                "E1"
              ]
            }
          },
          "factor_ratings": {
            "clean_price_discovery": 0,
            "competing_bids": null,
            "data_consideration_separability": 0,
            "disclosed_economics": 0,
            "explicit_model_use": 5,
            "independent_model_uplift": null,
            "precedent_value": 4,
            "rights_clarity": 2.5,
            "strategic_buyer_quality": 4
          },
          "rationale": "Qualitative Strategic Signal assessment based on the disclosed corpus and arrangement. Unknown factors remain null; no measured model uplift is claimed.",
          "score": 35
        }
      },
      "seller_data_owner": [
        "NAVER Cloud",
        "LG CNS",
        "KEPCO KDN",
        "Korea Hydro & Nuclear Power",
        "Financial Security Institute",
        "KISTI",
        "LG Uplus",
        "Other consortium members"
      ],
      "slug": "naver-cloud-cybersecurity-consortium",
      "source_code_software_asset": "unknown",
      "sources": [
        {
          "publisher": "NAVER",
          "title": "NAVER Cloud Consortium to Develop AI Models Tailored for Cybersecurity",
          "type": "primary",
          "url": "https://navercorp.com/en/media/pressReleasesDetail?seq=10034653"
        }
      ],
      "transaction_id": "SS-AIDT-2026-0004",
      "transaction_status": "announced_development"
    },
    {
      "access_structure": "Customized on-premise model development; Samsung says sensitive technology and operational data remain entirely within its boundaries. Samsung also participated in Mistral's financing round.",
      "analysis_boundary": "On-premise controls, target use cases, and strategic equity relationship are disclosed. Model ownership, exclusivity, sublicensing, and data-specific consideration are not.",
      "announcement_date": "2026-09-09",
      "asset_disposition": "Embedded / Continuing",
      "buyer_ai_developer": [
        "Mistral AI"
      ],
      "canonical_path": "/ai-data-transactions/deals/ss-aidt-2026-0005/",
      "confidence": "high",
      "corpus_description": "Sensitive semiconductor operational and manufacturing data used inside Samsung boundaries to customize on-premise models for defect detection, equipment optimization, and engineering workflows.",
      "corpus_renewal": "Renewable Corpus",
      "data_category": [
        "Manufacturing history",
        "Equipment and process data",
        "Engineering knowledge"
      ],
      "date_added": "2026-09-15",
      "date_last_reviewed": "2026-09-15",
      "decision_outcome_richness": "Exceptional: process interventions, yield, defect, and equipment outcomes can encode extremely high-value manufacturing decisions.",
      "disclosed_consideration": {
        "amount": null,
        "currency": null,
        "description": "Partnership economics and Samsung's individual equity investment were not disclosed separately.",
        "status": "not_disclosed"
      },
      "economics_scope": "bundled_with_strategic_equity_relationship",
      "estimated_economics": {
        "amount": null,
        "currency": null,
        "method": null,
        "status": "not_estimated"
      },
      "historical_depth": "not_disclosed",
      "industry": "Semiconductors",
      "likely_model_use": "Customize models for semiconductor defect detection, equipment optimization, technical support, and engineering decision workflows.",
      "likely_strategic_value": "Pairs a frontier-model developer with one of the world's most valuable proprietary manufacturing feedback loops while preserving on-premise control.",
      "m_and_a_implications": "Equity-linked, on-premise arrangements may become a preferred structure where the corpus is too strategic to sell or externalize.",
      "primary_transaction_structure": "Proprietary Data + Equity Partnership",
      "publication_review": {
        "audited_on": "2026-09-16",
        "human_approval_recorded": false,
        "notice": "Samsung announces customized on-premise models and a strategic equity stake. Processing data within Samsung boundaries does not establish what learning rights Mistral receives.",
        "status": "legacy_review_required"
      },
      "refreshability": "continuous_manufacturing_operations",
      "restrictions": {
        "de_identification": "not_disclosed",
        "employee_customer_data": "not_disclosed",
        "geographic_sovereignty": "On-premise / sovereign control",
        "governed_access": "On-premise deployment; sensitive data stays inside Samsung boundaries.",
        "privacy_pii": "not_disclosed",
        "trade_secret": "Explicit protection of sensitive semiconductor technology and operational data."
      },
      "rights": {
        "derived_model": "unknown",
        "exclusivity": "not_disclosed",
        "inference_retrieval": "yes_on_prem_use",
        "ownership_transfer": "no_disclosed_transfer",
        "post_training": "yes_customization",
        "retention": "restricted_to_samsung_boundaries",
        "sublicensing": "not_disclosed",
        "training": "yes_customized_on_prem_models"
      },
      "scores": {
        "asset": {
          "coverage_pct": 0,
          "factor_ratings": {
            "decision_outcome_richness": null,
            "decision_value_density": null,
            "domain_value": null,
            "historical_depth": null,
            "model_learning_usefulness": null,
            "non_replicability": null,
            "proprietary_advantage": null,
            "real_world_grounding": null,
            "refreshability": null,
            "rights_usability": null,
            "uniqueness_scarcity": null
          },
          "rationale": "Legacy score withdrawn pending evidence-linked scoring and human approval. The superseded v1.0.0 release retains the original values.",
          "score": null
        },
        "methodology_version": "1.0.0",
        "transaction_signal": {
          "coverage_pct": 0,
          "factor_ratings": {
            "clean_price_discovery": null,
            "competing_bids": null,
            "data_consideration_separability": null,
            "disclosed_economics": null,
            "explicit_model_use": null,
            "independent_model_uplift": null,
            "precedent_value": null,
            "rights_clarity": null,
            "strategic_buyer_quality": null
          },
          "rationale": "Legacy score withdrawn pending evidence-linked scoring and human approval. The superseded v1.0.0 release retains the original values.",
          "score": null
        }
      },
      "seller_data_owner": [
        "Samsung Electronics"
      ],
      "slug": "samsung-mistral-semiconductor-partnership",
      "source_code_software_asset": "unknown",
      "sources": [
        {
          "publisher": "Samsung",
          "title": "Samsung and Mistral AI Announce Strategic Partnership for Intelligence-Driven Semiconductor Infrastructure",
          "type": "primary",
          "url": "https://news.samsung.com/global/samsung-and-mistral-ai-announce-strategic-partnership-for-intelligence-driven-semiconductor-infrastructure"
        },
        {
          "publisher": "Reuters",
          "title": "French AI company Mistral hits $24 billion valuation in funding round",
          "type": "secondary",
          "url": "https://www.reuters.com/world/europe/french-ai-company-mistral-hits-24-billion-valuation-funding-round-2026-09-08/"
        }
      ],
      "transaction_id": "SS-AIDT-2026-0005",
      "transaction_status": "announced_active"
    },
    {
      "access_structure": "Built-in licensed/indexed data available to the financial-services product for grounded analysis and citations; specific provider contracts are not public.",
      "analysis_boundary": "Built-in providers and retrieval/citation use are disclosed. Training, post-training, ownership, and pricing are not claimed.",
      "announcement_date": "2026-09-10",
      "asset_disposition": "Embedded / Continuing",
      "buyer_ai_developer": [
        "OpenAI"
      ],
      "canonical_path": "/ai-data-transactions/deals/ss-aidt-2026-0006/",
      "confidence": "medium_high",
      "corpus_description": "Continuously updated proprietary financial datasets integrated as built-in, cited data sources in ChatGPT for Financial Services. Customer-connected subscription sources are excluded from this record.",
      "corpus_renewal": "Renewable Corpus",
      "data_category": [
        "Financial datasets",
        "Private-market data",
        "Fundamental company data",
        "Earnings and filings"
      ],
      "date_added": "2026-09-15",
      "date_last_reviewed": "2026-09-15",
      "decision_outcome_richness": "High for financial research and valuation, though less direct than operational intervention corpora.",
      "disclosed_consideration": {
        "amount": null,
        "currency": null,
        "description": "No provider-specific consideration disclosed.",
        "status": "not_disclosed"
      },
      "economics_scope": "not_disclosed",
      "estimated_economics": {
        "amount": null,
        "currency": null,
        "method": null,
        "status": "not_estimated"
      },
      "historical_depth": "Varies by provider; Daloopa reports 14 years of normalized data.",
      "industry": "Financial information services",
      "likely_model_use": "Ground inference, financial analysis, retrieval, and citations; no training rights are claimed.",
      "likely_strategic_value": "Authoritative licensed data reduces hallucination and makes a general model useful in high-value, source-sensitive financial workflows.",
      "m_and_a_implications": "Shows that renewable data vendors can license access repeatedly without transferring ownership, but undisclosed contracts provide weak standalone price discovery.",
      "primary_transaction_structure": "Proprietary Corpus License",
      "publication_review": {
        "audited_on": "2026-09-16",
        "human_approval_recorded": false,
        "notice": "The cited pages could not be retrieved for this audit. Source-specific rights and the boundary between built-in access and customer connectors remain unresolved.",
        "status": "legacy_review_required"
      },
      "refreshability": "continuous_provider_updates",
      "restrictions": {
        "de_identification": "not_applicable_or_not_disclosed",
        "employee_customer_data": "Customer-connected datasets are explicitly excluded from this record.",
        "geographic_sovereignty": "not_disclosed",
        "governed_access": "Product and provider controls apply; contract details are not public.",
        "privacy_pii": "not_disclosed",
        "trade_secret": "Provider-license restrictions not disclosed."
      },
      "rights": {
        "derived_model": "not_disclosed",
        "exclusivity": "not_disclosed",
        "inference_retrieval": "yes_explicit",
        "ownership_transfer": "no_disclosed_transfer",
        "post_training": "not_disclosed",
        "retention": "indexed_on_openai_infrastructure_for_some_sources",
        "sublicensing": "not_disclosed",
        "training": "not_disclosed"
      },
      "scores": {
        "asset": {
          "coverage_pct": 0,
          "factor_ratings": {
            "decision_outcome_richness": null,
            "decision_value_density": null,
            "domain_value": null,
            "historical_depth": null,
            "model_learning_usefulness": null,
            "non_replicability": null,
            "proprietary_advantage": null,
            "real_world_grounding": null,
            "refreshability": null,
            "rights_usability": null,
            "uniqueness_scarcity": null
          },
          "rationale": "Legacy score withdrawn pending evidence-linked scoring and human approval. The superseded v1.0.0 release retains the original values.",
          "score": null
        },
        "methodology_version": "1.0.0",
        "transaction_signal": {
          "coverage_pct": 0,
          "factor_ratings": {
            "clean_price_discovery": null,
            "competing_bids": null,
            "data_consideration_separability": null,
            "disclosed_economics": null,
            "explicit_model_use": null,
            "independent_model_uplift": null,
            "precedent_value": null,
            "rights_clarity": null,
            "strategic_buyer_quality": null
          },
          "rationale": "Legacy score withdrawn pending evidence-linked scoring and human approval. The superseded v1.0.0 release retains the original values.",
          "score": null
        }
      },
      "seller_data_owner": [
        "LSEG",
        "PitchBook",
        "Daloopa",
        "Crunchbase",
        "Quartr",
        "LSEG News"
      ],
      "slug": "openai-built-in-financial-data-partnerships",
      "source_code_software_asset": "no",
      "sources": [
        {
          "publisher": "OpenAI",
          "title": "ChatGPT for Financial Services",
          "type": "primary",
          "url": "https://help.openai.com/en/articles/12608093-chatgpt-for-financial-services"
        },
        {
          "publisher": "Reuters",
          "title": "OpenAI launches ChatGPT for financial services industry",
          "type": "secondary",
          "url": "https://www.reuters.com/business/openai-launches-chatgpt-financial-services-industry-2026-09-10/"
        },
        {
          "publisher": "Daloopa",
          "title": "Daloopa data infrastructure",
          "type": "primary",
          "url": "https://daloopa.com/"
        }
      ],
      "transaction_id": "SS-AIDT-2026-0006",
      "transaction_status": "launched_active"
    },
    {
      "access_structure": "Three-year joint scientific laboratory to develop frontier models using TotalEnergies geoscience data and expertise. Specific licenses, data custody and allocation of resulting model rights are not disclosed.",
      "analysis_boundary": "Program duration, investment lower bound, joint lab and geoscience history are disclosed. Scores are qualitative analysis, not measured uplift or a valuation of the dataset. Ownership, retention and derived-model rights are not disclosed.",
      "announcement_date": "2026-09-15",
      "asset_disposition": "Embedded / Continuing",
      "buyer_ai_developer": [
        "Mistral AI"
      ],
      "canonical_path": "/ai-data-transactions/deals/ss-aidt-2026-0007/",
      "confidence": "high",
      "corpus_description": "Nearly a century of TotalEnergies geoscience data, knowledge, and reservoir-engineering expertise used in a joint laboratory to develop frontier and agentic AI models for exploration and reservoir decisions.",
      "corpus_renewal": "Renewable Corpus",
      "data_category": [
        "Geological / subsurface data",
        "Reservoir engineering",
        "Decision histories",
        "Scientific data"
      ],
      "date_added": "2026-09-15",
      "date_last_reviewed": "2026-09-16",
      "decision_outcome_richness": "Strategic Signal analysis: reservoir decisions are economically consequential, but decision-linked outcome labeling in the corpus has not been demonstrated publicly.",
      "disclosed_consideration": {
        "amount": 100000000,
        "amount_qualifier": "greater_than",
        "currency": "EUR",
        "description": "More than €100 million over three years; explicitly a program-wide investment, not a dataset price.",
        "status": "disclosed_program_commitment"
      },
      "economics_scope": "bundled_research_compute_people_and_implementation",
      "estimated_economics": {
        "amount": null,
        "currency": null,
        "method": null,
        "status": "not_estimated"
      },
      "historical_depth": "Nearly one century",
      "industry": "Energy & geoscience",
      "likely_model_use": "Develop frontier and agentic models that generate exploration scenarios, interpret subsurface data, optimize reservoirs, and support expert decisions.",
      "likely_strategic_value": "A rare, longitudinal state–decision–action–outcome corpus in a domain where individual decisions can move billions of euros of value.",
      "m_and_a_implications": "Large program economics confirm strategic importance but do not price the corpus separately; transactions for energy data should distinguish data value from scientists, compute, and implementation.",
      "primary_transaction_structure": "Strategic Model Co-Development",
      "publication_review": {
        "approval_channel": "owner_conversation",
        "decision_id": "conversation-20260916-5768A8330103",
        "packet_sha256": "656620707ca856653b6dd4b75854d841b3c4106e4e89e51e1088de26891e85e2",
        "record_sha256": "f40e71b98171603e1825f3e02d1584b2c8e0822e8f4d59cc9e102b8ea534dd6a",
        "reviewed_at": "2026-09-16T21:17:28.112Z",
        "reviewer": "Mike Ye",
        "status": "human_reviewed"
      },
      "refreshability": "continuing_exploration_and_reservoir_operations",
      "restrictions": {
        "de_identification": "not_disclosed",
        "employee_customer_data": "not_applicable_or_not_disclosed",
        "geographic_sovereignty": "European strategic-technology context; binding geographic terms not disclosed.",
        "governed_access": "Joint laboratory; technical access controls and contractual rights allocation are not disclosed.",
        "privacy_pii": "not_applicable_or_not_disclosed",
        "trade_secret": "Announcement emphasizes protecting intellectual property; specific binding provisions are not disclosed."
      },
      "rights": {
        "derived_model": "unknown_joint_program",
        "exclusivity": "not_disclosed",
        "inference_retrieval": "unknown",
        "ownership_transfer": "not_disclosed",
        "post_training": "unknown",
        "retention": "not_disclosed",
        "sublicensing": "not_disclosed",
        "training": "yes_develop_frontier_models_on_corpus"
      },
      "scores": {
        "asset": {
          "coverage_pct": 88,
          "factor_evidence": {
            "decision_outcome_richness": {
              "basis": "analysis",
              "rationale": "Analysis: operational context suggests useful decision relationships; missing outcome-label evidence limits the rating.",
              "source_ids": [
                "E1"
              ]
            },
            "decision_value_density": {
              "basis": "analysis",
              "rationale": "Analysis: mistakes in this domain can have major economic consequences. This is domain judgment, not a disclosed corpus valuation.",
              "source_ids": [
                "E1"
              ]
            },
            "domain_value": {
              "basis": "analysis",
              "rationale": "Analysis: the disclosed domain supports economically consequential enterprise activities; this is not a model-uplift result.",
              "source_ids": [
                "E1"
              ]
            },
            "historical_depth": {
              "basis": "analysis",
              "rationale": "Analysis: the announced near-century geoscience history suggests considerable temporal depth; uniform record completeness is not demonstrated.",
              "source_ids": [
                "E1"
              ]
            },
            "model_learning_usefulness": {
              "basis": "analysis",
              "rationale": "Analysis: the announcement directly links this corpus to model development; improvement magnitude is unmeasured.",
              "source_ids": [
                "E1"
              ]
            },
            "non_replicability": {
              "basis": "analysis",
              "rationale": "Analysis: reproducing owner-specific operating history would require comparable activity and time.",
              "source_ids": [
                "E1"
              ]
            },
            "proprietary_advantage": {
              "basis": "analysis",
              "rationale": "Analysis: member-owned operational or scientific data offers context beyond ordinary public web corpora.",
              "source_ids": [
                "E1"
              ]
            },
            "real_world_grounding": {
              "basis": "analysis",
              "rationale": "Analysis: the described corporate operational or scientific records are grounded in real activity rather than synthetic-only tasks.",
              "source_ids": [
                "E1"
              ]
            },
            "uniqueness_scarcity": {
              "basis": "analysis",
              "rationale": "Analysis: access to this owner-contributed domain corpus is difficult to reproduce; exclusivity is not established.",
              "source_ids": [
                "E1"
              ]
            }
          },
          "factor_ratings": {
            "decision_outcome_richness": 3.5,
            "decision_value_density": 4.5,
            "domain_value": 5,
            "historical_depth": 4.5,
            "model_learning_usefulness": 4,
            "non_replicability": 4,
            "proprietary_advantage": 4,
            "real_world_grounding": 4.5,
            "refreshability": null,
            "rights_usability": null,
            "uniqueness_scarcity": 4.5
          },
          "rationale": "Qualitative Strategic Signal assessment based on the disclosed corpus and arrangement. Unknown factors remain null; no measured model uplift is claimed.",
          "score": 85
        },
        "methodology_version": "1.0.0",
        "transaction_signal": {
          "coverage_pct": 85,
          "factor_evidence": {
            "clean_price_discovery": {
              "basis": "analysis",
              "rationale": "Analysis: no separately priced corporate dataset is identified; the announcement offers little clean data-price discovery.",
              "source_ids": [
                "E1"
              ]
            },
            "data_consideration_separability": {
              "basis": "analysis",
              "rationale": "No separately quantified consideration for data is disclosed. Zero measures disclosure weakness, not zero data value.",
              "source_ids": [
                "E1"
              ]
            },
            "disclosed_economics": {
              "basis": "analysis",
              "rationale": "A program investment lower bound is disclosed, but its allocation to data is not disclosed.",
              "source_ids": [
                "E1"
              ]
            },
            "explicit_model_use": {
              "basis": "analysis",
              "rationale": "The company explicitly connects proprietary corporate data to developing the named AI models.",
              "source_ids": [
                "E1"
              ]
            },
            "precedent_value": {
              "basis": "analysis",
              "rationale": "Analysis: this is a reusable model-co-development structure, but repeatable standalone pricing is not established.",
              "source_ids": [
                "E1"
              ]
            },
            "rights_clarity": {
              "basis": "analysis",
              "rationale": "Analysis: the announced model-development purpose is clear, while detailed transfer, retention and licensing terms remain unknown.",
              "source_ids": [
                "E1"
              ]
            },
            "strategic_buyer_quality": {
              "basis": "analysis",
              "rationale": "Analysis: an identified model developer with a specific domain program makes the observation strategically relevant.",
              "source_ids": [
                "E1"
              ]
            }
          },
          "factor_ratings": {
            "clean_price_discovery": 1,
            "competing_bids": null,
            "data_consideration_separability": 0,
            "disclosed_economics": 4.5,
            "explicit_model_use": 5,
            "independent_model_uplift": null,
            "precedent_value": 4,
            "rights_clarity": 2.5,
            "strategic_buyer_quality": 4
          },
          "rationale": "Qualitative Strategic Signal assessment based on the disclosed corpus and arrangement. Unknown factors remain null; no measured model uplift is claimed.",
          "score": 56
        }
      },
      "seller_data_owner": [
        "TotalEnergies"
      ],
      "slug": "totalenergies-mistral-reservoir-models",
      "source_code_software_asset": "unknown",
      "sources": [
        {
          "publisher": "TotalEnergies",
          "title": "TotalEnergies Announces a Partnership with Mistral to Develop Frontier AI Models for Reservoir Exploration and Engineering",
          "type": "primary",
          "url": "https://totalenergies.com/newsroom/totalenergies-annonce-un-partenariat-avec-mistral-en-vue-de-developper-des-modeles-de-frontiere-dintelligence-artificielle-dedies-a-lexploration-et-a-lingenierie-des-reservoirs-498514"
        }
      ],
      "transaction_id": "SS-AIDT-2026-0007",
      "transaction_status": "announced_three_year_program"
    },
    {
      "access_structure": "Purpose-limited data license to Pathos within a three-party arrangement involving AstraZeneca.",
      "agreement_effective_date": "2025-04-17",
      "analysis_boundary": "This observation covers the data-license leg only. Payment completion, corpus renewal and many contractual restrictions remain unknown. Related fees and financing are context, not additional data transactions.",
      "announcement_date": "2025-04-23",
      "announcement_date_basis": "SEC filing date",
      "asset_disposition": "Embedded / Continuing",
      "buyer_ai_developer": [
        "Pathos AI"
      ],
      "canonical_path": "/ai-data-transactions/deals/ss-aidt-2025-0002/",
      "confidence": "high",
      "confidence_boundary": "High on filed historical terms; current payment performance is unknown.",
      "corpus_description": "A de-identified multimodal oncology dataset licensed for a foundation-model development program.",
      "corpus_renewal": "unknown",
      "data_category": [
        "Scientific / pharmaceutical",
        "De-identified multimodal oncology data"
      ],
      "date_added": "2026-09-16",
      "date_last_reviewed": "2026-09-16",
      "decision_outcome_richness": "unknown; the filing does not define longitudinal outcome fields",
      "disclosed_consideration": {
        "amount": 200000000,
        "currency": "USD",
        "description": "$200m data-license fees over three years, including $50m upfront payable. Up to 50% may be settled in Pathos preferred shares. Separate $35m flows run from AstraZeneca to Tempus and Tempus to Pathos.",
        "status": "disclosed_contractual_fees_not_verified_cash_receipts"
      },
      "economics_scope": "contractually_identified_data_license_fees_within_broader_arrangement",
      "estimated_economics": {
        "amount": null,
        "currency": null,
        "method": null,
        "status": "not_estimated"
      },
      "historical_depth": "not_disclosed",
      "industry": "Pharmaceuticals & biotechnology",
      "likely_model_use": "Develop and train an oncology foundation model.",
      "likely_strategic_value": "Analysis: a specified training license demonstrates that a proprietary corpus can carry contractual consideration.",
      "m_and_a_implications": "Analysis: useful license-fee precedent, not a clean cash sale valuation. Comparability depends on rights, payment form, scope and linked obligations; no annual run-rate or per-patient price is inferred.",
      "primary_transaction_structure": "Training Rights License",
      "refreshability": "not_disclosed",
      "restrictions": {
        "de_identification": "explicit",
        "employee_customer_data": "Detailed consent and patient-data terms not_disclosed",
        "geographic_sovereignty": "not_disclosed",
        "governed_access": "License purpose is development and training of the specified model.",
        "privacy_pii": "de_identified_dataset",
        "trade_secret": "not_disclosed"
      },
      "rights": {
        "derived_model": "Tempus receives model-use license with field restrictions",
        "exclusivity": "not_disclosed",
        "inference_retrieval": "not_disclosed",
        "ownership_transfer": "no_disclosed_transfer",
        "post_training": "not_disclosed",
        "retention": "not_disclosed",
        "sublicensing": "Tempus may sublicense model to AstraZeneca; dataset sublicensing not_disclosed",
        "training": "yes_explicit_purpose_limited"
      },
      "scores": {
        "asset": {
          "coverage_pct": 71,
          "factor_evidence": {
            "decision_value_density": {
              "basis": "analysis",
              "rationale": "Oncology research choices can be consequential; the rating does not assert actual patient-outcome improvement.",
              "source_ids": [
                "E1"
              ]
            },
            "domain_value": {
              "basis": "analysis",
              "rationale": "The use case lies in economically significant scientific research.",
              "source_ids": [
                "E1"
              ]
            },
            "model_learning_usefulness": {
              "basis": "analysis",
              "rationale": "The agreement explicitly links the dataset to the model training task.",
              "source_ids": [
                "E1"
              ]
            },
            "non_replicability": {
              "basis": "analysis",
              "rationale": "Reassembly may require specialized data collection; no reproduction-cost estimate is available.",
              "source_ids": [
                "E1"
              ]
            },
            "proprietary_advantage": {
              "basis": "analysis",
              "rationale": "The purpose-specific private license offers access beyond generic public material.",
              "source_ids": [
                "E1"
              ]
            },
            "real_world_grounding": {
              "basis": "analysis",
              "rationale": "A de-identified domain dataset provides real-world grounding; field detail is limited.",
              "source_ids": [
                "E1"
              ]
            },
            "rights_usability": {
              "basis": "analysis",
              "rationale": "Specified training permission is useful, but purpose limits and undisclosed retention constrain reuse.",
              "source_ids": [
                "E1"
              ]
            },
            "uniqueness_scarcity": {
              "basis": "analysis",
              "rationale": "The licensed private corpus is differentiated, but exact scarcity cannot be quantified.",
              "source_ids": [
                "E1"
              ]
            }
          },
          "factor_ratings": {
            "decision_outcome_richness": null,
            "decision_value_density": 4,
            "domain_value": 4.5,
            "historical_depth": null,
            "model_learning_usefulness": 4.5,
            "non_replicability": 3.5,
            "proprietary_advantage": 4,
            "real_world_grounding": 4,
            "refreshability": null,
            "rights_usability": 3.5,
            "uniqueness_scarcity": 4
          },
          "rationale": "Potentially valuable domain training corpus; uncertainty about contents, age and renewal materially limits coverage. Score is strategic analysis, not measured utility.",
          "score": 82
        },
        "methodology_version": "1.0.0",
        "transaction_signal": {
          "coverage_pct": 85,
          "factor_evidence": {
            "clean_price_discovery": {
              "basis": "analysis",
              "rationale": "Linked financing and reciprocal obligations weaken the isolated market-price interpretation.",
              "source_ids": [
                "E1"
              ]
            },
            "data_consideration_separability": {
              "basis": "analysis",
              "rationale": "The filing identifies a data-license payment leg even though it sits within a broader arrangement.",
              "source_ids": [
                "E1"
              ]
            },
            "disclosed_economics": {
              "basis": "analysis",
              "rationale": "Contractual fees and payment form are disclosed, but realized proceeds are not verified.",
              "source_ids": [
                "E1"
              ]
            },
            "explicit_model_use": {
              "basis": "analysis",
              "rationale": "Model development and training are the express permitted purposes.",
              "source_ids": [
                "E1"
              ]
            },
            "precedent_value": {
              "basis": "analysis",
              "rationale": "The contract structure offers a useful reference if future comparisons preserve its restrictions.",
              "source_ids": [
                "E1"
              ]
            },
            "rights_clarity": {
              "basis": "analysis",
              "rationale": "Purpose and some model rights are visible; exclusivity, retention and data sublicensing are not.",
              "source_ids": [
                "E1"
              ]
            },
            "strategic_buyer_quality": {
              "basis": "analysis",
              "rationale": "A specified specialist model developer provides credible use, without implying open-market bidding.",
              "source_ids": [
                "E1"
              ]
            }
          },
          "factor_ratings": {
            "clean_price_discovery": 2,
            "competing_bids": null,
            "data_consideration_separability": 4,
            "disclosed_economics": 4.5,
            "explicit_model_use": 5,
            "independent_model_uplift": null,
            "precedent_value": 4,
            "rights_clarity": 3.5,
            "strategic_buyer_quality": 3
          },
          "rationale": "Explicit fee allocation is valuable pricing evidence. Equity settlement, reciprocal commitments and missing comparable bids prevent treatment as a clean cash market price.",
          "score": 75
        }
      },
      "seller_data_owner": [
        "Tempus AI"
      ],
      "slug": "tempus-pathos-oncology-data-license",
      "source_code_software_asset": "no",
      "sources": [
        {
          "publisher": "Tempus AI / SEC",
          "title": "Form 8-K filed April 23, 2025, Item 8.01",
          "type": "primary",
          "url": "https://www.sec.gov/Archives/edgar/data/1717115/000119312525090062/d944168d8k.htm"
        }
      ],
      "transaction_id": "SS-AIDT-2025-0002",
      "transaction_status": "agreement_disclosed; current payment performance not verified",
      "publication_review": {
        "status": "human_reviewed",
        "approval_channel": "owner_conversation",
        "reviewer": "Mike Ye",
        "reviewed_at": "2026-09-16T21:38:20.346Z",
        "decision_id": "conversation-20260916-CD20DA31F0CF",
        "packet_sha256": "401f767bb940538d1a214a8130fd811dfa4275b939fcba6f15377ce87c2ac9b2",
        "record_sha256": "12bdd18b6c5a16d4e4612db7460712415283174fb14daa888dd8dd3ebb6a945b"
      }
    },
    {
      "transaction_id": "SS-AIDT-2026-0008",
      "slug": "bolt-arcee-forge-training-data-license",
      "canonical_path": "/ai-data-transactions/deals/ss-aidt-2026-0008/",
      "announcement_date": "2026-09-14",
      "buyer_ai_developer": [
        "Arcee AI"
      ],
      "seller_data_owner": [
        "Bolt / StackBlitz"
      ],
      "transaction_status": "announced_preview_training_window",
      "industry": "Software development & AI training data",
      "data_category": [
        "Source code and project files",
        "Developer prompts",
        "Tool-call traces",
        "Edit histories",
        "Error and fix traces"
      ],
      "corpus_description": "Opted-in Bolt Forge sessions containing developer prompts, code, project files and configuration, tool calls, edit histories and fix traces from individual Pro users during the September 14–October 14, 2026 preview window.",
      "primary_transaction_structure": "Training Rights License",
      "access_structure": "Bolt de-identifies opted-in Forge sessions inside its infrastructure and licenses resulting datasets to AI developers under a data license agreement. Arcee AI is the first named recipient; Bolt says the datasets may also be licensed to other AI developers.",
      "asset_disposition": "Embedded / Continuing",
      "corpus_renewal": "Renewable Corpus",
      "source_code_software_asset": "yes",
      "disclosed_consideration": {
        "status": "disclosed_non_cash_user_consideration",
        "amount": null,
        "currency": null,
        "description": "Individual Pro users receive up to 50× additional Forge usage during the preview; Bolt describes the allocation as payment for sharing. Bolt may also receive cash for licensed datasets, but no amount is disclosed."
      },
      "estimated_economics": {
        "status": "not_estimated",
        "amount": null,
        "currency": null,
        "method": null
      },
      "economics_scope": "In-kind usage allocation to contributing users; any Arcee-to-Bolt license payment is undisclosed.",
      "rights": {
        "training": "Explicit: the licensed sessions feed Arcee AI's first training run for a trillion-parameter-class model.",
        "post_training": "not_disclosed",
        "inference_retrieval": "not_disclosed",
        "ownership_transfer": "No ownership transfer is disclosed; the announced structure is a data license agreement.",
        "exclusivity": "Non-exclusive at Bolt level: Bolt says datasets may be licensed to other AI developers. Arcee-specific exclusivity terms are not disclosed.",
        "retention": "Users can request that Bolt stop using already-collected Forge content; deletion mechanics and treatment of trained-model effects are not disclosed.",
        "derived_model": "Arcee will train a trillion-parameter-class model; ownership and allocation of derived weights are not disclosed.",
        "sublicensing": "Bolt may license the datasets to additional AI developers; Arcee's sublicensing rights are not disclosed."
      },
      "restrictions": {
        "governed_access": "One-tap opt-in is required before collection; switching away from Forge stops new collection.",
        "geographic_sovereignty": "Sessions from the EEA, United Kingdom and Switzerland are excluded from training and dataset licensing.",
        "privacy_pii": "Bolt says personal information is stripped and the de-identification pipeline is tested with seeded data.",
        "trade_secret": "Bolt says secrets and sensitive data are stripped before material leaves its infrastructure; the effectiveness and contractual remedies are not independently verified.",
        "de_identification": "Explicitly required and performed by Bolt before transfer.",
        "employee_customer_data": "Team and Enterprise workspaces are excluded; the disclosed corpus comes from opted-in individual Pro users."
      },
      "refreshability": "Renewable while users opt into Forge; the first disclosed Arcee training window is September 14–October 14, 2026.",
      "historical_depth": "Newly collected preview corpus beginning September 14, 2026; no long operating history is disclosed.",
      "decision_outcome_richness": "The corpus records software-building sequences including attempts, edits, tool calls, errors and fixes, creating observable short-cycle action and outcome traces.",
      "likely_model_use": "Training open-weight coding and software-building models to complete real application-development workflows in fewer attempts.",
      "likely_strategic_value": "Captures real developer interaction and repair traces that cannot be reconstructed reliably from static public code repositories alone.",
      "m_and_a_implications": "Shows that a software platform can monetize consented workflow telemetry separately from subscriptions and can compensate contributors with compute or usage rather than cash.",
      "analysis_boundary": "The data-license structure, corpus fields, consent controls, initial Arcee recipient and model-training purpose are disclosed by Bolt. Cash economics, raw-data retention, derived-weight ownership, Arcee sublicensing and measured model uplift are not disclosed.",
      "confidence": "high",
      "scores": {
        "methodology_version": "1.0.0",
        "asset": {
          "score": 72,
          "coverage_pct": 100,
          "rationale": "Strategic Signal analysis. The workflow traces are unusually useful for software-agent learning, but the corpus is new, excludes enterprise workspaces and has no independent uplift evidence.",
          "factor_ratings": {
            "uniqueness_scarcity": 4,
            "historical_depth": 0.5,
            "decision_outcome_richness": 4,
            "decision_value_density": 2.5,
            "real_world_grounding": 4,
            "domain_value": 4,
            "refreshability": 4.5,
            "proprietary_advantage": 4,
            "model_learning_usefulness": 4.5,
            "non_replicability": 3.5,
            "rights_usability": 3.5
          },
          "factor_evidence": {
            "uniqueness_scarcity": {
              "basis": "analysis",
              "source_ids": [
                "E1"
              ],
              "rationale": "Static code is common, but consented end-to-end build, error and fix traces are difficult to scrape or reconstruct."
            },
            "historical_depth": {
              "basis": "disclosed_fact",
              "source_ids": [
                "E1"
              ],
              "rationale": "The disclosed corpus begins with a one-month preview window in September 2026, so historical depth is presently limited."
            },
            "decision_outcome_richness": {
              "basis": "analysis",
              "source_ids": [
                "E1"
              ],
              "rationale": "Edit histories, tool calls and fix traces expose repeated action-and-result sequences within software-building workflows."
            },
            "decision_value_density": {
              "basis": "analysis",
              "source_ids": [
                "E1"
              ],
              "rationale": "Individual coding decisions are usually lower-value than clinical or capital decisions, though successful repairs carry useful technical information."
            },
            "real_world_grounding": {
              "basis": "analysis",
              "source_ids": [
                "E1"
              ],
              "rationale": "The sessions arise from real user projects and tool interactions rather than a synthetic-only benchmark corpus."
            },
            "domain_value": {
              "basis": "analysis",
              "source_ids": [
                "E1"
              ],
              "rationale": "Coding-agent capability has broad economic value across software development, even though the disclosed users are not enterprise workspaces."
            },
            "refreshability": {
              "basis": "disclosed_fact",
              "source_ids": [
                "E1"
              ],
              "rationale": "Forge can continue collecting new opted-in sessions as users build, subject to the stated program and geography limits."
            },
            "proprietary_advantage": {
              "basis": "analysis",
              "source_ids": [
                "E1"
              ],
              "rationale": "Bolt controls the platform telemetry and de-identification pipeline that create a differentiated corpus beyond public repositories."
            },
            "model_learning_usefulness": {
              "basis": "analysis",
              "source_ids": [
                "E1"
              ],
              "rationale": "Bolt directly links the sessions to training a new model and explains that repair traces can reduce the number of attempts required."
            },
            "non_replicability": {
              "basis": "analysis",
              "source_ids": [
                "E1"
              ],
              "rationale": "Another builder could collect similar telemetry, but reproducing Bolt's exact user interactions and repair sequences would require comparable scale and consent."
            },
            "rights_usability": {
              "basis": "analysis",
              "source_ids": [
                "E1"
              ],
              "rationale": "Opt-in collection, de-identification and an explicit data license support usability, while retention and downstream model terms remain undisclosed."
            }
          }
        },
        "transaction_signal": {
          "score": 66,
          "coverage_pct": 85,
          "rationale": "Strategic Signal analysis. The arrangement cleanly proves training use and non-cash contributor consideration, but provides weak cash price discovery and no measured uplift.",
          "factor_ratings": {
            "disclosed_economics": 3,
            "clean_price_discovery": 1,
            "data_consideration_separability": 3.5,
            "explicit_model_use": 5,
            "rights_clarity": 4,
            "competing_bids": null,
            "strategic_buyer_quality": 3.5,
            "precedent_value": 4.5,
            "independent_model_uplift": null
          },
          "factor_evidence": {
            "disclosed_economics": {
              "basis": "disclosed_fact",
              "source_ids": [
                "E1"
              ],
              "rationale": "Bolt discloses up to 50× usage as contributor payment and says it may receive payment, but gives no cash value for the dataset license."
            },
            "clean_price_discovery": {
              "basis": "analysis",
              "source_ids": [
                "E1"
              ],
              "rationale": "The usage benefit is bundled with a preview program and no buyer payment amount is disclosed, limiting clean price discovery."
            },
            "data_consideration_separability": {
              "basis": "analysis",
              "source_ids": [
                "E1"
              ],
              "rationale": "Bolt explicitly describes the usage allocation as payment for sharing, but no cash-equivalent valuation is provided."
            },
            "explicit_model_use": {
              "basis": "disclosed_fact",
              "source_ids": [
                "E1"
              ],
              "rationale": "The announcement directly states that shared sessions feed Arcee's first training run for a trillion-parameter-class model."
            },
            "rights_clarity": {
              "basis": "analysis",
              "source_ids": [
                "E1"
              ],
              "rationale": "Training, opt-in, exclusions and Bolt's ability to license other developers are clear; retention and derived-model allocation are not."
            },
            "strategic_buyer_quality": {
              "basis": "analysis",
              "source_ids": [
                "E1"
              ],
              "rationale": "Arcee is a named model developer with a concrete disclosed training run, though it is not a frontier lab of the largest scale."
            },
            "precedent_value": {
              "basis": "analysis",
              "source_ids": [
                "E1"
              ],
              "rationale": "The data-for-usage exchange is a replicable template for software platforms monetizing consented workflow traces."
            }
          }
        }
      },
      "sources": [
        {
          "type": "primary",
          "publisher": "Bolt",
          "title": "What is Bolt Forge?",
          "url": "https://bolt.new/blog/what-is-bolt-forge"
        }
      ],
      "date_added": "2026-09-17",
      "date_last_reviewed": "2026-09-17",
      "publication_review": {
        "status": "human_reviewed",
        "approval_channel": "github_workflow",
        "reviewer": "Trailgenic",
        "reviewed_at": "2026-09-17T17:36:14.674Z",
        "decision_id": "35253728428-1",
        "packet_sha256": "e1891b409964b996e4b464fd5436cf7ef8ae43599ca7f8153171c06be5ce1ff9",
        "record_sha256": "c5a10eeece7f3e720d635e409999286246ebbd48d54636a21ffa50017c641d5e"
      }
    },
    {
      "transaction_id": "SS-AIDT-2026-0009",
      "slug": "silencio-tagalog-cebuano-voice-data-contract",
      "canonical_path": "/ai-data-transactions/deals/ss-aidt-2026-0009/",
      "announcement_date": "2026-09-07",
      "buyer_ai_developer": [
        "Undisclosed leading voice-AI company"
      ],
      "seller_data_owner": [
        "Silencio"
      ],
      "transaction_status": "contract_signed",
      "industry": "Voice AI & training data",
      "data_category": [
        "Human speech recordings",
        "Low-resource language data",
        "Single-speaker voice data",
        "Multi-speaker voice data"
      ],
      "corpus_description": "A contracted corpus of native-speaker Tagalog and Cebuano single- and multi-speaker voice recordings assembled through Silencio's distributed contributor network for a leading voice-AI company.",
      "primary_transaction_structure": "Proprietary Corpus License",
      "access_structure": "Dataset contract and delivery following procurement, legal, security and sample evaluation. The buyer, delivery volume, term, exclusivity and detailed license language are not disclosed.",
      "asset_disposition": "Stranded / Separable",
      "corpus_renewal": "Finite Corpus",
      "source_code_software_asset": "no",
      "disclosed_consideration": {
        "status": "disclosed_contract_value",
        "amount": 250000,
        "currency": "USD",
        "description": "Dataset-specific contract value disclosed as $250,000; the buyer is unnamed."
      },
      "estimated_economics": {
        "status": "not_estimated",
        "amount": null,
        "currency": null,
        "method": null
      },
      "economics_scope": "Specific Tagalog and Cebuano single- and multi-speaker voice-data contract.",
      "rights": {
        "training": "The buyer is a voice-AI company and Silencio states AI labs train on its data; the exact contract grant is not disclosed.",
        "post_training": "not_disclosed",
        "inference_retrieval": "not_disclosed",
        "ownership_transfer": "not_disclosed",
        "exclusivity": "not_disclosed",
        "retention": "not_disclosed",
        "derived_model": "not_disclosed",
        "sublicensing": "not_disclosed"
      },
      "restrictions": {
        "governed_access": "Procurement, legal and security review are disclosed; binding access controls are not.",
        "geographic_sovereignty": "not_disclosed",
        "privacy_pii": "Voice recordings can carry biometric and identity risk; contract-specific privacy terms are not disclosed.",
        "trade_secret": "not_disclosed",
        "de_identification": "not_disclosed",
        "employee_customer_data": "not_applicable; contributors record data for compensation rather than supplying enterprise employee or customer records."
      },
      "refreshability": "The contracted delivery is finite, while Silencio's contributor network can produce follow-on recordings and adjacent-language corpora.",
      "historical_depth": "Silencio began gathering data in November 2025; the age and duration represented in this specific delivery are not disclosed.",
      "decision_outcome_richness": "The corpus provides speech examples rather than a state–decision–action–outcome history; decision richness is limited.",
      "likely_model_use": "Training or improving speech recognition, multilingual voice generation or voice-agent performance in Tagalog and Cebuano.",
      "likely_strategic_value": "Low-resource native-speaker recordings are difficult to scrape and can fill coverage gaps that synthetic or high-resource-language data does not resolve.",
      "m_and_a_implications": "The disclosed $250,000 contract provides a rare observable price point for a defined low-resource-language voice corpus, though volume, hours and exact rights are unknown.",
      "analysis_boundary": "Contract value, languages, speaker formats and buyer type are disclosed by an investor in Silencio. Buyer identity, dataset size, ownership, exclusivity, retention, sublicensing, derived-model rights and independent model uplift are not disclosed.",
      "confidence": "medium_high",
      "scores": {
        "methodology_version": "1.0.0",
        "asset": {
          "score": 63,
          "coverage_pct": 100,
          "rationale": "Strategic Signal analysis. The recordings are scarce, consent-oriented and useful for model coverage, but they contain little decision-outcome structure and have limited disclosed historical depth.",
          "factor_ratings": {
            "uniqueness_scarcity": 4.5,
            "historical_depth": 1,
            "decision_outcome_richness": 0.5,
            "decision_value_density": 1,
            "real_world_grounding": 5,
            "domain_value": 4,
            "refreshability": 4,
            "proprietary_advantage": 4.5,
            "model_learning_usefulness": 4.5,
            "non_replicability": 4,
            "rights_usability": 3.5
          },
          "factor_evidence": {
            "uniqueness_scarcity": {
              "basis": "analysis",
              "source_ids": [
                "E1",
                "E2"
              ],
              "rationale": "Native-speaker Tagalog and Cebuano recordings cannot be assembled reliably from the open web at the same coverage and consent standard."
            },
            "historical_depth": {
              "basis": "disclosed_fact",
              "source_ids": [
                "E1"
              ],
              "rationale": "Silencio says collection began in November 2025, so the disclosed business has limited temporal depth relative to longitudinal operating datasets."
            },
            "decision_outcome_richness": {
              "basis": "analysis",
              "source_ids": [
                "E1"
              ],
              "rationale": "Speech recordings contain linguistic observations but not a disclosed sequence of economically consequential decisions and outcomes."
            },
            "decision_value_density": {
              "basis": "analysis",
              "source_ids": [
                "E1"
              ],
              "rationale": "The corpus improves language coverage but does not encode high-value underwriting, clinical, industrial or capital-allocation decisions."
            },
            "real_world_grounding": {
              "basis": "disclosed_fact",
              "source_ids": [
                "E1",
                "E2"
              ],
              "rationale": "The recordings are contributed by native speakers in relevant markets rather than described as synthetic speech."
            },
            "domain_value": {
              "basis": "analysis",
              "source_ids": [
                "E1",
                "E2"
              ],
              "rationale": "Multilingual speech capability is commercially valuable for voice agents and speech systems, especially in underserved languages."
            },
            "refreshability": {
              "basis": "analysis",
              "source_ids": [
                "E1",
                "E2"
              ],
              "rationale": "Silencio's distributed network can collect follow-on speech and related languages, though this contract is a finite delivery."
            },
            "proprietary_advantage": {
              "basis": "analysis",
              "source_ids": [
                "E1",
                "E2"
              ],
              "rationale": "Silencio controls consented contributor relationships and collection operations that are unavailable through ordinary web scraping."
            },
            "model_learning_usefulness": {
              "basis": "analysis",
              "source_ids": [
                "E1",
                "E2"
              ],
              "rationale": "The corpus directly addresses model coverage in languages with little public training data; the magnitude of improvement is unmeasured."
            },
            "non_replicability": {
              "basis": "analysis",
              "source_ids": [
                "E1",
                "E2"
              ],
              "rationale": "A competitor could recruit speakers, but reproducing geographic coverage, consent and collection at comparable speed requires substantial network infrastructure."
            },
            "rights_usability": {
              "basis": "analysis",
              "source_ids": [
                "E1",
                "E2"
              ],
              "rationale": "The data is marketed as consent-cleared and licensing-ready, while contract-specific retention, exclusivity and downstream rights remain unknown."
            }
          }
        },
        "transaction_signal": {
          "score": 89,
          "coverage_pct": 79,
          "rationale": "Strategic Signal analysis. The contract supplies unusually clean dataset-specific cash economics for a defined language corpus, but detailed rights and the buyer identity remain undisclosed.",
          "factor_ratings": {
            "disclosed_economics": 5,
            "clean_price_discovery": 4.5,
            "data_consideration_separability": 5,
            "explicit_model_use": 4.5,
            "rights_clarity": 3,
            "competing_bids": null,
            "strategic_buyer_quality": null,
            "precedent_value": 4,
            "independent_model_uplift": null
          },
          "factor_evidence": {
            "disclosed_economics": {
              "basis": "disclosed_fact",
              "source_ids": [
                "E1"
              ],
              "rationale": "The issuer-responsible announcement states a $250,000 contract value for the defined Tagalog and Cebuano corpus."
            },
            "clean_price_discovery": {
              "basis": "analysis",
              "source_ids": [
                "E1"
              ],
              "rationale": "The cash amount is tied to a specific corpus, although undisclosed volume and contract term prevent a per-hour benchmark."
            },
            "data_consideration_separability": {
              "basis": "disclosed_fact",
              "source_ids": [
                "E1"
              ],
              "rationale": "The announcement presents the $250,000 amount as a voice-data contract rather than a bundled software deployment."
            },
            "explicit_model_use": {
              "basis": "analysis",
              "source_ids": [
                "E1",
                "E2"
              ],
              "rationale": "The buyer is identified as a leading voice-AI company and Silencio says AI labs train on its network data; exact training language is not quoted from the contract."
            },
            "rights_clarity": {
              "basis": "analysis",
              "source_ids": [
                "E1",
                "E2"
              ],
              "rationale": "A data contract and consent-oriented supply are clear, but ownership, exclusivity, retention and sublicensing are undisclosed."
            },
            "precedent_value": {
              "basis": "analysis",
              "source_ids": [
                "E1"
              ],
              "rationale": "A dataset-specific cash price for named low-resource languages can inform future speech-data transactions, subject to missing volume normalization."
            }
          }
        }
      },
      "sources": [
        {
          "type": "primary_syndicated",
          "publisher": "Advanced Blockchain AG via EQS News",
          "title": "Portfolio company Silencio signs record voice-data contract as AI labs turn to low-resource languages",
          "url": "https://www.eqs-news.com/news/corporate/advanced-blockchain-ag-portfolio-company-silencio-signs-record-voice-data-contract-as-ai-labs-turn-to-low-resource-languages/87e56ca5-9883-4f82-8b2e-d265764a5dc7_en"
        },
        {
          "type": "primary",
          "publisher": "Silencio",
          "title": "Silencio AI",
          "url": "https://www.silencio.network/"
        }
      ],
      "date_added": "2026-09-17",
      "date_last_reviewed": "2026-09-17",
      "publication_review": {
        "status": "human_reviewed",
        "approval_channel": "github_workflow",
        "reviewer": "Trailgenic",
        "reviewed_at": "2026-09-17T17:37:52.520Z",
        "decision_id": "35253885258-1",
        "packet_sha256": "b084f0f7a0ae807c847525e517d07eaafda15abc1df076c4c07397f0a2558baf",
        "record_sha256": "165b8c0418f97c60c4332558ccb0a13968fb6c5ecb635a68041e380d662f75a0"
      }
    },
    {
      "transaction_id": "SS-AIDT-2026-0010",
      "slug": "silencio-african-language-voice-data-license",
      "canonical_path": "/ai-data-transactions/deals/ss-aidt-2026-0010/",
      "announcement_date": "2026-09-07",
      "buyer_ai_developer": [
        "Undisclosed leading AI developer"
      ],
      "seller_data_owner": [
        "Silencio"
      ],
      "transaction_status": "license_signed",
      "industry": "Voice AI & training data",
      "data_category": [
        "Human speech recordings",
        "Low-resource language data",
        "African language datasets",
        "Transcription-ready voice data"
      ],
      "corpus_description": "Native-speaker voice datasets covering Swahili, Amharic, Hausa, Wolof and Yoruba, assembled through Silencio's distributed contributor network under a separate license to an unnamed leading AI developer.",
      "primary_transaction_structure": "Proprietary Corpus License",
      "access_structure": "A separate data licensing agreement for five named African-language datasets. Economics, volume, delivery term, exclusivity and detailed license language are not disclosed.",
      "asset_disposition": "Stranded / Separable",
      "corpus_renewal": "Finite Corpus",
      "source_code_software_asset": "no",
      "disclosed_consideration": {
        "status": "not_disclosed",
        "amount": null,
        "currency": null,
        "description": "The announcement discloses the license but not its consideration."
      },
      "estimated_economics": {
        "status": "not_estimated",
        "amount": null,
        "currency": null,
        "method": null
      },
      "economics_scope": "Separate license for five African-language datasets; no transaction economics are disclosed.",
      "rights": {
        "training": "Silencio states that AI labs train on its data and identifies the counterparty as an AI developer; the exact contract grant is not disclosed.",
        "post_training": "not_disclosed",
        "inference_retrieval": "not_disclosed",
        "ownership_transfer": "not_disclosed",
        "exclusivity": "not_disclosed",
        "retention": "not_disclosed",
        "derived_model": "not_disclosed",
        "sublicensing": "not_disclosed"
      },
      "restrictions": {
        "governed_access": "A commercial data license is disclosed; binding access controls are not.",
        "geographic_sovereignty": "not_disclosed",
        "privacy_pii": "Voice recordings can carry biometric and identity risk; contract-specific privacy terms are not disclosed.",
        "trade_secret": "not_disclosed",
        "de_identification": "not_disclosed",
        "employee_customer_data": "not_applicable; contributors record data for compensation rather than supplying enterprise employee or customer records."
      },
      "refreshability": "The initial five-language delivery is finite; the contributor network supports follow-on collection, paid transcription and additional languages.",
      "historical_depth": "Silencio began gathering data in November 2025; the age and duration represented in these specific datasets are not disclosed.",
      "decision_outcome_richness": "The corpus provides speech examples rather than a state–decision–action–outcome history; decision richness is limited.",
      "likely_model_use": "Training or improving speech recognition, multilingual voice generation or voice-agent performance across five underrepresented African languages.",
      "likely_strategic_value": "Native-speaker recordings in languages with little public data can expand model coverage and reduce dependence on synthetic or translated substitutes.",
      "m_and_a_implications": "Confirms that one proprietary collection network can support repeat, multi-language AI-data licenses; undisclosed economics prevent valuation benchmarking for this agreement.",
      "analysis_boundary": "The separate license, five initial languages and buyer type are disclosed by an investor in Silencio. Buyer identity, consideration, dataset size, ownership, exclusivity, retention, sublicensing, derived-model rights and independent uplift are not disclosed.",
      "confidence": "medium_high",
      "scores": {
        "methodology_version": "1.0.0",
        "asset": {
          "score": 63,
          "coverage_pct": 100,
          "rationale": "Strategic Signal analysis. The recordings are scarce, consent-oriented and useful for model coverage, but they contain little decision-outcome structure and have limited disclosed historical depth.",
          "factor_ratings": {
            "uniqueness_scarcity": 4.5,
            "historical_depth": 1,
            "decision_outcome_richness": 0.5,
            "decision_value_density": 1,
            "real_world_grounding": 5,
            "domain_value": 4,
            "refreshability": 4,
            "proprietary_advantage": 4.5,
            "model_learning_usefulness": 4.5,
            "non_replicability": 4,
            "rights_usability": 3.5
          },
          "factor_evidence": {
            "uniqueness_scarcity": {
              "basis": "analysis",
              "source_ids": [
                "E1",
                "E2"
              ],
              "rationale": "Native-speaker recordings across the five named languages are difficult to source at scale from public web material."
            },
            "historical_depth": {
              "basis": "disclosed_fact",
              "source_ids": [
                "E1"
              ],
              "rationale": "Silencio says collection began in November 2025, so the disclosed business has limited temporal depth relative to longitudinal operating datasets."
            },
            "decision_outcome_richness": {
              "basis": "analysis",
              "source_ids": [
                "E1"
              ],
              "rationale": "Speech recordings contain linguistic observations but not a disclosed sequence of economically consequential decisions and outcomes."
            },
            "decision_value_density": {
              "basis": "analysis",
              "source_ids": [
                "E1"
              ],
              "rationale": "The corpus improves language coverage but does not encode high-value underwriting, clinical, industrial or capital-allocation decisions."
            },
            "real_world_grounding": {
              "basis": "disclosed_fact",
              "source_ids": [
                "E1",
                "E2"
              ],
              "rationale": "The recordings are contributed by speakers in relevant markets rather than described as synthetic speech."
            },
            "domain_value": {
              "basis": "analysis",
              "source_ids": [
                "E1",
                "E2"
              ],
              "rationale": "Multilingual speech capability is commercially valuable for voice agents and speech systems, especially in underserved languages."
            },
            "refreshability": {
              "basis": "analysis",
              "source_ids": [
                "E1",
                "E2"
              ],
              "rationale": "The network can collect follow-on speech and additional languages even though the initial five-language delivery is finite."
            },
            "proprietary_advantage": {
              "basis": "analysis",
              "source_ids": [
                "E1",
                "E2"
              ],
              "rationale": "Silencio controls consented contributor relationships and collection operations that are unavailable through ordinary web scraping."
            },
            "model_learning_usefulness": {
              "basis": "analysis",
              "source_ids": [
                "E1",
                "E2"
              ],
              "rationale": "The datasets directly address model coverage in languages with little public training data; improvement magnitude is unmeasured."
            },
            "non_replicability": {
              "basis": "analysis",
              "source_ids": [
                "E1",
                "E2"
              ],
              "rationale": "A competitor could recruit speakers, but reproducing multi-country coverage, consent and collection speed requires substantial network infrastructure."
            },
            "rights_usability": {
              "basis": "analysis",
              "source_ids": [
                "E1",
                "E2"
              ],
              "rationale": "The data is marketed as consent-cleared and licensing-ready, while contract-specific retention, exclusivity and downstream rights remain unknown."
            }
          }
        },
        "transaction_signal": {
          "score": null,
          "coverage_pct": 47,
          "rationale": "Strategic Signal analysis. A separable AI-data license is explicit, but undisclosed economics, buyer identity and most contract rights leave evidence coverage below the methodology's 70% threshold.",
          "factor_ratings": {
            "disclosed_economics": null,
            "clean_price_discovery": null,
            "data_consideration_separability": 4.5,
            "explicit_model_use": 4.5,
            "rights_clarity": 3,
            "competing_bids": null,
            "strategic_buyer_quality": null,
            "precedent_value": 4,
            "independent_model_uplift": null
          },
          "factor_evidence": {
            "data_consideration_separability": {
              "basis": "disclosed_fact",
              "source_ids": [
                "E1"
              ],
              "rationale": "The announcement explicitly describes a separate data licensing agreement for five named language datasets."
            },
            "explicit_model_use": {
              "basis": "analysis",
              "source_ids": [
                "E1",
                "E2"
              ],
              "rationale": "The counterparty is described as a leading AI developer and Silencio says AI labs train on its network data; contract wording is not public."
            },
            "rights_clarity": {
              "basis": "analysis",
              "source_ids": [
                "E1",
                "E2"
              ],
              "rationale": "The existence and scope languages of the license are clear, but ownership, exclusivity, retention and sublicensing are undisclosed."
            },
            "precedent_value": {
              "basis": "analysis",
              "source_ids": [
                "E1"
              ],
              "rationale": "The agreement demonstrates repeat multi-language licensing from a renewable contributor network, even though it offers no price benchmark."
            }
          }
        }
      },
      "sources": [
        {
          "type": "primary_syndicated",
          "publisher": "Advanced Blockchain AG via EQS News",
          "title": "Portfolio company Silencio signs record voice-data contract as AI labs turn to low-resource languages",
          "url": "https://www.eqs-news.com/news/corporate/advanced-blockchain-ag-portfolio-company-silencio-signs-record-voice-data-contract-as-ai-labs-turn-to-low-resource-languages/87e56ca5-9883-4f82-8b2e-d265764a5dc7_en"
        },
        {
          "type": "primary",
          "publisher": "Silencio",
          "title": "Silencio AI",
          "url": "https://www.silencio.network/"
        }
      ],
      "date_added": "2026-09-17",
      "date_last_reviewed": "2026-09-17",
      "publication_review": {
        "status": "human_reviewed",
        "approval_channel": "github_workflow",
        "reviewer": "Trailgenic",
        "reviewed_at": "2026-09-17T17:38:43.226Z",
        "decision_id": "35253967579-1",
        "packet_sha256": "edd54770424ddcc1d311f83df2a2a5fce9d23adc3996636c30b8222479b3b11b",
        "record_sha256": "4b81ae9d51db5e8e069e1517003afc4e36600684cd8c67cc4e403d839067a698"
      }
    },
    {
      "transaction_id": "SS-AIDT-2026-0011",
      "slug": "harell-data-controlled-biotech-model-training",
      "canonical_path": "/ai-data-transactions/deals/ss-aidt-2026-0011/",
      "announcement_date": "2026-09-15",
      "buyer_ai_developer": [
        "Authorized Harell Data model builders and scientific research teams"
      ],
      "seller_data_owner": [
        "A-Alpha Bio",
        "Adaptive Biotechnologies"
      ],
      "transaction_status": "active_multi_year_controlled_learning_platform",
      "industry": "Biotechnology, scientific AI, and AI infrastructure",
      "data_category": [
        "Antibody-antigen affinity and structural data",
        "T-cell-receptor binding data",
        "Hidden scientific validation datasets"
      ],
      "corpus_description": "Proprietary scientific datasets hosted in Harell Data's controlled environment, initially including A-Alpha Bio antibody-antigen affinity and structural data and Adaptive Biotechnologies T-cell-receptor binding data, plus hidden validation questions retained by data partners.",
      "primary_transaction_structure": "Federated / Controlled Learning Rights",
      "access_structure": "Authorized model builders bring models into Harell Data's CoreWeave-powered environment for training, fine-tuning, validation, and inference. Raw datasets remain in place and cannot be viewed or extracted. Builders receive trained models and model IP; data owners receive revenue share from each metered training run, while downstream models may be listed as Models as a Service on Harell's marketplace.",
      "asset_disposition": "Embedded / Continuing",
      "corpus_renewal": "Renewable Corpus",
      "source_code_software_asset": "no",
      "disclosed_consideration": {
        "status": "disclosed_revenue_share_structure_amount_not_disclosed",
        "amount": null,
        "currency": null,
        "description": "Harell states that data owners receive a revenue share on every training job executed against their dataset; the percentage, minimum commitment, and cash amount are not disclosed. Model builders can also collect a fee when users run marketplace-listed models."
      },
      "estimated_economics": {
        "status": "not_estimated",
        "amount": null,
        "currency": null,
        "method": null
      },
      "economics_scope": "Per-training-run data-owner revenue share and usage fees for downstream Models as a Service; CoreWeave contract value and exact revenue-share percentages are not disclosed.",
      "rights": {
        "training": "Conditional and permitted inside Harell's controlled environment for authorized builders; raw data may not be viewed or extracted.",
        "post_training": "Fine-tuning is expressly permitted inside the controlled platform; other post-training rights are not disclosed.",
        "inference_retrieval": "Inference is expressly supported inside Harell; raw-record retrieval and extraction are prohibited by the disclosed architecture.",
        "ownership_transfer": "No dataset ownership transfer is disclosed. Raw datasets remain with data providers and do not leave the Harell environment.",
        "exclusivity": "not_disclosed",
        "retention": "Raw data remains inside the Harell environment; contract duration, deletion timing, and trained-model memorization controls are not disclosed.",
        "derived_model": "Model builders keep their trained models and associated intellectual property and may list models under a Models as a Service arrangement.",
        "sublicensing": "Data sublicensing is not disclosed. Marketplace use of trained models is disclosed, but this does not establish a right to sublicense the underlying datasets."
      },
      "restrictions": {
        "governed_access": "Training, fine-tuning, validation, and inference execute inside Harell's controlled platform; raw datasets cannot be viewed or extracted, and compute is attributable and metered per run.",
        "geographic_sovereignty": "not_disclosed",
        "privacy_pii": "The disclosed initial assets are scientific binding datasets. Patient-level content, consent terms, and privacy controls are not disclosed in the reviewed sources.",
        "trade_secret": "Raw proprietary datasets remain inside the platform and are not delivered to model builders; contractual remedies and technical exfiltration testing are not disclosed.",
        "de_identification": "not_disclosed",
        "employee_customer_data": "No employee or ordinary customer-operational corpus is identified; the assets are proprietary experimental scientific datasets from named data partners."
      },
      "refreshability": "Harell describes a growing partner and dataset ecosystem and per-run marketplace model; the cadence and exact update process for each initial dataset are not disclosed.",
      "historical_depth": "The data partners generated substantial proprietary experimental datasets, but the reviewed sources do not quantify collection start dates or longitudinal depth.",
      "decision_outcome_richness": "The initial datasets link biological sequences or structures to measured antibody-antigen affinity and T-cell-receptor binding outcomes, supporting model learning and blinded validation against experimental results.",
      "likely_model_use": "Training, fine-tuning, validating, and serving protein-language, structure-prediction, antibody-engineering, and immune-receptor models without transferring raw proprietary records.",
      "likely_strategic_value": "Provides governed access to scarce experimental biology data whose scale and measured binding outcomes cannot be reproduced from public literature alone, while preserving data-provider control and builder ownership of resulting models.",
      "m_and_a_implications": "Establishes a repeatable data-market structure in which scientific asset owners monetize controlled learning runs without selling or delivering the corpus, potentially increasing strategic value for experimental-data generators and secure model-training platforms.",
      "analysis_boundary": "The named data partners, initial dataset categories, controlled in-platform training and inference, raw-data non-extraction, builder model ownership, multi-year infrastructure agreement, and per-run data-owner revenue share are disclosed. Exact corpus sizes, dataset-specific update cadence, access prices, revenue-share percentages, exclusivity, geography, retention periods, patient-level content, downstream memorization controls, and measured model uplift are not disclosed. Chronology correction: the founder post is dated September 15, 2026 and already describes the same named datasets and learning-rights architecture; the September 23 release concerns supporting infrastructure. The first contract execution date, original provider agreement dates and historical page publication/version timestamps are not independently established. This record is excluded from the September 17, 2026 forecast cohort under the conservative first-disclosure treatment; the correction does not create or remove a registry event.",
      "confidence": "high",
      "scores": {
        "methodology_version": "1.0.0",
        "asset": {
          "score": 84,
          "coverage_pct": 92,
          "rationale": "Strategic Signal analysis. The named scientific corpora combine scarce proprietary experimental measurements with high-value biological outcomes and direct model-training utility; exact historical depth and corpus scale remain partly undisclosed.",
          "factor_ratings": {
            "uniqueness_scarcity": 4,
            "historical_depth": null,
            "decision_outcome_richness": 4,
            "decision_value_density": 4,
            "real_world_grounding": 5,
            "domain_value": 5,
            "refreshability": 3,
            "proprietary_advantage": 4.5,
            "model_learning_usefulness": 4.5,
            "non_replicability": 4,
            "rights_usability": 4
          },
          "factor_evidence": {
            "uniqueness_scarcity": {
              "basis": "analysis",
              "source_ids": [
                "E2",
                "E3"
              ],
              "rationale": "The platform exposes named proprietary antibody and immune-receptor datasets, and Harell states that A-Alpha Bio's scale exceeds publicly available alternatives."
            },
            "decision_outcome_richness": {
              "basis": "analysis",
              "source_ids": [
                "E2",
                "E3"
              ],
              "rationale": "Binding-affinity and receptor-binding measurements connect biological inputs to experimentally observed outcomes suitable for learning and blinded validation."
            },
            "decision_value_density": {
              "basis": "analysis",
              "source_ids": [
                "E2",
                "E3"
              ],
              "rationale": "Individual experimental outcomes can shape high-value protein engineering, immune medicine, and drug-discovery decisions."
            },
            "real_world_grounding": {
              "basis": "disclosed_fact",
              "source_ids": [
                "E2",
                "E3"
              ],
              "rationale": "The assets are experimentally generated affinity, structural, and immune-receptor data rather than synthetic-only benchmarks."
            },
            "domain_value": {
              "basis": "analysis",
              "source_ids": [
                "E1",
                "E3"
              ],
              "rationale": "Protein engineering and immune-driven medicine are economically consequential domains with direct scientific and therapeutic applications."
            },
            "refreshability": {
              "basis": "analysis",
              "source_ids": [
                "E1",
                "E3"
              ],
              "rationale": "Harell describes a growing ecosystem and future datasets, but does not disclose a contractual refresh cadence for the initial assets."
            },
            "proprietary_advantage": {
              "basis": "analysis",
              "source_ids": [
                "E1",
                "E2"
              ],
              "rationale": "Named data partners control experimental assets that builders cannot view or extract and that exceed public alternatives in scale or specificity."
            },
            "model_learning_usefulness": {
              "basis": "disclosed_fact",
              "source_ids": [
                "E1",
                "E3"
              ],
              "rationale": "The arrangement expressly supports model training, fine-tuning, inference, and blinded validation on the proprietary scientific data."
            },
            "non_replicability": {
              "basis": "analysis",
              "source_ids": [
                "E2",
                "E3"
              ],
              "rationale": "Reproducing the precise experimental measurements would require comparable biological platforms, samples, protocols, time, and cost."
            },
            "rights_usability": {
              "basis": "analysis",
              "source_ids": [
                "E1"
              ],
              "rationale": "Controlled in-platform execution makes the data usable for training while preserving raw-data boundaries, though detailed contract terms remain undisclosed."
            }
          }
        },
        "transaction_signal": {
          "score": 71,
          "coverage_pct": 85,
          "rationale": "Strategic Signal analysis. The arrangement cleanly separates controlled data use from raw-data transfer and ties data-owner compensation to training runs, but provides no dollar value, share percentage, bid process, or independent uplift result.",
          "factor_ratings": {
            "disclosed_economics": 2.5,
            "clean_price_discovery": 1.5,
            "data_consideration_separability": 4.5,
            "explicit_model_use": 5,
            "rights_clarity": 4.5,
            "competing_bids": null,
            "strategic_buyer_quality": 3,
            "precedent_value": 5,
            "independent_model_uplift": null
          },
          "factor_evidence": {
            "disclosed_economics": {
              "basis": "disclosed_fact",
              "source_ids": [
                "E1",
                "E4"
              ],
              "rationale": "A per-training-run revenue share and downstream model fees are disclosed, but no dollar amount, percentage, or minimum commitment is provided."
            },
            "clean_price_discovery": {
              "basis": "analysis",
              "source_ids": [
                "E1"
              ],
              "rationale": "Metered runs create an attributable usage base, but the private CoreWeave agreement and undisclosed share terms do not reveal a clean market price."
            },
            "data_consideration_separability": {
              "basis": "analysis",
              "source_ids": [
                "E1",
                "E4"
              ],
              "rationale": "Data-owner compensation is expressly linked to each training job against a dataset, making the data consideration unusually separable from general software subscriptions."
            },
            "explicit_model_use": {
              "basis": "disclosed_fact",
              "source_ids": [
                "E1"
              ],
              "rationale": "Training, fine-tuning, validation, and inference are all expressly identified as supported workloads."
            },
            "rights_clarity": {
              "basis": "analysis",
              "source_ids": [
                "E1"
              ],
              "rationale": "Raw-data non-extraction, controlled execution, builder model ownership, and marketplace use are clear; retention, sublicensing, and exclusivity remain undisclosed."
            },
            "strategic_buyer_quality": {
              "basis": "analysis",
              "source_ids": [
                "E1",
                "E4"
              ],
              "rationale": "The platform targets serious scientific model builders and runs on a scaled AI cloud, but no named frontier model developer is disclosed as a customer."
            },
            "precedent_value": {
              "basis": "analysis",
              "source_ids": [
                "E1",
                "E4"
              ],
              "rationale": "Per-run compensation for in-place proprietary data use is a repeatable template for scientific and other high-value data markets."
            }
          }
        }
      },
      "sources": [
        {
          "type": "primary",
          "publisher": "CoreWeave and Harell Data",
          "title": "Harell Data Selects CoreWeave to Power Its Secure Platform for AI Model Training",
          "url": "https://investors.coreweave.com/news/news-details/2026/Harell-Data-Selects-CoreWeave-to-Power-Its-Secure-Platform-for-AI-Model-Training/default.aspx"
        },
        {
          "type": "primary",
          "publisher": "Harell Data",
          "title": "Data Partners",
          "url": "https://harelldata.com/data-partners/"
        },
        {
          "type": "primary",
          "publisher": "Harell Data",
          "title": "Why we started Harell Data",
          "url": "https://harelldata.com/why-we-started-harell-data/"
        },
        {
          "type": "secondary",
          "publisher": "GeekWire",
          "title": "Adaptive Biotech co-founder raises $15M for new startup to rethink how AI trains on science",
          "url": "https://www.geekwire.com/2026/adaptive-biotech-co-founder-raises-15m-for-new-startup-to-rethink-how-ai-trains-on-science/"
        }
      ],
      "date_added": "2026-09-26",
      "date_last_reviewed": "2026-09-30",
      "announcement_date_basis": "Date displayed on Harell's founder post describing the same two live datasets and governed learning arrangement. This is the earliest source-stated disclosure identified in this review, not a verified agreement execution date or proof of first-ever public availability. The September 23 CoreWeave release is recorded separately as infrastructure news.",
      "event_history": [
        {
          "date": "2026-09-15",
          "event": "Founder post describes the existing controlled learning platform, A-Alpha Bio and Adaptive Biotechnologies datasets, retained data, model IP and per-training compensation.",
          "count_as_new_transaction": true
        },
        {
          "date": "2026-09-23",
          "event": "CoreWeave announces a multi-year infrastructure agreement supporting Harell's platform; no separately evidenced new underlying data-rights arrangement is counted.",
          "count_as_new_transaction": false
        }
      ],
      "publication_review": {
        "status": "human_reviewed",
        "approval_channel": "owner_conversation",
        "reviewer": "Mike Ye",
        "reviewed_at": "2026-09-30T19:34:37.142Z",
        "decision_id": "conversation-20260930-2AEBFCD969B7",
        "packet_sha256": "56a04ca0b550881fe5f41794246d96f8b36ed91f5268a6fc2d7e7ba0c2ba3724",
        "record_sha256": "d4ef8da3e3de774ff279956f7058b3a294d0b60b6726fdc91d8baee8a3799611"
      }
    },
    {
      "transaction_id": "SS-AIDT-2026-0012",
      "slug": "tempus-recursion-oncology-data-license",
      "canonical_path": "/ai-data-transactions/deals/ss-aidt-2026-0012/",
      "announcement_date": "2023-11-03",
      "buyer_ai_developer": [
        "Recursion Pharmaceuticals"
      ],
      "seller_data_owner": [
        "Tempus AI"
      ],
      "transaction_status": "active_amended_six_year_license",
      "industry": "Healthcare, oncology, biotechnology and AI drug discovery",
      "data_category": [
        "De-identified clinical oncology data",
        "Molecular and genomic oncology data",
        "Patient-centric multimodal records"
      ],
      "corpus_description": "Tempus's proprietary de-identified patient-centric multimodal oncology database of clinical and molecular records, originally described by Recursion as more than 20 petabytes, licensed for therapeutic-development AI/ML use subject to unique-record caps and retention limits.",
      "primary_transaction_structure": "Training Rights License",
      "access_structure": "Recursion receives limited licensed access to Tempus's proprietary de-identified clinical and molecular data for therapeutic product development, including model training, improvement, modification and derivative works. Downloaded records are subject to per-time and aggregate unique-record caps and historically a 180-day retention limit. The September 2026 amendment extends the term to six years, removes convenience termination and lowers the remaining unique-record cap.",
      "asset_disposition": "Embedded / Continuing",
      "corpus_renewal": "Renewable Corpus",
      "source_code_software_asset": "no",
      "disclosed_consideration": {
        "status": "disclosed_contractual_license_fees",
        "amount": 42000000,
        "currency": "USD",
        "description": "The 2026 amendment sets $14 million annual license fees on each of the third, fourth and fifth anniversaries, at least $4 million of each payable in cash and the remainder in cash and/or Recursion shares. The original 2023 deal disclosed up to $160 million over five years."
      },
      "estimated_economics": {
        "status": "not_estimated",
        "amount": null,
        "currency": null,
        "method": null
      },
      "economics_scope": "$42 million of amended third-through-fifth anniversary license fees is explicitly disclosed; amounts already paid under the original schedule and the economic effect of the reduced unique-record cap are not restated here.",
      "rights": {
        "training": "Permitted for Recursion and affiliates to develop and train AI/ML models solely for therapeutic product development, subject to the agreement's limits.",
        "post_training": "Permitted to improve and modify Recursion AI/ML models and develop embeddings, training and test data, subject to contractual restrictions.",
        "inference_retrieval": "Licensed data may be accessed and downloaded within stated caps for permitted therapeutic-development uses; unrestricted retrieval or redistribution is not permitted.",
        "ownership_transfer": "No ownership transfer. Tempus retains ownership of Licensed Data and Tempus Materials.",
        "exclusivity": "not_disclosed",
        "retention": "The filed 2023 disclosure states downloaded records may be retained for 180 days; the 2026 amendment does not disclose a change to that limit.",
        "derived_model": "Recursion may create derivative works of its AI/ML models and use, deploy, commercialize or license those models subject to the agreement; underlying licensed data remain Tempus property.",
        "sublicensing": "Certain third-party therapeutic collaborations are allowed under defined conditions, but no general right to sublicense the underlying dataset is established."
      },
      "restrictions": {
        "governed_access": "Use is restricted to therapeutic product development within a secure Recursion environment and subject to download, aggregate unique-record and retention controls.",
        "geographic_sovereignty": "not_disclosed",
        "privacy_pii": "The licensed corpus is disclosed as de-identified clinical and molecular data and governed by healthcare-data restrictions.",
        "trade_secret": "Tempus retains ownership of its proprietary database and related materials; permitted uses do not authorize unrestricted reproduction or transfer.",
        "de_identification": "Licensed records are de-identified; the agreement references HIPAA-aligned de-identification treatment for relevant data.",
        "employee_customer_data": "The asset contains patient-centric clinical and molecular oncology records rather than ordinary employee or customer-operational data."
      },
      "refreshability": "The arrangement spans six years and grants continuing licensed access subject to aggregate unique-record limits; the exact refresh cadence and amended unique-record count are not disclosed.",
      "historical_depth": "Tempus's oncology database aggregates clinical and molecular patient records accumulated over time; the exact date range and longitudinal completeness are not disclosed in the amendment.",
      "decision_outcome_richness": "The corpus supports therapeutic hypotheses, biomarker strategies, patient stratification and model learning from linked real-world clinical and molecular observations.",
      "likely_model_use": "Training, improving, modifying and creating derivative works of Recursion causal AI/ML models for therapeutic product development, plus biomarker and patient-stratification work.",
      "likely_strategic_value": "Provides Recursion with large-scale patient-centric multimodal oncology data that complements its proprietary interventional biology and chemistry data and directly supports AI-driven drug discovery.",
      "m_and_a_implications": "Demonstrates that a proprietary healthcare-data asset can command large, separately stated multi-year licensing economics when explicit AI-training and derivative-model rights are granted without selling the underlying corpus.",
      "analysis_boundary": "Primary SEC filings establish the proprietary corpus, explicit AI/ML training and derivative-work rights, original and amended economics, six-year term, removal of convenience termination, lower aggregate unique-record cap, ownership boundaries and de-identification. The amended exact record cap, exclusivity, current refresh cadence, geographic restrictions and independent model-uplift evidence are not disclosed.",
      "confidence": "high",
      "scores": {
        "methodology_version": "1.0.0",
        "asset": {
          "score": 94,
          "coverage_pct": 100,
          "rationale": "Strategic Signal analysis. Tempus's large proprietary multimodal oncology corpus combines real-world clinical and molecular observations with explicit AI-training utility and strong therapeutic decision relevance.",
          "factor_ratings": {
            "uniqueness_scarcity": 4.5,
            "historical_depth": 4.5,
            "decision_outcome_richness": 4.5,
            "decision_value_density": 5,
            "real_world_grounding": 5,
            "domain_value": 5,
            "refreshability": 4.5,
            "proprietary_advantage": 4.5,
            "model_learning_usefulness": 5,
            "non_replicability": 4,
            "rights_usability": 4.5
          },
          "factor_evidence": {
            "uniqueness_scarcity": {
              "basis": "analysis",
              "source_ids": [
                "E2",
                "E3"
              ],
              "rationale": "Evidence supports uniqueness scarcity as a material characteristic of the Tempus oncology corpus and its AI-training use, with remaining limits preserved explicitly."
            },
            "historical_depth": {
              "basis": "analysis",
              "source_ids": [
                "E2",
                "E3"
              ],
              "rationale": "Evidence supports historical depth as a material characteristic of the Tempus oncology corpus and its AI-training use, with remaining limits preserved explicitly."
            },
            "decision_outcome_richness": {
              "basis": "analysis",
              "source_ids": [
                "E2",
                "E3"
              ],
              "rationale": "Evidence supports decision outcome richness as a material characteristic of the Tempus oncology corpus and its AI-training use, with remaining limits preserved explicitly."
            },
            "decision_value_density": {
              "basis": "analysis",
              "source_ids": [
                "E2",
                "E3"
              ],
              "rationale": "Evidence supports decision value density as a material characteristic of the Tempus oncology corpus and its AI-training use, with remaining limits preserved explicitly."
            },
            "real_world_grounding": {
              "basis": "disclosed_fact",
              "source_ids": [
                "E2",
                "E3"
              ],
              "rationale": "Evidence supports real world grounding as a material characteristic of the Tempus oncology corpus and its AI-training use, with remaining limits preserved explicitly."
            },
            "domain_value": {
              "basis": "analysis",
              "source_ids": [
                "E2",
                "E3"
              ],
              "rationale": "Evidence supports domain value as a material characteristic of the Tempus oncology corpus and its AI-training use, with remaining limits preserved explicitly."
            },
            "refreshability": {
              "basis": "analysis",
              "source_ids": [
                "E2",
                "E3"
              ],
              "rationale": "Evidence supports refreshability as a material characteristic of the Tempus oncology corpus and its AI-training use, with remaining limits preserved explicitly."
            },
            "proprietary_advantage": {
              "basis": "analysis",
              "source_ids": [
                "E2",
                "E3"
              ],
              "rationale": "Evidence supports proprietary advantage as a material characteristic of the Tempus oncology corpus and its AI-training use, with remaining limits preserved explicitly."
            },
            "model_learning_usefulness": {
              "basis": "disclosed_fact",
              "source_ids": [
                "E2",
                "E3"
              ],
              "rationale": "Evidence supports model learning usefulness as a material characteristic of the Tempus oncology corpus and its AI-training use, with remaining limits preserved explicitly."
            },
            "non_replicability": {
              "basis": "analysis",
              "source_ids": [
                "E2",
                "E3"
              ],
              "rationale": "Evidence supports non replicability as a material characteristic of the Tempus oncology corpus and its AI-training use, with remaining limits preserved explicitly."
            },
            "rights_usability": {
              "basis": "analysis",
              "source_ids": [
                "E2",
                "E3"
              ],
              "rationale": "Evidence supports rights usability as a material characteristic of the Tempus oncology corpus and its AI-training use, with remaining limits preserved explicitly."
            }
          }
        },
        "transaction_signal": {
          "score": 96,
          "coverage_pct": 85,
          "rationale": "Strategic Signal analysis. The agreement has unusually explicit model-use rights and separately disclosed multi-year license economics, strengthened by a filed 2026 amendment; no competitive process or independent uplift study is disclosed.",
          "factor_ratings": {
            "disclosed_economics": 5,
            "clean_price_discovery": 4,
            "data_consideration_separability": 5,
            "explicit_model_use": 5,
            "rights_clarity": 5,
            "competing_bids": null,
            "strategic_buyer_quality": 4.5,
            "precedent_value": 5,
            "independent_model_uplift": null
          },
          "factor_evidence": {
            "disclosed_economics": {
              "basis": "disclosed_fact",
              "source_ids": [
                "E1",
                "E2",
                "E3"
              ],
              "rationale": "The filed agreement and amendment provide sufficient evidence to rate disclosed economics for this separately priced AI data-rights arrangement."
            },
            "clean_price_discovery": {
              "basis": "analysis",
              "source_ids": [
                "E1",
                "E2",
                "E3"
              ],
              "rationale": "The filed agreement and amendment provide sufficient evidence to rate clean price discovery for this separately priced AI data-rights arrangement."
            },
            "data_consideration_separability": {
              "basis": "analysis",
              "source_ids": [
                "E1",
                "E2",
                "E3"
              ],
              "rationale": "The filed agreement and amendment provide sufficient evidence to rate data consideration separability for this separately priced AI data-rights arrangement."
            },
            "explicit_model_use": {
              "basis": "disclosed_fact",
              "source_ids": [
                "E1",
                "E2",
                "E3"
              ],
              "rationale": "The filed agreement and amendment provide sufficient evidence to rate explicit model use for this separately priced AI data-rights arrangement."
            },
            "rights_clarity": {
              "basis": "disclosed_fact",
              "source_ids": [
                "E1",
                "E2",
                "E3"
              ],
              "rationale": "The filed agreement and amendment provide sufficient evidence to rate rights clarity for this separately priced AI data-rights arrangement."
            },
            "strategic_buyer_quality": {
              "basis": "analysis",
              "source_ids": [
                "E1",
                "E2",
                "E3"
              ],
              "rationale": "The filed agreement and amendment provide sufficient evidence to rate strategic buyer quality for this separately priced AI data-rights arrangement."
            },
            "precedent_value": {
              "basis": "analysis",
              "source_ids": [
                "E1",
                "E2",
                "E3"
              ],
              "rationale": "The filed agreement and amendment provide sufficient evidence to rate precedent value for this separately priced AI data-rights arrangement."
            }
          }
        }
      },
      "sources": [
        {
          "type": "primary",
          "publisher": "Recursion Pharmaceuticals / SEC",
          "title": "September 15, 2026 Form 8-K — Tempus Master Agreement Amendment",
          "url": "https://www.sec.gov/Archives/edgar/data/1601830/000160183026000112/rxrx-20260915.htm"
        },
        {
          "type": "primary",
          "publisher": "Recursion Pharmaceuticals / SEC",
          "title": "November 9, 2023 Form 8-K — Tempus Master Agreement",
          "url": "https://www.sec.gov/Archives/edgar/data/1601830/000160183023000068/rxrx-20231109.htm"
        },
        {
          "type": "primary",
          "publisher": "Recursion Pharmaceuticals / SEC",
          "title": "Tempus Master Agreement Exhibit",
          "url": "https://www.sec.gov/Archives/edgar/data/1601830/000160183023000069/exhibit104-masteragreement.htm"
        },
        {
          "type": "primary",
          "publisher": "Recursion Pharmaceuticals",
          "title": "Recursion Announces Data Collaboration Deal with Tempus",
          "url": "https://www.sec.gov/Archives/edgar/data/1601830/000160183023000068/exhibit992-q0323.htm"
        }
      ],
      "date_added": "2026-09-29",
      "date_last_reviewed": "2026-09-29",
      "publication_review": {
        "status": "human_reviewed",
        "approval_channel": "owner_conversation",
        "reviewer": "Mike Ye",
        "reviewed_at": "2026-09-29T18:02:23.696Z",
        "decision_id": "conversation-20260929-1C674D15A8B4",
        "packet_sha256": "23a5d5b381767ab928eeb336bd5bad568134f4a95c5968247b856aa36360be0e",
        "record_sha256": "c4b83271b0cdc8d5f6cbf96f6f12176eaaba7ff0ca039c4dd314f7dd504b3461"
      }
    },
    {
      "transaction_id": "SS-AIDT-2026-0013",
      "slug": "guideai-vrad-exclusive-imaging-ai-rights",
      "canonical_path": "/ai-data-transactions/deals/ss-aidt-2026-0013/",
      "announcement_date": "2026-09-28",
      "buyer_ai_developer": [
        "GuideAI Health"
      ],
      "seller_data_owner": [
        "vRad (Virtual Radiologic)"
      ],
      "transaction_status": "active_renewed_exclusive_data_agreement",
      "industry": "Healthcare, medical imaging and clinical AI",
      "data_category": [
        "De-identified radiology imaging data",
        "Routine CT imaging",
        "Vascular disease imaging corpus"
      ],
      "corpus_description": "De-identified imaging data generated across vRad's national network of more than 2,100 U.S. hospitals and healthcare facilities, made exclusively available to GuideAI for development, training and validation of AI models for vascular-disease detection and characterization.",
      "primary_transaction_structure": "Training Rights License",
      "access_structure": "Under the renewed data exclusivity agreement, GuideAI retains exclusive rights to use de-identified imaging generated across vRad's network to develop AI models. The issuer states that the dataset provides the foundation to train and validate GuideAI algorithms; detailed delivery, retention and downstream model-rights mechanics are not disclosed.",
      "asset_disposition": "Embedded / Continuing",
      "corpus_renewal": "Renewable Corpus",
      "source_code_software_asset": "no",
      "disclosed_consideration": {
        "status": "not_disclosed",
        "amount": null,
        "currency": null,
        "description": "The renewed agreement's financial consideration, minimum commitments and pricing formula are not disclosed."
      },
      "estimated_economics": {
        "status": "not_estimated",
        "amount": null,
        "currency": null,
        "method": null
      },
      "economics_scope": "No agreement value, per-study pricing, revenue share, equity component or other economic terms are disclosed.",
      "rights": {
        "training": "Permitted and expressly contemplated: GuideAI states that the imaging corpus provides a foundation to train its vascular-disease algorithms.",
        "post_training": "Validation of GuideAI algorithms is expressly contemplated; fine-tuning and other post-training rights are not separately described.",
        "inference_retrieval": "GuideAI has exclusive access to use the de-identified imaging data for AI-model development; record-level retrieval mechanics are not disclosed.",
        "ownership_transfer": "No ownership transfer of vRad imaging data is disclosed.",
        "exclusivity": "GuideAI states that it retains exclusive rights to develop AI models using the de-identified imaging data generated across vRad's network.",
        "retention": "not_disclosed",
        "derived_model": "GuideAI is permitted to develop AI models from the licensed imaging corpus; detailed ownership language for resulting models is not disclosed.",
        "sublicensing": "not_disclosed"
      },
      "restrictions": {
        "governed_access": "Access is limited to the renewed vRad data partnership and AI-model development scope; detailed technical access controls are not disclosed.",
        "geographic_sovereignty": "The disclosed source network consists of U.S. hospitals and healthcare facilities; processing-location restrictions are not disclosed.",
        "privacy_pii": "The licensed imaging data are disclosed as de-identified, with healthcare privacy and data-protection compliance identified as relevant risks.",
        "trade_secret": "The dataset is treated as an exclusive commercial data asset; technical confidentiality controls are not disclosed.",
        "de_identification": "The announcement expressly describes the imaging data as de-identified.",
        "employee_customer_data": "The asset consists of de-identified patient imaging generated in clinical care rather than employee or ordinary commercial customer data."
      },
      "refreshability": "The renewed partnership covers imaging generated across an ongoing network of more than 2,100 hospitals and healthcare facilities, indicating a renewable flow; exact refresh cadence is not disclosed.",
      "historical_depth": "The arrangement is a renewal of an existing data partnership, but the reviewed source does not disclose the original start date, number of studies or years of imaging history.",
      "decision_outcome_richness": "The imaging corpus contains clinically meaningful disease signals used for detection and characterization, but the disclosure does not establish linked treatment interventions and outcomes.",
      "likely_model_use": "Training, validating and expanding GuideAI models for detection and characterization of peripheral arterial and other vascular diseases from routine CT imaging.",
      "likely_strategic_value": "Exclusive access to a large, diverse, real-world national imaging corpus can improve model coverage and create a proprietary data advantage in vascular-disease AI.",
      "m_and_a_implications": "Shows how a large clinical-network data owner can grant exclusive AI-model development rights without transferring ownership of the underlying corpus, potentially creating strategic scarcity for specialized healthcare AI developers.",
      "analysis_boundary": "The issuer release establishes the renewed exclusive rights, de-identified imaging scope, more than 2,100-hospital network, approximately 500-radiologist workflow, and training/validation purpose. Economics, term length, exact corpus size, delivery architecture, retention, derived-model ownership, sublicensing, geographic processing controls and independent model-uplift evidence are not disclosed.",
      "confidence": "high",
      "scores": {
        "methodology_version": "1.0.0",
        "asset": {
          "score": 93,
          "coverage_pct": 100,
          "rationale": "Strategic Signal analysis. The exclusive national imaging corpus is large, real-world, renewable and directly useful for clinical-model training, though exact longitudinal depth and patient-level outcome linkage are not disclosed.",
          "factor_ratings": {
            "uniqueness_scarcity": 4.5,
            "historical_depth": 4,
            "decision_outcome_richness": 4,
            "decision_value_density": 5,
            "real_world_grounding": 5,
            "domain_value": 5,
            "refreshability": 5,
            "proprietary_advantage": 5,
            "model_learning_usefulness": 5,
            "non_replicability": 4.5,
            "rights_usability": 4
          },
          "factor_evidence": {
            "uniqueness_scarcity": {
              "basis": "analysis",
              "source_ids": [
                "E1"
              ],
              "rationale": "The issuer disclosure provides sufficient evidence to rate uniqueness scarcity for the exclusive vRad imaging corpus while preserving undisclosed limits."
            },
            "historical_depth": {
              "basis": "analysis",
              "source_ids": [
                "E1"
              ],
              "rationale": "The issuer disclosure provides sufficient evidence to rate historical depth for the exclusive vRad imaging corpus while preserving undisclosed limits."
            },
            "decision_outcome_richness": {
              "basis": "analysis",
              "source_ids": [
                "E1"
              ],
              "rationale": "The issuer disclosure provides sufficient evidence to rate decision outcome richness for the exclusive vRad imaging corpus while preserving undisclosed limits."
            },
            "decision_value_density": {
              "basis": "analysis",
              "source_ids": [
                "E1"
              ],
              "rationale": "The issuer disclosure provides sufficient evidence to rate decision value density for the exclusive vRad imaging corpus while preserving undisclosed limits."
            },
            "real_world_grounding": {
              "basis": "disclosed_fact",
              "source_ids": [
                "E1"
              ],
              "rationale": "The issuer disclosure provides sufficient evidence to rate real world grounding for the exclusive vRad imaging corpus while preserving undisclosed limits."
            },
            "domain_value": {
              "basis": "analysis",
              "source_ids": [
                "E1"
              ],
              "rationale": "The issuer disclosure provides sufficient evidence to rate domain value for the exclusive vRad imaging corpus while preserving undisclosed limits."
            },
            "refreshability": {
              "basis": "disclosed_fact",
              "source_ids": [
                "E1"
              ],
              "rationale": "The issuer disclosure provides sufficient evidence to rate refreshability for the exclusive vRad imaging corpus while preserving undisclosed limits."
            },
            "proprietary_advantage": {
              "basis": "disclosed_fact",
              "source_ids": [
                "E1"
              ],
              "rationale": "The issuer disclosure provides sufficient evidence to rate proprietary advantage for the exclusive vRad imaging corpus while preserving undisclosed limits."
            },
            "model_learning_usefulness": {
              "basis": "disclosed_fact",
              "source_ids": [
                "E1"
              ],
              "rationale": "The issuer disclosure provides sufficient evidence to rate model learning usefulness for the exclusive vRad imaging corpus while preserving undisclosed limits."
            },
            "non_replicability": {
              "basis": "analysis",
              "source_ids": [
                "E1"
              ],
              "rationale": "The issuer disclosure provides sufficient evidence to rate non replicability for the exclusive vRad imaging corpus while preserving undisclosed limits."
            },
            "rights_usability": {
              "basis": "analysis",
              "source_ids": [
                "E1"
              ],
              "rationale": "The issuer disclosure provides sufficient evidence to rate rights usability for the exclusive vRad imaging corpus while preserving undisclosed limits."
            }
          }
        },
        "transaction_signal": {
          "score": 55,
          "coverage_pct": 85,
          "rationale": "Strategic Signal analysis. Exclusive model-development rights are unusually explicit, but economics and market price discovery are absent, materially limiting the transaction signal despite strong AI-use and rights clarity.",
          "factor_ratings": {
            "disclosed_economics": 0,
            "clean_price_discovery": 0,
            "data_consideration_separability": 4,
            "explicit_model_use": 5,
            "rights_clarity": 4.5,
            "competing_bids": null,
            "strategic_buyer_quality": 3.5,
            "precedent_value": 4.5,
            "independent_model_uplift": null
          },
          "factor_evidence": {
            "disclosed_economics": {
              "basis": "disclosed_fact",
              "source_ids": [
                "E1"
              ],
              "rationale": "The renewed agreement provides sufficient evidence to rate disclosed economics for this exclusive AI-model development data-rights arrangement."
            },
            "clean_price_discovery": {
              "basis": "analysis",
              "source_ids": [
                "E1"
              ],
              "rationale": "The renewed agreement provides sufficient evidence to rate clean price discovery for this exclusive AI-model development data-rights arrangement."
            },
            "data_consideration_separability": {
              "basis": "analysis",
              "source_ids": [
                "E1"
              ],
              "rationale": "The renewed agreement provides sufficient evidence to rate data consideration separability for this exclusive AI-model development data-rights arrangement."
            },
            "explicit_model_use": {
              "basis": "disclosed_fact",
              "source_ids": [
                "E1"
              ],
              "rationale": "The renewed agreement provides sufficient evidence to rate explicit model use for this exclusive AI-model development data-rights arrangement."
            },
            "rights_clarity": {
              "basis": "disclosed_fact",
              "source_ids": [
                "E1"
              ],
              "rationale": "The renewed agreement provides sufficient evidence to rate rights clarity for this exclusive AI-model development data-rights arrangement."
            },
            "strategic_buyer_quality": {
              "basis": "analysis",
              "source_ids": [
                "E1"
              ],
              "rationale": "The renewed agreement provides sufficient evidence to rate strategic buyer quality for this exclusive AI-model development data-rights arrangement."
            },
            "precedent_value": {
              "basis": "analysis",
              "source_ids": [
                "E1"
              ],
              "rationale": "The renewed agreement provides sufficient evidence to rate precedent value for this exclusive AI-model development data-rights arrangement."
            }
          }
        }
      },
      "sources": [
        {
          "type": "primary_syndicated",
          "publisher": "GuideAI Health",
          "title": "GuideAI Health Corp. and vRad Renew Data Exclusivity Agreement to Develop AI Models",
          "url": "https://www.globenewswire.com/news-release/2026/09/28/3369827/0/en/guideai-health-corp-and-vrad-renew-data-exclusivity-agreement-to-develop-ai-models-for-the-detection-and-characterization-of-multiple-vascular-diseases.html"
        }
      ],
      "date_added": "2026-09-29",
      "date_last_reviewed": "2026-09-29",
      "publication_review": {
        "status": "human_reviewed",
        "approval_channel": "owner_conversation",
        "reviewer": "Mike Ye",
        "reviewed_at": "2026-09-29T18:02:23.716Z",
        "decision_id": "conversation-20260929-22F00D6C2ABE",
        "packet_sha256": "8534eef99edbad6cbffa7a0a9b29177561506eb77207ffc61f8d68ac30f125f4",
        "record_sha256": "498d385cb70c4250f97bb7a0856be2a422c066b9f4d49cd94f96f875fa453cc3"
      }
    },
    {
      "transaction_id": "SS-AIDT-2026-0014",
      "slug": "onemednet-precision-medicine-rwd-training-license",
      "canonical_path": "/ai-data-transactions/deals/ss-aidt-2026-0014/",
      "announcement_date": "2026-09-30",
      "buyer_ai_developer": [
        "Undisclosed commercially scaled AI-enabled precision-medicine company"
      ],
      "seller_data_owner": [
        "OneMedNet"
      ],
      "transaction_status": "active_multiyear_data_license",
      "industry": "Healthcare, precision medicine and clinical AI",
      "data_category": [
        "De-identified multimodal clinical real-world data",
        "Medical imaging",
        "Longitudinal clinical records and patient journeys"
      ],
      "corpus_description": "De-identified, multimodal, longitudinal clinical real-world data sourced directly through OneMedNet's network of more than 2,300 healthcare partner sites, including medical imaging and longitudinal clinical journeys, licensed for AI model training and incorporation into a commercial derivative dataset.",
      "primary_transaction_structure": "Training Rights License",
      "access_structure": "Under a seven-figure multiyear agreement, OneMedNet will deliver de-identified multimodal longitudinal clinical RWD through its iRWD platform to an undisclosed commercially scaled AI-enabled precision-medicine company. The customer intends to train clinical and diagnostic AI models and create a licensable derivative dataset combining its molecular data with OneMedNet imaging and longitudinal clinical journeys. OneMedNet retains source-data rights and receives incremental annual sublicensing revenue when the derivative solution is licensed to end customers.",
      "asset_disposition": "Embedded / Continuing",
      "corpus_renewal": "Renewable Corpus",
      "source_code_software_asset": "no",
      "disclosed_consideration": {
        "status": "reported_range_only",
        "amount": null,
        "currency": null,
        "description": "Seven-figure multiyear agreement, plus incremental annual sublicensing revenue to OneMedNet for each downstream end-customer license; exact contract value, currency, payment form, annual schedule and per-license revenue share are not disclosed in the reviewed issuer announcement."
      },
      "estimated_economics": {
        "status": "not_estimated",
        "amount": null,
        "currency": null,
        "method": null
      },
      "economics_scope": "The disclosed seven-figure amount applies to the multiyear data license. OneMedNet separately states that it receives recurring sublicensing revenue on each downstream license of the derivative product and that the agreement includes an option to expand data volume and contract value; exact values are not disclosed.",
      "rights": {
        "training": "Permitted and explicit: the customer intends to use OneMedNet data to train AI-enabled clinical and diagnostic models.",
        "post_training": "not_disclosed",
        "inference_retrieval": "The customer receives licensed access to the data through OneMedNet's iRWD platform; detailed retrieval mechanics and inference-only permissions are not separately disclosed.",
        "ownership_transfer": "Prohibited for the source data: OneMedNet expressly states that it retains source-data rights.",
        "exclusivity": "not_disclosed",
        "retention": "not_disclosed",
        "derived_model": "Permitted in substance: the customer may train clinical and diagnostic AI models and create a licensable derivative dataset combining its molecular data with OneMedNet imaging and longitudinal clinical journeys; ownership terms for resulting models are not separately disclosed.",
        "sublicensing": "Conditional and explicit for the derivative product: the customer may license the derivative dataset to end customers, OneMedNet receives incremental annual sublicensing revenue, and end customers may not relicense or resell the underlying OneMedNet data."
      },
      "restrictions": {
        "governed_access": "Data are delivered under a multiyear commercial license through OneMedNet's iRWD platform; OneMedNet retains source-data rights and derivative end customers may not relicense or resell the underlying OneMedNet data.",
        "geographic_sovereignty": "not_disclosed",
        "privacy_pii": "The licensed clinical data are disclosed as de-identified; the release does not disclose patient-level consent terms or the detailed privacy architecture.",
        "trade_secret": "The source corpus and derivative commercialization rights are governed by the commercial license; additional confidentiality terms are not disclosed.",
        "de_identification": "OneMedNet states that data are de-identified at source and designed to meet standards for AI training and life-sciences commercial use.",
        "employee_customer_data": "The asset consists of de-identified healthcare patient data and clinical records rather than employee or ordinary commercial customer data."
      },
      "refreshability": "The arrangement is multiyear, uses a live network of more than 2,300 healthcare partner sites, includes recurring downstream licensing and provides an option to expand data volume, supporting a renewable rather than one-time corpus.",
      "historical_depth": "OneMedNet describes longitudinal clinical journeys and a network encompassing more than 90 million patient journeys and 270 million studies, but the specific licensed subset's years of history and patient count are not disclosed.",
      "decision_outcome_richness": "The licensed corpus links imaging and longitudinal clinical records across patient journeys and is intended for clinical and diagnostic model development. The reviewed disclosure does not establish accessible context-intervention-outcome linkage sufficient to infer canonical Human Response H2 or H3.",
      "likely_model_use": "Training clinical and diagnostic AI models and constructing a commercial derivative dataset that combines OneMedNet imaging and longitudinal clinical journeys with the customer's molecular datasets.",
      "likely_strategic_value": "The agreement gives the AI customer recurring access to direct-from-source multimodal clinical data for model learning while creating a derivative-data product with recurring downstream monetization, making both the corpus and the rights architecture strategically valuable.",
      "m_and_a_implications": "The transaction provides observable evidence that proprietary healthcare data can support a layered monetization model: multiyear training-license economics plus participation in downstream derivative-data revenue while the source owner retains the underlying data rights.",
      "analysis_boundary": "The issuer release establishes a seven-figure multiyear license, explicit AI-model training, derivative-dataset commercialization, recurring downstream sublicensing revenue, retained source-data rights, de-identification, more than 2,300 healthcare partner sites, and an option to expand volume and value. The counterparty name, exact contract value, annual payment schedule, sublicensing rate, licensed subset size, historical years, exclusivity, retention, post-training and evaluation rights, geographic processing terms, patient consent details, model ownership and independent model-uplift evidence are not disclosed. Longitudinal patient journeys do not by themselves establish Human Response H2/H3 linkage. Currency correction: the reviewed issuer announcement uses 'seven figure' without explicitly identifying a currency or settlement form. The earlier USD field is withdrawn in this successor record; no numeric range, cash realization or currency conversion is inferred.",
      "confidence": "high",
      "scores": {
        "methodology_version": "1.0.0",
        "asset": {
          "score": 94,
          "coverage_pct": 100,
          "rationale": "Strategic Signal analysis. OneMedNet's direct-from-source multimodal clinical corpus is large, longitudinal, renewable, real-world and explicitly useful for AI training, with strong scarcity and derivative-product utility; exact licensed-subset depth and intervention-to-outcome linkage remain undisclosed.",
          "factor_ratings": {
            "uniqueness_scarcity": 4.5,
            "historical_depth": 4,
            "decision_outcome_richness": 4,
            "decision_value_density": 5,
            "real_world_grounding": 5,
            "domain_value": 5,
            "refreshability": 5,
            "proprietary_advantage": 5,
            "model_learning_usefulness": 5,
            "non_replicability": 4.5,
            "rights_usability": 4.5
          },
          "factor_evidence": {
            "uniqueness_scarcity": {
              "basis": "analysis",
              "source_ids": [
                "E1"
              ],
              "rationale": "The direct-from-source network, multimodal imaging plus clinical journeys, and scale across more than 2,300 healthcare sites support a high scarcity rating while the exact licensed subset remains undisclosed."
            },
            "historical_depth": {
              "basis": "analysis",
              "source_ids": [
                "E1"
              ],
              "rationale": "Longitudinal clinical journeys and the broader platform scale support substantial history, but the announcement does not disclose the licensed subset's exact years of coverage."
            },
            "decision_outcome_richness": {
              "basis": "analysis",
              "source_ids": [
                "E1"
              ],
              "rationale": "The corpus includes longitudinal clinical journeys and imaging relevant to diagnosis, but the disclosure does not establish treatment-intervention and outcome linkage for canonical Human Response staging."
            },
            "decision_value_density": {
              "basis": "analysis",
              "source_ids": [
                "E1"
              ],
              "rationale": "Clinical and diagnostic model development operates in a high-consequence healthcare domain where the licensed data can influence valuable model-development decisions."
            },
            "real_world_grounding": {
              "basis": "disclosed_fact",
              "source_ids": [
                "E1"
              ],
              "rationale": "OneMedNet states the data are first-party, direct-from-source real-world clinical data drawn from a live network of more than 2,300 healthcare partner sites."
            },
            "domain_value": {
              "basis": "analysis",
              "source_ids": [
                "E1"
              ],
              "rationale": "Precision medicine, clinical diagnostics and life-sciences data products are economically valuable applications, supporting a maximum domain-value rating."
            },
            "refreshability": {
              "basis": "disclosed_fact",
              "source_ids": [
                "E1"
              ],
              "rationale": "The multiyear agreement, live provider network, expansion option and recurring downstream licenses establish a renewable data relationship rather than a one-time corpus."
            },
            "proprietary_advantage": {
              "basis": "disclosed_fact",
              "source_ids": [
                "E1"
              ],
              "rationale": "OneMedNet retains source-data rights while licensing a direct-from-source corpus that the customer will combine with proprietary molecular data into a derivative product."
            },
            "model_learning_usefulness": {
              "basis": "disclosed_fact",
              "source_ids": [
                "E1"
              ],
              "rationale": "The customer explicitly intends to use the licensed data to train AI-enabled clinical and diagnostic models."
            },
            "non_replicability": {
              "basis": "analysis",
              "source_ids": [
                "E1"
              ],
              "rationale": "Replicating a de-identified multimodal corpus across more than 2,300 direct healthcare sites would require substantial access, integration and compliance infrastructure."
            },
            "rights_usability": {
              "basis": "analysis",
              "source_ids": [
                "E1"
              ],
              "rationale": "Training and derivative commercialization are expressly permitted and downstream sublicensing economics are defined, although retention, exclusivity and model-ownership details remain undisclosed."
            }
          }
        },
        "transaction_signal": {
          "score": 78,
          "coverage_pct": 85,
          "rationale": "Strategic Signal analysis. The transaction discloses a seven-figure multiyear data license, explicit model-training use, cleanly separable data consideration, derivative commercialization and recurring sublicensing economics. The exact price, counterparty identity, competitive process and independent model-uplift evidence remain undisclosed.",
          "factor_ratings": {
            "disclosed_economics": 3,
            "clean_price_discovery": 2,
            "data_consideration_separability": 5,
            "explicit_model_use": 5,
            "rights_clarity": 4.5,
            "competing_bids": null,
            "strategic_buyer_quality": 3.5,
            "precedent_value": 5,
            "independent_model_uplift": null
          },
          "factor_evidence": {
            "disclosed_economics": {
              "basis": "disclosed_fact",
              "source_ids": [
                "E1"
              ],
              "rationale": "The issuer discloses a seven-figure multiyear data license plus recurring annual sublicensing revenue, but not an exact contract amount or payment schedule."
            },
            "clean_price_discovery": {
              "basis": "analysis",
              "source_ids": [
                "E1"
              ],
              "rationale": "The disclosure isolates the data-license economics from the derivative sublicensing stream, but gives only a seven-figure range and no competitive bidding evidence."
            },
            "data_consideration_separability": {
              "basis": "disclosed_fact",
              "source_ids": [
                "E1"
              ],
              "rationale": "The announced consideration is explicitly for the multiyear data license, with a separately described downstream sublicensing revenue stream."
            },
            "explicit_model_use": {
              "basis": "disclosed_fact",
              "source_ids": [
                "E1"
              ],
              "rationale": "The customer explicitly intends to train AI-enabled clinical and diagnostic models using the licensed OneMedNet data."
            },
            "rights_clarity": {
              "basis": "disclosed_fact",
              "source_ids": [
                "E1"
              ],
              "rationale": "The announcement clearly establishes training, derivative commercialization, retained source-data rights and downstream restrictions, while several secondary rights remain undisclosed."
            },
            "strategic_buyer_quality": {
              "basis": "analysis",
              "source_ids": [
                "E1"
              ],
              "rationale": "The counterparty is unnamed but described as a commercially scaled AI-enabled precision-medicine company with an established life-sciences customer base."
            },
            "precedent_value": {
              "basis": "analysis",
              "source_ids": [
                "E1"
              ],
              "rationale": "A multiyear training license plus recurring participation in derivative-data sublicensing provides a repeatable precedent for source-data owners monetizing downstream AI data products."
            }
          }
        }
      },
      "sources": [
        {
          "type": "primary_syndicated",
          "publisher": "OneMedNet",
          "title": "OneMedNet Signs Seven Figure Multiyear Agreement to Power a Partner Sold Derivative Data Solution",
          "url": "https://www.globenewswire.com/news-release/2026/09/30/3371902/0/en/onemednet-signs-seven-figure-multiyear-agreement-to-power-a-partner-sold-derivative-data-solution.html"
        }
      ],
      "date_added": "2026-09-30",
      "date_last_reviewed": "2026-09-30",
      "publication_review": {
        "status": "human_reviewed",
        "approval_channel": "owner_conversation",
        "reviewer": "Mike Ye",
        "reviewed_at": "2026-09-30T19:34:37.164Z",
        "decision_id": "conversation-20260930-1BE00DE10512",
        "packet_sha256": "ab15069afc2980e4e693ed1554c3ee7c9091af7fc4fa43e2021a8c482500df80",
        "record_sha256": "780e019ff810f2e9758e4fa33c443b9018cd250c1647c418e50dc08d66ce1c57"
      }
    }
  ]
}
