{
  "dataset": {
    "approved_record_count": 15,
    "author": "Mike Ye",
    "canonical_url": "https://strategicsignal.ai/ai-data-transactions/",
    "methodology_version": "1.0.0",
    "name": "Strategic Signal — Corporate AI Data Transactions",
    "notes": "15 approved observations; 5 legacy observations remain provisional and excluded from approved benchmarks. Filter publication_review.status=human_reviewed for approved longitudinal measures.",
    "provisional_record_count": 5,
    "published_on": "2026-09-15",
    "publisher": "Strategic Signal",
    "record_count": 20,
    "review_standard": "AI-discovered; human approval required. Legacy observations remain provisional until approved.",
    "updated_on": "2026-10-07",
    "version": "1.0.20"
  },
  "transactions": [
    {
      "access_structure": "Astex, Bristol Myers Squibb and Takeda joined the existing AbbVie and Johnson & Johnson initiative. Local computation permits model learning without pooling the underlying records.",
      "analysis_boundary": "March 27, 2025 founding is context, not a second transaction here. September 14, 2026 performance evidence is a later update. Uplift is consortium-reported, not independently replicated; no price is disclosed.",
      "announcement_date": "2025-10-01",
      "asset_disposition": "Embedded / Continuing",
      "buyer_ai_developer": [
        "AISB Federated OpenFold3 Initiative",
        "AlQuraishi Lab",
        "Apheris"
      ],
      "canonical_path": "/ai-data-transactions/deals/ss-aidt-2025-0001/",
      "confidence": "high",
      "corpus_description": "Proprietary experimentally determined protein–small molecule structures supplied for federated fine-tuning. This observation records the October 2025 consortium expansion.",
      "corpus_renewal": "Renewable Corpus",
      "data_category": [
        "Scientific / pharmaceutical",
        "Protein–small molecule structures"
      ],
      "date_added": "2026-09-15",
      "date_last_reviewed": "2026-09-16",
      "decision_outcome_richness": "Analysis: experimental structures directly constrain molecular predictions; they are not a complete history of clinical decisions or drug-development outcomes.",
      "disclosed_consideration": {
        "amount": null,
        "currency": null,
        "description": "No data-specific consideration disclosed.",
        "status": "not_disclosed"
      },
      "economics_scope": "not_disclosed",
      "estimated_economics": {
        "amount": null,
        "currency": null,
        "method": null,
        "status": "not_estimated"
      },
      "event_history": [
        {
          "count_as_new_transaction": false,
          "date": "2025-03-27",
          "event": "Founding initiative"
        },
        {
          "count_as_new_transaction": true,
          "date": "2025-10-01",
          "event": "Three new contributors join the existing initiative"
        },
        {
          "count_as_new_transaction": false,
          "date": "2026-09-14",
          "event": "Consortium-reported performance update"
        }
      ],
      "historical_depth": "not_disclosed",
      "industry": "Pharmaceuticals & biotechnology",
      "likely_model_use": "Fine-tune OpenFold3 for protein–ligand prediction.",
      "likely_strategic_value": "Analysis: multi-owner scientific archives can supply differentiated training examples while retaining source-data control.",
      "m_and_a_implications": "Analysis: rights to learn can be valuable without an asset sale. Reported predictive improvement is not a cash valuation or evidence of clinical benefit.",
      "model_uplift": {
        "baseline": "OpenFold3 Preview 2",
        "evaluation_n": 1056,
        "evidence_status": "consortium_reported_not_independently_replicated",
        "limitations": "Private held-out evaluation; consortium-funded work; clinical utility and independent replication not established.",
        "metrics": [
          {
            "baseline_pct": 35.6,
            "change_percentage_points": 16.5,
            "name": "fraction PL-lDDT >= 0.8",
            "updated_pct": 52.1
          },
          {
            "baseline_pct": 28.9,
            "change_percentage_points": 17.9,
            "name": "fraction ligand bisyRMSD <= 2 angstrom",
            "updated_pct": 46.8
          }
        ],
        "reported_on": "2026-09-14",
        "training_structures": 20167
      },
      "primary_transaction_structure": "Federated / Controlled Learning Rights",
      "refreshability": "continuing_network; contribution schedule not disclosed",
      "restrictions": {
        "de_identification": "not_disclosed",
        "employee_customer_data": "not_applicable",
        "geographic_sovereignty": "not_disclosed",
        "governed_access": "Local computation and contributor-defined access policies.",
        "privacy_pii": "Confidential scientific data; no patient-data entitlement inferred.",
        "trade_secret": "Raw structures stay under contributor control."
      },
      "rights": {
        "derived_model": "yes_model_improvement; current network describes member ownership, not full contract terms",
        "exclusivity": "not_disclosed",
        "inference_retrieval": "unknown",
        "ownership_transfer": "no_disclosed_transfer",
        "post_training": "yes_explicit_fine_tuning",
        "retention": "raw_data_remains_with_contributors; model retention terms not_disclosed",
        "sublicensing": "not_disclosed",
        "training": "yes_explicit"
      },
      "scores": {
        "asset": {
          "coverage_pct": 84,
          "factor_evidence": {
            "decision_outcome_richness": {
              "basis": "analysis",
              "rationale": "A structure records an experimental result, but does not establish downstream commercial or clinical outcomes.",
              "source_ids": [
                "E1"
              ]
            },
            "decision_value_density": {
              "basis": "analysis",
              "rationale": "Better molecular prioritization can influence expensive research choices; a qualitative domain judgment.",
              "source_ids": [
                "E1"
              ]
            },
            "domain_value": {
              "basis": "analysis",
              "rationale": "Drug-discovery research makes predictive accuracy economically consequential.",
              "source_ids": [
                "E1"
              ]
            },
            "model_learning_usefulness": {
              "basis": "analysis",
              "rationale": "The learning objective is closely matched to the described scientific observations.",
              "source_ids": [
                "E1"
              ]
            },
            "non_replicability": {
              "basis": "analysis",
              "rationale": "Reproducing experimental observations requires laboratory effort; cost is not quantified.",
              "source_ids": [
                "E1"
              ]
            },
            "proprietary_advantage": {
              "basis": "analysis",
              "rationale": "Combining otherwise separate private archives offers access unavailable from public corpora alone.",
              "source_ids": [
                "E1"
              ]
            },
            "real_world_grounding": {
              "basis": "analysis",
              "rationale": "The described observations originate in physical experiments rather than generated scenarios.",
              "source_ids": [
                "E1"
              ]
            },
            "rights_usability": {
              "basis": "analysis",
              "rationale": "Local learning is explicitly enabled, while undisclosed downstream terms limit certainty.",
              "source_ids": [
                "E1"
              ]
            },
            "uniqueness_scarcity": {
              "basis": "analysis",
              "rationale": "Nonpublic experimental observations are hard to substitute with generic text.",
              "source_ids": [
                "E1"
              ]
            }
          },
          "factor_ratings": {
            "decision_outcome_richness": 3.5,
            "decision_value_density": 4.5,
            "domain_value": 4.5,
            "historical_depth": null,
            "model_learning_usefulness": 4.5,
            "non_replicability": 4,
            "proprietary_advantage": 4.5,
            "real_world_grounding": 5,
            "refreshability": null,
            "rights_usability": 4,
            "uniqueness_scarcity": 4.5
          },
          "rationale": "Strong experimental grounding and learning fit. Structural results are not full decision histories. Unknown archive age and contribution cadence remain unscored.",
          "score": 87
        },
        "methodology_version": "1.0.0",
        "transaction_signal": {
          "coverage_pct": 85,
          "factor_evidence": {
            "clean_price_discovery": {
              "basis": "analysis",
              "rationale": "No observable price or auction is provided by this announcement.",
              "source_ids": [
                "E1"
              ]
            },
            "data_consideration_separability": {
              "basis": "analysis",
              "rationale": "No separately quantified data consideration is available.",
              "source_ids": [
                "E1"
              ]
            },
            "disclosed_economics": {
              "basis": "analysis",
              "rationale": "The reviewed announcement supplies no monetary amount usable for valuation.",
              "source_ids": [
                "E1"
              ]
            },
            "explicit_model_use": {
              "basis": "analysis",
              "rationale": "Fine-tuning is the stated purpose, not an inference from product deployment.",
              "source_ids": [
                "E1"
              ]
            },
            "precedent_value": {
              "basis": "analysis",
              "rationale": "A multi-owner structure offers a repeatable pattern for otherwise inaccessible scientific corpora.",
              "source_ids": [
                "E1"
              ]
            },
            "rights_clarity": {
              "basis": "analysis",
              "rationale": "The permitted learning method is clear; the full contractual allocation remains unknown.",
              "source_ids": [
                "E1"
              ]
            },
            "strategic_buyer_quality": {
              "basis": "analysis",
              "rationale": "Established research participants support execution relevance, not evidence of a market-clearing price.",
              "source_ids": [
                "E1"
              ]
            }
          },
          "factor_ratings": {
            "clean_price_discovery": 0,
            "competing_bids": null,
            "data_consideration_separability": 0,
            "disclosed_economics": 0,
            "explicit_model_use": 5,
            "independent_model_uplift": null,
            "precedent_value": 4.5,
            "rights_clarity": 4,
            "strategic_buyer_quality": 4
          },
          "rationale": "Clear learning arrangement, weak price discovery. Consortium-reported uplift is recorded separately and earns no independent-replication points.",
          "score": 40
        }
      },
      "seller_data_owner": [
        "AbbVie",
        "Johnson & Johnson",
        "Bristol Myers Squibb",
        "Takeda",
        "Astex Pharmaceuticals"
      ],
      "slug": "openfold-pharma-federated-training-consortium",
      "source_code_software_asset": "no",
      "sources": [
        {
          "publisher": "Apheris",
          "title": "AISB initiative expansion — October 1, 2025",
          "type": "primary",
          "url": "https://www.apheris.com/resources/aisb-network-expands-federated-openfold3-initiative-with-three-new-pharma-contributors"
        },
        {
          "publisher": "Apheris",
          "title": "Founding initiative — March 27, 2025",
          "type": "primary",
          "url": "https://www.apheris.com/resources/alquraishi-lab-s-openfold3-to-be-fine-tuned-with-pharma-industry-data-in-a-secure-ai-collaboration"
        },
        {
          "publisher": "Apheris",
          "title": "Consortium performance report — September 14, 2026",
          "type": "primary",
          "url": "https://www.apheris.com/resources/federated-training-dramatically-improves-the-accuracy-of-protein-ligand-co-folding-on-private-pharma-structures"
        },
        {
          "publisher": "Apheris",
          "title": "AISB network governance — current description",
          "type": "primary",
          "url": "https://www.apheris.com/networks/aisb"
        }
      ],
      "transaction_id": "SS-AIDT-2025-0001",
      "transaction_status": "active",
      "publication_review": {
        "status": "human_reviewed",
        "approval_channel": "owner_conversation",
        "reviewer": "Mike Ye",
        "reviewed_at": "2026-09-16T21:38:20.361Z",
        "decision_id": "conversation-20260916-81BF2CAD028B",
        "packet_sha256": "4e1f58d72f53b3ad5140e1c8b01fbb7347a8af99dc3f26fc29a5d28f52422173",
        "record_sha256": "0e4eef8e3f50a13a1e520e21f1545515918d9e29722134c414ceab9f6c02dc16"
      }
    },
    {
      "access_structure": "Reported transfer of a finite archive; detailed contract terms and buyer identity were not disclosed.",
      "analysis_boundary": "The corpus and consideration description are attributed to the former CEO through reporting. Buyer, contract rights, and privacy controls remain unknown.",
      "announcement_date": "2026-04-16",
      "asset_disposition": "Stranded / Separable",
      "buyer_ai_developer": [
        "Undisclosed AI buyer"
      ],
      "canonical_path": "/ai-data-transactions/deals/ss-aidt-2026-0001/",
      "confidence": "medium",
      "corpus_description": "Thirteen years of the shuttered company's Slack messages, internal email, and Jira tickets, reported sold as AI training data.",
      "corpus_renewal": "Finite Corpus",
      "data_category": [
        "Internal communications",
        "Product / engineering workflow",
        "Institutional memory"
      ],
      "date_added": "2026-09-15",
      "date_last_reviewed": "2026-09-15",
      "decision_outcome_richness": "Moderate-to-high: communications and tickets can connect plans, execution, and product outcomes, but outcome labeling is not disclosed.",
      "disclosed_consideration": {
        "amount": null,
        "currency": "USD",
        "description": "Seller described proceeds as hundreds of thousands of dollars; no exact figure was disclosed.",
        "status": "reported_range_only"
      },
      "economics_scope": "reported_dataset_specific",
      "estimated_economics": {
        "amount": null,
        "currency": null,
        "method": null,
        "status": "not_estimated"
      },
      "historical_depth": "13 years (reported)",
      "industry": "Media technology",
      "likely_model_use": "Training AI systems to navigate realistic workplace communication, engineering, and project-management tasks.",
      "likely_strategic_value": "A coherent company history provides temporal and organizational context that synthetic office tasks lack.",
      "m_and_a_implications": "Shows that a failed company's collaboration exhaust may remain separately monetizable after operating value disappears.",
      "primary_transaction_structure": "Data Asset Acquisition",
      "publication_review": {
        "audited_on": "2026-09-16",
        "human_approval_recorded": false,
        "notice": "Sale reporting is supported by indexed Forbes excerpts, but full text and an independent corroborating source remain to be captured. Detailed contract rights are not verified.",
        "status": "legacy_review_required"
      },
      "refreshability": "none_company_shuttered",
      "restrictions": {
        "de_identification": "not_disclosed",
        "employee_customer_data": "not_disclosed",
        "geographic_sovereignty": "not_disclosed",
        "governed_access": "not_disclosed",
        "privacy_pii": "not_disclosed",
        "trade_secret": "not_disclosed"
      },
      "rights": {
        "derived_model": "not_disclosed",
        "exclusivity": "not_disclosed",
        "inference_retrieval": "not_disclosed",
        "ownership_transfer": "unknown",
        "post_training": "not_disclosed",
        "retention": "not_disclosed",
        "sublicensing": "not_disclosed",
        "training": "yes_reported"
      },
      "scores": {
        "asset": {
          "coverage_pct": 0,
          "factor_ratings": {
            "decision_outcome_richness": null,
            "decision_value_density": null,
            "domain_value": null,
            "historical_depth": null,
            "model_learning_usefulness": null,
            "non_replicability": null,
            "proprietary_advantage": null,
            "real_world_grounding": null,
            "refreshability": null,
            "rights_usability": null,
            "uniqueness_scarcity": null
          },
          "rationale": "Legacy score withdrawn pending evidence-linked scoring and human approval. The superseded v1.0.0 release retains the original values.",
          "score": null
        },
        "methodology_version": "1.0.0",
        "transaction_signal": {
          "coverage_pct": 0,
          "factor_ratings": {
            "clean_price_discovery": null,
            "competing_bids": null,
            "data_consideration_separability": null,
            "disclosed_economics": null,
            "explicit_model_use": null,
            "independent_model_uplift": null,
            "precedent_value": null,
            "rights_clarity": null,
            "strategic_buyer_quality": null
          },
          "rationale": "Legacy score withdrawn pending evidence-linked scoring and human approval. The superseded v1.0.0 release retains the original values.",
          "score": null
        }
      },
      "seller_data_owner": [
        "Cielo24"
      ],
      "slug": "cielo24-institutional-memory-sale",
      "source_code_software_asset": "unknown",
      "sources": [
        {
          "publisher": "Forbes",
          "title": "AI's New Training Data: Your Old Work Slacks And Emails",
          "type": "secondary",
          "url": "https://www.forbes.com/sites/annatong/2026/04/16/ais-new-training-data-your-old-work-slacks-and-emails/"
        },
        {
          "publisher": "Fast Company",
          "title": "Shuttered startups are selling old Slack chats and emails to AI companies",
          "type": "secondary",
          "url": "https://www.fastcompany.com/91528808/shuttered-startups-are-selling-old-slack-chats-and-emails-to-ai-companies"
        }
      ],
      "transaction_id": "SS-AIDT-2026-0001",
      "transaction_status": "reported_completed"
    },
    {
      "access_structure": "Bankruptcy asset auction. Google was selected at $10 million; Mercor was alternate at $7.5 million. micro1 later noticed a $12.5 million competing bid. No sale order had been entered as of the review date.",
      "analysis_boundary": "Bid amounts, selected/alternate bidders, asset scope, and pending status are sourced facts. Model-use and valuation implications beyond disclosed purpose are Strategic Signal analysis.",
      "announcement_date": "2026-08-14",
      "asset_disposition": "Stranded / Separable",
      "buyer_ai_developer": [
        "Google LLC (selected bidder; challenged)"
      ],
      "canonical_path": "/ai-data-transactions/deals/ss-aidt-2026-0002/",
      "confidence": "high",
      "corpus_description": "A deidentified archive spanning roughly 34 years, reported to include about 100 million emails, 500 million Teams messages, 30 million lines of source code, operational data, documents, and internally developed software. Customer databases are outside scope.",
      "corpus_renewal": "Finite Corpus",
      "data_category": [
        "Internal communications",
        "Source code and software",
        "Operational records",
        "Financial and workforce records"
      ],
      "date_added": "2026-09-15",
      "date_last_reviewed": "2026-09-15",
      "decision_outcome_richness": "High: communications, code, operations, scheduling, finance, and maintenance-adjacent workflows create a dense cross-functional operating history.",
      "disclosed_consideration": {
        "amount": 10000000,
        "currency": "USD",
        "description": "Google selected bid; court approval pending. Later micro1 notice proposed $12.5 million.",
        "status": "disclosed_bid_not_closed"
      },
      "economics_scope": "dataset_and_related_internal_software_specific",
      "estimated_economics": {
        "amount": null,
        "currency": null,
        "method": null,
        "status": "not_estimated"
      },
      "historical_depth": "Approximately 34 years",
      "industry": "Aviation",
      "likely_model_use": "Train and improve AI systems and productivity products on real enterprise work, software, communication, and operational sequences.",
      "likely_strategic_value": "Large, temporally coherent, multi-modal corporate history with explicit training use and measurable auction demand.",
      "m_and_a_implications": "Provides unusually clean evidence that a defunct company's data and institutional memory can be auctioned separately from its operating business, while exposing privacy and chain-of-title diligence as value-critical.",
      "primary_transaction_structure": "Data Asset Acquisition",
      "publication_review": {
        "audited_on": "2026-09-16",
        "human_approval_recorded": false,
        "notice": "Stretto reports an auction and pending approval through September 11. Treat this as secondary reporting. Underlying filing and current status remain to be verified.",
        "status": "legacy_review_required"
      },
      "refreshability": "none_airline_ceased_operations",
      "restrictions": {
        "de_identification": "required_before_transfer",
        "employee_customer_data": "Customer datasets excluded; treatment of employee and labor data remains contested.",
        "geographic_sovereignty": "not_disclosed_for_google_bid",
        "governed_access": "Independent third-party deidentification contemplated before delivery; final restrictions remain subject to court approval.",
        "privacy_pii": "Customer databases excluded; incidental consumer data to be deidentified. Employee-data objections remain unresolved.",
        "trade_secret": "Contract-counterparty and labor objections remain on file."
      },
      "rights": {
        "derived_model": "yes_intended_use_if_sale_closes",
        "exclusivity": "asset_sale_subject_to_sale_order",
        "inference_retrieval": "not_disclosed",
        "ownership_transfer": "yes_if_sale_closes",
        "post_training": "yes_general_model_improvement",
        "retention": "not_final_pending_sale_order",
        "sublicensing": "not_final_pending_sale_order",
        "training": "yes_explicit"
      },
      "scores": {
        "asset": {
          "coverage_pct": 0,
          "factor_ratings": {
            "decision_outcome_richness": null,
            "decision_value_density": null,
            "domain_value": null,
            "historical_depth": null,
            "model_learning_usefulness": null,
            "non_replicability": null,
            "proprietary_advantage": null,
            "real_world_grounding": null,
            "refreshability": null,
            "rights_usability": null,
            "uniqueness_scarcity": null
          },
          "rationale": "Legacy score withdrawn pending evidence-linked scoring and human approval. The superseded v1.0.0 release retains the original values.",
          "score": null
        },
        "methodology_version": "1.0.0",
        "transaction_signal": {
          "coverage_pct": 0,
          "factor_ratings": {
            "clean_price_discovery": null,
            "competing_bids": null,
            "data_consideration_separability": null,
            "disclosed_economics": null,
            "explicit_model_use": null,
            "independent_model_uplift": null,
            "precedent_value": null,
            "rights_clarity": null,
            "strategic_buyer_quality": null
          },
          "rationale": "Legacy score withdrawn pending evidence-linked scoring and human approval. The superseded v1.0.0 release retains the original values.",
          "score": null
        }
      },
      "seller_data_owner": [
        "Spirit Aviation Holdings / Spirit Airlines"
      ],
      "slug": "spirit-airlines-data-auction-google",
      "source_code_software_asset": "yes",
      "sources": [
        {
          "publisher": "Research Suite by Stretto",
          "title": "Spirit deidentified-data auction results and docket chronology",
          "type": "secondary",
          "url": "https://chapter11cases.com/blogs/news/the-spirit-airlines-deidentified-data-sale-auction-results-objections-and-the-september-30-hearing"
        },
        {
          "publisher": "Reuters",
          "title": "Google to buy Spirit Airlines business data for $10 million",
          "type": "secondary",
          "url": "https://www.reuters.com/legal/litigation/google-buy-spirit-airlines-business-data-10-million-2026-08-17/"
        },
        {
          "publisher": "WIRED",
          "title": "Spirit Airlines Wants to Sell Its Data to Google",
          "type": "secondary",
          "url": "https://www.wired.com/story/spirit-airlines-wants-to-sell-its-data-to-google-former-flight-attendants-are-freaked-out"
        }
      ],
      "transaction_id": "SS-AIDT-2026-0002",
      "transaction_status": "pending_court_approval"
    },
    {
      "access_structure": "Collaborative physical-AI development; detailed data-access boundaries and model-rights allocation were not disclosed.",
      "analysis_boundary": "The announced combination of operational data and robot foundation models is fact. Training, retention, exclusivity, and derived-model rights are unknown.",
      "announcement_date": "2026-09-02",
      "asset_disposition": "Embedded / Continuing",
      "buyer_ai_developer": [
        "FieldAI"
      ],
      "canonical_path": "/ai-data-transactions/deals/ss-aidt-2026-0003/",
      "confidence": "medium_high",
      "corpus_description": "Caterpillar operational data, engineering capability, and industry expertise combined with FieldAI robot foundation models for complex jobsites and manufacturing environments.",
      "corpus_renewal": "Renewable Corpus",
      "data_category": [
        "Industrial operational data",
        "Sensor and robotics data",
        "Engineering knowledge"
      ],
      "date_added": "2026-09-15",
      "date_last_reviewed": "2026-09-15",
      "decision_outcome_richness": "High: autonomous inspections and operations can generate state–action–outcome data in safety-critical environments.",
      "disclosed_consideration": {
        "amount": null,
        "currency": null,
        "description": "No consideration disclosed.",
        "status": "not_disclosed"
      },
      "economics_scope": "not_disclosed",
      "estimated_economics": {
        "amount": null,
        "currency": null,
        "method": null,
        "status": "not_estimated"
      },
      "historical_depth": "not_disclosed",
      "industry": "Industrial equipment & construction",
      "likely_model_use": "Improve robotic autonomy, inspection, situational awareness, and digital-twin systems for industrial environments.",
      "likely_strategic_value": "Connects foundation-model capability to hard-to-reproduce industrial environments and continuous field feedback.",
      "m_and_a_implications": "Industrial incumbents may structure data-bearing co-development instead of selling operating histories, preserving control while giving AI partners domain access.",
      "primary_transaction_structure": "Strategic Model Co-Development",
      "publication_review": {
        "audited_on": "2026-09-16",
        "human_approval_recorded": false,
        "notice": "Caterpillar explicitly mentions operational data and FieldAI foundation models; separately material learning rights are not disclosed. Qualification remains unresolved.",
        "status": "legacy_review_required"
      },
      "refreshability": "live_and_recurring_operational_data_likely",
      "restrictions": {
        "de_identification": "not_disclosed",
        "employee_customer_data": "not_disclosed",
        "geographic_sovereignty": "not_disclosed",
        "governed_access": "not_disclosed",
        "privacy_pii": "not_disclosed",
        "trade_secret": "Engineering and operational data are described but controls are not disclosed."
      },
      "rights": {
        "derived_model": "unknown",
        "exclusivity": "not_disclosed",
        "inference_retrieval": "yes_operational_application",
        "ownership_transfer": "no_disclosed_transfer",
        "post_training": "unknown",
        "retention": "not_disclosed",
        "sublicensing": "not_disclosed",
        "training": "unknown"
      },
      "scores": {
        "asset": {
          "coverage_pct": 0,
          "factor_ratings": {
            "decision_outcome_richness": null,
            "decision_value_density": null,
            "domain_value": null,
            "historical_depth": null,
            "model_learning_usefulness": null,
            "non_replicability": null,
            "proprietary_advantage": null,
            "real_world_grounding": null,
            "refreshability": null,
            "rights_usability": null,
            "uniqueness_scarcity": null
          },
          "rationale": "Legacy score withdrawn pending evidence-linked scoring and human approval. The superseded v1.0.0 release retains the original values.",
          "score": null
        },
        "methodology_version": "1.0.0",
        "transaction_signal": {
          "coverage_pct": 0,
          "factor_ratings": {
            "clean_price_discovery": null,
            "competing_bids": null,
            "data_consideration_separability": null,
            "disclosed_economics": null,
            "explicit_model_use": null,
            "independent_model_uplift": null,
            "precedent_value": null,
            "rights_clarity": null,
            "strategic_buyer_quality": null
          },
          "rationale": "Legacy score withdrawn pending evidence-linked scoring and human approval. The superseded v1.0.0 release retains the original values.",
          "score": null
        }
      },
      "seller_data_owner": [
        "Caterpillar"
      ],
      "slug": "caterpillar-fieldai-industrial-autonomy",
      "source_code_software_asset": "unknown",
      "sources": [
        {
          "publisher": "Caterpillar",
          "title": "Caterpillar and FieldAI Advance AI-Powered Industrial Innovation",
          "type": "primary",
          "url": "https://www.caterpillar.com/en/news/corporate-press-releases/h/caterpillar-and-fieldai-advance-ai-powered-industrial-innovation.html"
        },
        {
          "publisher": "PR Newswire",
          "title": "Caterpillar and FieldAI Advance AI-Powered Industrial Innovation",
          "type": "primary_syndicated",
          "url": "https://www.prnewswire.com/news-releases/caterpillar-and-fieldai-advance-ai-powered-industrial-innovation-302866862.html"
        }
      ],
      "transaction_id": "SS-AIDT-2026-0003",
      "transaction_status": "announced_active"
    },
    {
      "access_structure": "A 33-organization consortium is developing cybersecurity foundation models using member operational and security data. Custody and member-level licensing terms are not disclosed.",
      "analysis_boundary": "The 830 TB planned corpus, contributors and training purpose are company disclosures. Scores are qualitative Strategic Signal analysis, not measured model performance. Federated topology and contract terms are unknown.",
      "announcement_date": "2026-09-03",
      "asset_disposition": "Embedded / Continuing",
      "buyer_ai_developer": [
        "NAVER Cloud consortium",
        "NAVER Cloud",
        "LG AI Research"
      ],
      "canonical_path": "/ai-data-transactions/deals/ss-aidt-2026-0004/",
      "confidence": "high",
      "corpus_description": "Approximately 830 TB of high-quality real-world operational and security data from consortium members across power, finance, telecommunications, science, semiconductors, defense, and aerospace.",
      "corpus_renewal": "Renewable Corpus",
      "data_category": [
        "Cybersecurity records",
        "Critical-infrastructure operations",
        "Industrial systems data"
      ],
      "date_added": "2026-09-15",
      "date_last_reviewed": "2026-09-16",
      "decision_outcome_richness": "Strategic Signal analysis: security operations and planned field trials may connect events to responses; intervention/outcome labels are not disclosed.",
      "disclosed_consideration": {
        "amount": null,
        "currency": null,
        "description": "Government GPU support is disclosed, alongside consortium computing resources. Dataset-specific cash consideration is not disclosed.",
        "status": "in_kind_and_government_compute"
      },
      "economics_scope": "bundled_program_support",
      "estimated_economics": {
        "amount": null,
        "currency": null,
        "method": null,
        "status": "not_estimated"
      },
      "historical_depth": "not_disclosed",
      "industry": "Cybersecurity & critical infrastructure",
      "likely_model_use": "Train defensive and offensive cybersecurity foundation models and validate them in seven critical-industry domains.",
      "likely_strategic_value": "Pools otherwise unobtainable cross-industry operational security data at sovereign scale.",
      "m_and_a_implications": "Consortium structures can aggregate embedded national-infrastructure data without a conventional acquisition, making governance and model rights central valuation terms.",
      "primary_transaction_structure": "Strategic Model Co-Development",
      "publication_review": {
        "approval_channel": "owner_conversation",
        "decision_id": "conversation-20260916-2D6377E44A57",
        "packet_sha256": "0d3511e1d7c9a9ce6cd876440fe1b414e967551ce11a7404834f16cf82849ba7",
        "record_sha256": "9e66511ee6acde56b1620a2e82c7786e746dd09abcef983c751844558ad77bff",
        "reviewed_at": "2026-09-16T21:17:28.097Z",
        "reviewer": "Mike Ye",
        "status": "human_reviewed"
      },
      "refreshability": "consortium_operational_sources_likely_renewable",
      "restrictions": {
        "de_identification": "not_disclosed",
        "employee_customer_data": "not_disclosed",
        "geographic_sovereignty": "Korean program; contractual geographic restrictions not_disclosed",
        "governed_access": "Consortium/government program; detailed member-level controls are not disclosed.",
        "privacy_pii": "not_disclosed",
        "trade_secret": "Operational security and critical-infrastructure data; controls not disclosed."
      },
      "rights": {
        "derived_model": "planned_open_source_models_commercial_use_terms_not_disclosed",
        "exclusivity": "not_disclosed",
        "inference_retrieval": "yes_field_demonstrations",
        "ownership_transfer": "no_disclosed_transfer",
        "post_training": "unknown",
        "retention": "not_disclosed",
        "sublicensing": "not_disclosed",
        "training": "yes_explicit"
      },
      "scores": {
        "asset": {
          "coverage_pct": 88,
          "factor_evidence": {
            "decision_outcome_richness": {
              "basis": "analysis",
              "rationale": "Analysis: operational context suggests useful decision relationships; missing outcome-label evidence limits the rating.",
              "source_ids": [
                "E1"
              ]
            },
            "decision_value_density": {
              "basis": "analysis",
              "rationale": "Analysis: mistakes in this domain can have major economic consequences. This is domain judgment, not a disclosed corpus valuation.",
              "source_ids": [
                "E1"
              ]
            },
            "domain_value": {
              "basis": "analysis",
              "rationale": "Analysis: the disclosed domain supports economically consequential enterprise activities; this is not a model-uplift result.",
              "source_ids": [
                "E1"
              ]
            },
            "model_learning_usefulness": {
              "basis": "analysis",
              "rationale": "Analysis: the announcement directly links this corpus to model development; improvement magnitude is unmeasured.",
              "source_ids": [
                "E1"
              ]
            },
            "non_replicability": {
              "basis": "analysis",
              "rationale": "Analysis: reproducing owner-specific operating history would require comparable activity and time.",
              "source_ids": [
                "E1"
              ]
            },
            "proprietary_advantage": {
              "basis": "analysis",
              "rationale": "Analysis: member-owned operational or scientific data offers context beyond ordinary public web corpora.",
              "source_ids": [
                "E1"
              ]
            },
            "real_world_grounding": {
              "basis": "analysis",
              "rationale": "Analysis: the described corporate operational or scientific records are grounded in real activity rather than synthetic-only tasks.",
              "source_ids": [
                "E1"
              ]
            },
            "refreshability": {
              "basis": "analysis",
              "rationale": "Analysis: planned field trials and model refinement indicate recurring operational input; no perpetual refresh entitlement is claimed.",
              "source_ids": [
                "E1"
              ]
            },
            "uniqueness_scarcity": {
              "basis": "analysis",
              "rationale": "Analysis: access to this owner-contributed domain corpus is difficult to reproduce; exclusivity is not established.",
              "source_ids": [
                "E1"
              ]
            }
          },
          "factor_ratings": {
            "decision_outcome_richness": 3,
            "decision_value_density": 4,
            "domain_value": 4.5,
            "historical_depth": null,
            "model_learning_usefulness": 4.5,
            "non_replicability": 4,
            "proprietary_advantage": 4.5,
            "real_world_grounding": 4.5,
            "refreshability": 4,
            "rights_usability": null,
            "uniqueness_scarcity": 4
          },
          "rationale": "Qualitative Strategic Signal assessment based on the disclosed corpus and arrangement. Unknown factors remain null; no measured model uplift is claimed.",
          "score": 81
        },
        "methodology_version": "1.0.0",
        "transaction_signal": {
          "coverage_pct": 85,
          "factor_evidence": {
            "clean_price_discovery": {
              "basis": "analysis",
              "rationale": "Analysis: no separately priced corporate dataset is identified; the announcement offers little clean data-price discovery.",
              "source_ids": [
                "E1"
              ]
            },
            "data_consideration_separability": {
              "basis": "analysis",
              "rationale": "No separately quantified consideration for data is disclosed. Zero measures disclosure weakness, not zero data value.",
              "source_ids": [
                "E1"
              ]
            },
            "disclosed_economics": {
              "basis": "analysis",
              "rationale": "No dataset-specific cash amount is disclosed. This rating measures public price disclosure, not whether payment exists.",
              "source_ids": [
                "E1"
              ]
            },
            "explicit_model_use": {
              "basis": "analysis",
              "rationale": "The company explicitly connects proprietary corporate data to developing the named AI models.",
              "source_ids": [
                "E1"
              ]
            },
            "precedent_value": {
              "basis": "analysis",
              "rationale": "Analysis: this is a reusable model-co-development structure, but repeatable standalone pricing is not established.",
              "source_ids": [
                "E1"
              ]
            },
            "rights_clarity": {
              "basis": "analysis",
              "rationale": "Analysis: the announced model-development purpose is clear, while detailed transfer, retention and licensing terms remain unknown.",
              "source_ids": [
                "E1"
              ]
            },
            "strategic_buyer_quality": {
              "basis": "analysis",
              "rationale": "Analysis: an identified model developer with a specific domain program makes the observation strategically relevant.",
              "source_ids": [
                "E1"
              ]
            }
          },
          "factor_ratings": {
            "clean_price_discovery": 0,
            "competing_bids": null,
            "data_consideration_separability": 0,
            "disclosed_economics": 0,
            "explicit_model_use": 5,
            "independent_model_uplift": null,
            "precedent_value": 4,
            "rights_clarity": 2.5,
            "strategic_buyer_quality": 4
          },
          "rationale": "Qualitative Strategic Signal assessment based on the disclosed corpus and arrangement. Unknown factors remain null; no measured model uplift is claimed.",
          "score": 35
        }
      },
      "seller_data_owner": [
        "NAVER Cloud",
        "LG CNS",
        "KEPCO KDN",
        "Korea Hydro & Nuclear Power",
        "Financial Security Institute",
        "KISTI",
        "LG Uplus",
        "Other consortium members"
      ],
      "slug": "naver-cloud-cybersecurity-consortium",
      "source_code_software_asset": "unknown",
      "sources": [
        {
          "publisher": "NAVER",
          "title": "NAVER Cloud Consortium to Develop AI Models Tailored for Cybersecurity",
          "type": "primary",
          "url": "https://navercorp.com/en/media/pressReleasesDetail?seq=10034653"
        }
      ],
      "transaction_id": "SS-AIDT-2026-0004",
      "transaction_status": "announced_development"
    },
    {
      "access_structure": "Customized on-premise model development; Samsung says sensitive technology and operational data remain entirely within its boundaries. Samsung also participated in Mistral's financing round.",
      "analysis_boundary": "On-premise controls, target use cases, and strategic equity relationship are disclosed. Model ownership, exclusivity, sublicensing, and data-specific consideration are not.",
      "announcement_date": "2026-09-09",
      "asset_disposition": "Embedded / Continuing",
      "buyer_ai_developer": [
        "Mistral AI"
      ],
      "canonical_path": "/ai-data-transactions/deals/ss-aidt-2026-0005/",
      "confidence": "high",
      "corpus_description": "Sensitive semiconductor operational and manufacturing data used inside Samsung boundaries to customize on-premise models for defect detection, equipment optimization, and engineering workflows.",
      "corpus_renewal": "Renewable Corpus",
      "data_category": [
        "Manufacturing history",
        "Equipment and process data",
        "Engineering knowledge"
      ],
      "date_added": "2026-09-15",
      "date_last_reviewed": "2026-09-15",
      "decision_outcome_richness": "Exceptional: process interventions, yield, defect, and equipment outcomes can encode extremely high-value manufacturing decisions.",
      "disclosed_consideration": {
        "amount": null,
        "currency": null,
        "description": "Partnership economics and Samsung's individual equity investment were not disclosed separately.",
        "status": "not_disclosed"
      },
      "economics_scope": "bundled_with_strategic_equity_relationship",
      "estimated_economics": {
        "amount": null,
        "currency": null,
        "method": null,
        "status": "not_estimated"
      },
      "historical_depth": "not_disclosed",
      "industry": "Semiconductors",
      "likely_model_use": "Customize models for semiconductor defect detection, equipment optimization, technical support, and engineering decision workflows.",
      "likely_strategic_value": "Pairs a frontier-model developer with one of the world's most valuable proprietary manufacturing feedback loops while preserving on-premise control.",
      "m_and_a_implications": "Equity-linked, on-premise arrangements may become a preferred structure where the corpus is too strategic to sell or externalize.",
      "primary_transaction_structure": "Proprietary Data + Equity Partnership",
      "publication_review": {
        "audited_on": "2026-09-16",
        "human_approval_recorded": false,
        "notice": "Samsung announces customized on-premise models and a strategic equity stake. Processing data within Samsung boundaries does not establish what learning rights Mistral receives.",
        "status": "legacy_review_required"
      },
      "refreshability": "continuous_manufacturing_operations",
      "restrictions": {
        "de_identification": "not_disclosed",
        "employee_customer_data": "not_disclosed",
        "geographic_sovereignty": "On-premise / sovereign control",
        "governed_access": "On-premise deployment; sensitive data stays inside Samsung boundaries.",
        "privacy_pii": "not_disclosed",
        "trade_secret": "Explicit protection of sensitive semiconductor technology and operational data."
      },
      "rights": {
        "derived_model": "unknown",
        "exclusivity": "not_disclosed",
        "inference_retrieval": "yes_on_prem_use",
        "ownership_transfer": "no_disclosed_transfer",
        "post_training": "yes_customization",
        "retention": "restricted_to_samsung_boundaries",
        "sublicensing": "not_disclosed",
        "training": "yes_customized_on_prem_models"
      },
      "scores": {
        "asset": {
          "coverage_pct": 0,
          "factor_ratings": {
            "decision_outcome_richness": null,
            "decision_value_density": null,
            "domain_value": null,
            "historical_depth": null,
            "model_learning_usefulness": null,
            "non_replicability": null,
            "proprietary_advantage": null,
            "real_world_grounding": null,
            "refreshability": null,
            "rights_usability": null,
            "uniqueness_scarcity": null
          },
          "rationale": "Legacy score withdrawn pending evidence-linked scoring and human approval. The superseded v1.0.0 release retains the original values.",
          "score": null
        },
        "methodology_version": "1.0.0",
        "transaction_signal": {
          "coverage_pct": 0,
          "factor_ratings": {
            "clean_price_discovery": null,
            "competing_bids": null,
            "data_consideration_separability": null,
            "disclosed_economics": null,
            "explicit_model_use": null,
            "independent_model_uplift": null,
            "precedent_value": null,
            "rights_clarity": null,
            "strategic_buyer_quality": null
          },
          "rationale": "Legacy score withdrawn pending evidence-linked scoring and human approval. The superseded v1.0.0 release retains the original values.",
          "score": null
        }
      },
      "seller_data_owner": [
        "Samsung Electronics"
      ],
      "slug": "samsung-mistral-semiconductor-partnership",
      "source_code_software_asset": "unknown",
      "sources": [
        {
          "publisher": "Samsung",
          "title": "Samsung and Mistral AI Announce Strategic Partnership for Intelligence-Driven Semiconductor Infrastructure",
          "type": "primary",
          "url": "https://news.samsung.com/global/samsung-and-mistral-ai-announce-strategic-partnership-for-intelligence-driven-semiconductor-infrastructure"
        },
        {
          "publisher": "Reuters",
          "title": "French AI company Mistral hits $24 billion valuation in funding round",
          "type": "secondary",
          "url": "https://www.reuters.com/world/europe/french-ai-company-mistral-hits-24-billion-valuation-funding-round-2026-09-08/"
        }
      ],
      "transaction_id": "SS-AIDT-2026-0005",
      "transaction_status": "announced_active"
    },
    {
      "access_structure": "Built-in licensed/indexed data available to the financial-services product for grounded analysis and citations; specific provider contracts are not public.",
      "analysis_boundary": "Built-in providers and retrieval/citation use are disclosed. Training, post-training, ownership, and pricing are not claimed.",
      "announcement_date": "2026-09-10",
      "asset_disposition": "Embedded / Continuing",
      "buyer_ai_developer": [
        "OpenAI"
      ],
      "canonical_path": "/ai-data-transactions/deals/ss-aidt-2026-0006/",
      "confidence": "medium_high",
      "corpus_description": "Continuously updated proprietary financial datasets integrated as built-in, cited data sources in ChatGPT for Financial Services. Customer-connected subscription sources are excluded from this record.",
      "corpus_renewal": "Renewable Corpus",
      "data_category": [
        "Financial datasets",
        "Private-market data",
        "Fundamental company data",
        "Earnings and filings"
      ],
      "date_added": "2026-09-15",
      "date_last_reviewed": "2026-09-15",
      "decision_outcome_richness": "High for financial research and valuation, though less direct than operational intervention corpora.",
      "disclosed_consideration": {
        "amount": null,
        "currency": null,
        "description": "No provider-specific consideration disclosed.",
        "status": "not_disclosed"
      },
      "economics_scope": "not_disclosed",
      "estimated_economics": {
        "amount": null,
        "currency": null,
        "method": null,
        "status": "not_estimated"
      },
      "historical_depth": "Varies by provider; Daloopa reports 14 years of normalized data.",
      "industry": "Financial information services",
      "likely_model_use": "Ground inference, financial analysis, retrieval, and citations; no training rights are claimed.",
      "likely_strategic_value": "Authoritative licensed data reduces hallucination and makes a general model useful in high-value, source-sensitive financial workflows.",
      "m_and_a_implications": "Shows that renewable data vendors can license access repeatedly without transferring ownership, but undisclosed contracts provide weak standalone price discovery.",
      "primary_transaction_structure": "Proprietary Corpus License",
      "publication_review": {
        "audited_on": "2026-09-16",
        "human_approval_recorded": false,
        "notice": "The cited pages could not be retrieved for this audit. Source-specific rights and the boundary between built-in access and customer connectors remain unresolved.",
        "status": "legacy_review_required"
      },
      "refreshability": "continuous_provider_updates",
      "restrictions": {
        "de_identification": "not_applicable_or_not_disclosed",
        "employee_customer_data": "Customer-connected datasets are explicitly excluded from this record.",
        "geographic_sovereignty": "not_disclosed",
        "governed_access": "Product and provider controls apply; contract details are not public.",
        "privacy_pii": "not_disclosed",
        "trade_secret": "Provider-license restrictions not disclosed."
      },
      "rights": {
        "derived_model": "not_disclosed",
        "exclusivity": "not_disclosed",
        "inference_retrieval": "yes_explicit",
        "ownership_transfer": "no_disclosed_transfer",
        "post_training": "not_disclosed",
        "retention": "indexed_on_openai_infrastructure_for_some_sources",
        "sublicensing": "not_disclosed",
        "training": "not_disclosed"
      },
      "scores": {
        "asset": {
          "coverage_pct": 0,
          "factor_ratings": {
            "decision_outcome_richness": null,
            "decision_value_density": null,
            "domain_value": null,
            "historical_depth": null,
            "model_learning_usefulness": null,
            "non_replicability": null,
            "proprietary_advantage": null,
            "real_world_grounding": null,
            "refreshability": null,
            "rights_usability": null,
            "uniqueness_scarcity": null
          },
          "rationale": "Legacy score withdrawn pending evidence-linked scoring and human approval. The superseded v1.0.0 release retains the original values.",
          "score": null
        },
        "methodology_version": "1.0.0",
        "transaction_signal": {
          "coverage_pct": 0,
          "factor_ratings": {
            "clean_price_discovery": null,
            "competing_bids": null,
            "data_consideration_separability": null,
            "disclosed_economics": null,
            "explicit_model_use": null,
            "independent_model_uplift": null,
            "precedent_value": null,
            "rights_clarity": null,
            "strategic_buyer_quality": null
          },
          "rationale": "Legacy score withdrawn pending evidence-linked scoring and human approval. The superseded v1.0.0 release retains the original values.",
          "score": null
        }
      },
      "seller_data_owner": [
        "LSEG",
        "PitchBook",
        "Daloopa",
        "Crunchbase",
        "Quartr",
        "LSEG News"
      ],
      "slug": "openai-built-in-financial-data-partnerships",
      "source_code_software_asset": "no",
      "sources": [
        {
          "publisher": "OpenAI",
          "title": "ChatGPT for Financial Services",
          "type": "primary",
          "url": "https://help.openai.com/en/articles/12608093-chatgpt-for-financial-services"
        },
        {
          "publisher": "Reuters",
          "title": "OpenAI launches ChatGPT for financial services industry",
          "type": "secondary",
          "url": "https://www.reuters.com/business/openai-launches-chatgpt-financial-services-industry-2026-09-10/"
        },
        {
          "publisher": "Daloopa",
          "title": "Daloopa data infrastructure",
          "type": "primary",
          "url": "https://daloopa.com/"
        }
      ],
      "transaction_id": "SS-AIDT-2026-0006",
      "transaction_status": "launched_active"
    },
    {
      "access_structure": "Three-year joint scientific laboratory to develop frontier models using TotalEnergies geoscience data and expertise. Specific licenses, data custody and allocation of resulting model rights are not disclosed.",
      "analysis_boundary": "Program duration, investment lower bound, joint lab and geoscience history are disclosed. Scores are qualitative analysis, not measured uplift or a valuation of the dataset. Ownership, retention and derived-model rights are not disclosed.",
      "announcement_date": "2026-09-15",
      "asset_disposition": "Embedded / Continuing",
      "buyer_ai_developer": [
        "Mistral AI"
      ],
      "canonical_path": "/ai-data-transactions/deals/ss-aidt-2026-0007/",
      "confidence": "high",
      "corpus_description": "Nearly a century of TotalEnergies geoscience data, knowledge, and reservoir-engineering expertise used in a joint laboratory to develop frontier and agentic AI models for exploration and reservoir decisions.",
      "corpus_renewal": "Renewable Corpus",
      "data_category": [
        "Geological / subsurface data",
        "Reservoir engineering",
        "Decision histories",
        "Scientific data"
      ],
      "date_added": "2026-09-15",
      "date_last_reviewed": "2026-09-16",
      "decision_outcome_richness": "Strategic Signal analysis: reservoir decisions are economically consequential, but decision-linked outcome labeling in the corpus has not been demonstrated publicly.",
      "disclosed_consideration": {
        "amount": 100000000,
        "amount_qualifier": "greater_than",
        "currency": "EUR",
        "description": "More than €100 million over three years; explicitly a program-wide investment, not a dataset price.",
        "status": "disclosed_program_commitment"
      },
      "economics_scope": "bundled_research_compute_people_and_implementation",
      "estimated_economics": {
        "amount": null,
        "currency": null,
        "method": null,
        "status": "not_estimated"
      },
      "historical_depth": "Nearly one century",
      "industry": "Energy & geoscience",
      "likely_model_use": "Develop frontier and agentic models that generate exploration scenarios, interpret subsurface data, optimize reservoirs, and support expert decisions.",
      "likely_strategic_value": "A rare, longitudinal state–decision–action–outcome corpus in a domain where individual decisions can move billions of euros of value.",
      "m_and_a_implications": "Large program economics confirm strategic importance but do not price the corpus separately; transactions for energy data should distinguish data value from scientists, compute, and implementation.",
      "primary_transaction_structure": "Strategic Model Co-Development",
      "publication_review": {
        "approval_channel": "owner_conversation",
        "decision_id": "conversation-20260916-5768A8330103",
        "packet_sha256": "656620707ca856653b6dd4b75854d841b3c4106e4e89e51e1088de26891e85e2",
        "record_sha256": "f40e71b98171603e1825f3e02d1584b2c8e0822e8f4d59cc9e102b8ea534dd6a",
        "reviewed_at": "2026-09-16T21:17:28.112Z",
        "reviewer": "Mike Ye",
        "status": "human_reviewed"
      },
      "refreshability": "continuing_exploration_and_reservoir_operations",
      "restrictions": {
        "de_identification": "not_disclosed",
        "employee_customer_data": "not_applicable_or_not_disclosed",
        "geographic_sovereignty": "European strategic-technology context; binding geographic terms not disclosed.",
        "governed_access": "Joint laboratory; technical access controls and contractual rights allocation are not disclosed.",
        "privacy_pii": "not_applicable_or_not_disclosed",
        "trade_secret": "Announcement emphasizes protecting intellectual property; specific binding provisions are not disclosed."
      },
      "rights": {
        "derived_model": "unknown_joint_program",
        "exclusivity": "not_disclosed",
        "inference_retrieval": "unknown",
        "ownership_transfer": "not_disclosed",
        "post_training": "unknown",
        "retention": "not_disclosed",
        "sublicensing": "not_disclosed",
        "training": "yes_develop_frontier_models_on_corpus"
      },
      "scores": {
        "asset": {
          "coverage_pct": 88,
          "factor_evidence": {
            "decision_outcome_richness": {
              "basis": "analysis",
              "rationale": "Analysis: operational context suggests useful decision relationships; missing outcome-label evidence limits the rating.",
              "source_ids": [
                "E1"
              ]
            },
            "decision_value_density": {
              "basis": "analysis",
              "rationale": "Analysis: mistakes in this domain can have major economic consequences. This is domain judgment, not a disclosed corpus valuation.",
              "source_ids": [
                "E1"
              ]
            },
            "domain_value": {
              "basis": "analysis",
              "rationale": "Analysis: the disclosed domain supports economically consequential enterprise activities; this is not a model-uplift result.",
              "source_ids": [
                "E1"
              ]
            },
            "historical_depth": {
              "basis": "analysis",
              "rationale": "Analysis: the announced near-century geoscience history suggests considerable temporal depth; uniform record completeness is not demonstrated.",
              "source_ids": [
                "E1"
              ]
            },
            "model_learning_usefulness": {
              "basis": "analysis",
              "rationale": "Analysis: the announcement directly links this corpus to model development; improvement magnitude is unmeasured.",
              "source_ids": [
                "E1"
              ]
            },
            "non_replicability": {
              "basis": "analysis",
              "rationale": "Analysis: reproducing owner-specific operating history would require comparable activity and time.",
              "source_ids": [
                "E1"
              ]
            },
            "proprietary_advantage": {
              "basis": "analysis",
              "rationale": "Analysis: member-owned operational or scientific data offers context beyond ordinary public web corpora.",
              "source_ids": [
                "E1"
              ]
            },
            "real_world_grounding": {
              "basis": "analysis",
              "rationale": "Analysis: the described corporate operational or scientific records are grounded in real activity rather than synthetic-only tasks.",
              "source_ids": [
                "E1"
              ]
            },
            "uniqueness_scarcity": {
              "basis": "analysis",
              "rationale": "Analysis: access to this owner-contributed domain corpus is difficult to reproduce; exclusivity is not established.",
              "source_ids": [
                "E1"
              ]
            }
          },
          "factor_ratings": {
            "decision_outcome_richness": 3.5,
            "decision_value_density": 4.5,
            "domain_value": 5,
            "historical_depth": 4.5,
            "model_learning_usefulness": 4,
            "non_replicability": 4,
            "proprietary_advantage": 4,
            "real_world_grounding": 4.5,
            "refreshability": null,
            "rights_usability": null,
            "uniqueness_scarcity": 4.5
          },
          "rationale": "Qualitative Strategic Signal assessment based on the disclosed corpus and arrangement. Unknown factors remain null; no measured model uplift is claimed.",
          "score": 85
        },
        "methodology_version": "1.0.0",
        "transaction_signal": {
          "coverage_pct": 85,
          "factor_evidence": {
            "clean_price_discovery": {
              "basis": "analysis",
              "rationale": "Analysis: no separately priced corporate dataset is identified; the announcement offers little clean data-price discovery.",
              "source_ids": [
                "E1"
              ]
            },
            "data_consideration_separability": {
              "basis": "analysis",
              "rationale": "No separately quantified consideration for data is disclosed. Zero measures disclosure weakness, not zero data value.",
              "source_ids": [
                "E1"
              ]
            },
            "disclosed_economics": {
              "basis": "analysis",
              "rationale": "A program investment lower bound is disclosed, but its allocation to data is not disclosed.",
              "source_ids": [
                "E1"
              ]
            },
            "explicit_model_use": {
              "basis": "analysis",
              "rationale": "The company explicitly connects proprietary corporate data to developing the named AI models.",
              "source_ids": [
                "E1"
              ]
            },
            "precedent_value": {
              "basis": "analysis",
              "rationale": "Analysis: this is a reusable model-co-development structure, but repeatable standalone pricing is not established.",
              "source_ids": [
                "E1"
              ]
            },
            "rights_clarity": {
              "basis": "analysis",
              "rationale": "Analysis: the announced model-development purpose is clear, while detailed transfer, retention and licensing terms remain unknown.",
              "source_ids": [
                "E1"
              ]
            },
            "strategic_buyer_quality": {
              "basis": "analysis",
              "rationale": "Analysis: an identified model developer with a specific domain program makes the observation strategically relevant.",
              "source_ids": [
                "E1"
              ]
            }
          },
          "factor_ratings": {
            "clean_price_discovery": 1,
            "competing_bids": null,
            "data_consideration_separability": 0,
            "disclosed_economics": 4.5,
            "explicit_model_use": 5,
            "independent_model_uplift": null,
            "precedent_value": 4,
            "rights_clarity": 2.5,
            "strategic_buyer_quality": 4
          },
          "rationale": "Qualitative Strategic Signal assessment based on the disclosed corpus and arrangement. Unknown factors remain null; no measured model uplift is claimed.",
          "score": 56
        }
      },
      "seller_data_owner": [
        "TotalEnergies"
      ],
      "slug": "totalenergies-mistral-reservoir-models",
      "source_code_software_asset": "unknown",
      "sources": [
        {
          "publisher": "TotalEnergies",
          "title": "TotalEnergies Announces a Partnership with Mistral to Develop Frontier AI Models for Reservoir Exploration and Engineering",
          "type": "primary",
          "url": "https://totalenergies.com/newsroom/totalenergies-annonce-un-partenariat-avec-mistral-en-vue-de-developper-des-modeles-de-frontiere-dintelligence-artificielle-dedies-a-lexploration-et-a-lingenierie-des-reservoirs-498514"
        }
      ],
      "transaction_id": "SS-AIDT-2026-0007",
      "transaction_status": "announced_three_year_program"
    },
    {
      "access_structure": "Purpose-limited data license to Pathos within a three-party arrangement involving AstraZeneca.",
      "agreement_effective_date": "2025-04-17",
      "analysis_boundary": "This observation covers the data-license leg only. Payment completion, corpus renewal and many contractual restrictions remain unknown. Related fees and financing are context, not additional data transactions.",
      "announcement_date": "2025-04-23",
      "announcement_date_basis": "SEC filing date",
      "asset_disposition": "Embedded / Continuing",
      "buyer_ai_developer": [
        "Pathos AI"
      ],
      "canonical_path": "/ai-data-transactions/deals/ss-aidt-2025-0002/",
      "confidence": "high",
      "confidence_boundary": "High on filed historical terms; current payment performance is unknown.",
      "corpus_description": "A de-identified multimodal oncology dataset licensed for a foundation-model development program.",
      "corpus_renewal": "unknown",
      "data_category": [
        "Scientific / pharmaceutical",
        "De-identified multimodal oncology data"
      ],
      "date_added": "2026-09-16",
      "date_last_reviewed": "2026-09-16",
      "decision_outcome_richness": "unknown; the filing does not define longitudinal outcome fields",
      "disclosed_consideration": {
        "amount": 200000000,
        "currency": "USD",
        "description": "$200m data-license fees over three years, including $50m upfront payable. Up to 50% may be settled in Pathos preferred shares. Separate $35m flows run from AstraZeneca to Tempus and Tempus to Pathos.",
        "status": "disclosed_contractual_fees_not_verified_cash_receipts"
      },
      "economics_scope": "contractually_identified_data_license_fees_within_broader_arrangement",
      "estimated_economics": {
        "amount": null,
        "currency": null,
        "method": null,
        "status": "not_estimated"
      },
      "historical_depth": "not_disclosed",
      "industry": "Pharmaceuticals & biotechnology",
      "likely_model_use": "Develop and train an oncology foundation model.",
      "likely_strategic_value": "Analysis: a specified training license demonstrates that a proprietary corpus can carry contractual consideration.",
      "m_and_a_implications": "Analysis: useful license-fee precedent, not a clean cash sale valuation. Comparability depends on rights, payment form, scope and linked obligations; no annual run-rate or per-patient price is inferred.",
      "primary_transaction_structure": "Training Rights License",
      "refreshability": "not_disclosed",
      "restrictions": {
        "de_identification": "explicit",
        "employee_customer_data": "Detailed consent and patient-data terms not_disclosed",
        "geographic_sovereignty": "not_disclosed",
        "governed_access": "License purpose is development and training of the specified model.",
        "privacy_pii": "de_identified_dataset",
        "trade_secret": "not_disclosed"
      },
      "rights": {
        "derived_model": "Tempus receives model-use license with field restrictions",
        "exclusivity": "not_disclosed",
        "inference_retrieval": "not_disclosed",
        "ownership_transfer": "no_disclosed_transfer",
        "post_training": "not_disclosed",
        "retention": "not_disclosed",
        "sublicensing": "Tempus may sublicense model to AstraZeneca; dataset sublicensing not_disclosed",
        "training": "yes_explicit_purpose_limited"
      },
      "scores": {
        "asset": {
          "coverage_pct": 71,
          "factor_evidence": {
            "decision_value_density": {
              "basis": "analysis",
              "rationale": "Oncology research choices can be consequential; the rating does not assert actual patient-outcome improvement.",
              "source_ids": [
                "E1"
              ]
            },
            "domain_value": {
              "basis": "analysis",
              "rationale": "The use case lies in economically significant scientific research.",
              "source_ids": [
                "E1"
              ]
            },
            "model_learning_usefulness": {
              "basis": "analysis",
              "rationale": "The agreement explicitly links the dataset to the model training task.",
              "source_ids": [
                "E1"
              ]
            },
            "non_replicability": {
              "basis": "analysis",
              "rationale": "Reassembly may require specialized data collection; no reproduction-cost estimate is available.",
              "source_ids": [
                "E1"
              ]
            },
            "proprietary_advantage": {
              "basis": "analysis",
              "rationale": "The purpose-specific private license offers access beyond generic public material.",
              "source_ids": [
                "E1"
              ]
            },
            "real_world_grounding": {
              "basis": "analysis",
              "rationale": "A de-identified domain dataset provides real-world grounding; field detail is limited.",
              "source_ids": [
                "E1"
              ]
            },
            "rights_usability": {
              "basis": "analysis",
              "rationale": "Specified training permission is useful, but purpose limits and undisclosed retention constrain reuse.",
              "source_ids": [
                "E1"
              ]
            },
            "uniqueness_scarcity": {
              "basis": "analysis",
              "rationale": "The licensed private corpus is differentiated, but exact scarcity cannot be quantified.",
              "source_ids": [
                "E1"
              ]
            }
          },
          "factor_ratings": {
            "decision_outcome_richness": null,
            "decision_value_density": 4,
            "domain_value": 4.5,
            "historical_depth": null,
            "model_learning_usefulness": 4.5,
            "non_replicability": 3.5,
            "proprietary_advantage": 4,
            "real_world_grounding": 4,
            "refreshability": null,
            "rights_usability": 3.5,
            "uniqueness_scarcity": 4
          },
          "rationale": "Potentially valuable domain training corpus; uncertainty about contents, age and renewal materially limits coverage. Score is strategic analysis, not measured utility.",
          "score": 82
        },
        "methodology_version": "1.0.0",
        "transaction_signal": {
          "coverage_pct": 85,
          "factor_evidence": {
            "clean_price_discovery": {
              "basis": "analysis",
              "rationale": "Linked financing and reciprocal obligations weaken the isolated market-price interpretation.",
              "source_ids": [
                "E1"
              ]
            },
            "data_consideration_separability": {
              "basis": "analysis",
              "rationale": "The filing identifies a data-license payment leg even though it sits within a broader arrangement.",
              "source_ids": [
                "E1"
              ]
            },
            "disclosed_economics": {
              "basis": "analysis",
              "rationale": "Contractual fees and payment form are disclosed, but realized proceeds are not verified.",
              "source_ids": [
                "E1"
              ]
            },
            "explicit_model_use": {
              "basis": "analysis",
              "rationale": "Model development and training are the express permitted purposes.",
              "source_ids": [
                "E1"
              ]
            },
            "precedent_value": {
              "basis": "analysis",
              "rationale": "The contract structure offers a useful reference if future comparisons preserve its restrictions.",
              "source_ids": [
                "E1"
              ]
            },
            "rights_clarity": {
              "basis": "analysis",
              "rationale": "Purpose and some model rights are visible; exclusivity, retention and data sublicensing are not.",
              "source_ids": [
                "E1"
              ]
            },
            "strategic_buyer_quality": {
              "basis": "analysis",
              "rationale": "A specified specialist model developer provides credible use, without implying open-market bidding.",
              "source_ids": [
                "E1"
              ]
            }
          },
          "factor_ratings": {
            "clean_price_discovery": 2,
            "competing_bids": null,
            "data_consideration_separability": 4,
            "disclosed_economics": 4.5,
            "explicit_model_use": 5,
            "independent_model_uplift": null,
            "precedent_value": 4,
            "rights_clarity": 3.5,
            "strategic_buyer_quality": 3
          },
          "rationale": "Explicit fee allocation is valuable pricing evidence. Equity settlement, reciprocal commitments and missing comparable bids prevent treatment as a clean cash market price.",
          "score": 75
        }
      },
      "seller_data_owner": [
        "Tempus AI"
      ],
      "slug": "tempus-pathos-oncology-data-license",
      "source_code_software_asset": "no",
      "sources": [
        {
          "publisher": "Tempus AI / SEC",
          "title": "Form 8-K filed April 23, 2025, Item 8.01",
          "type": "primary",
          "url": "https://www.sec.gov/Archives/edgar/data/1717115/000119312525090062/d944168d8k.htm"
        }
      ],
      "transaction_id": "SS-AIDT-2025-0002",
      "transaction_status": "agreement_disclosed; current payment performance not verified",
      "publication_review": {
        "status": "human_reviewed",
        "approval_channel": "owner_conversation",
        "reviewer": "Mike Ye",
        "reviewed_at": "2026-09-16T21:38:20.346Z",
        "decision_id": "conversation-20260916-CD20DA31F0CF",
        "packet_sha256": "401f767bb940538d1a214a8130fd811dfa4275b939fcba6f15377ce87c2ac9b2",
        "record_sha256": "12bdd18b6c5a16d4e4612db7460712415283174fb14daa888dd8dd3ebb6a945b"
      }
    },
    {
      "transaction_id": "SS-AIDT-2026-0008",
      "slug": "bolt-arcee-forge-training-data-license",
      "canonical_path": "/ai-data-transactions/deals/ss-aidt-2026-0008/",
      "announcement_date": "2026-09-14",
      "buyer_ai_developer": [
        "Arcee AI"
      ],
      "seller_data_owner": [
        "Bolt / StackBlitz"
      ],
      "transaction_status": "announced_preview_training_window",
      "industry": "Software development & AI training data",
      "data_category": [
        "Source code and project files",
        "Developer prompts",
        "Tool-call traces",
        "Edit histories",
        "Error and fix traces"
      ],
      "corpus_description": "Opted-in Bolt Forge sessions containing developer prompts, code, project files and configuration, tool calls, edit histories and fix traces from individual Pro users during the September 14–October 14, 2026 preview window.",
      "primary_transaction_structure": "Training Rights License",
      "access_structure": "Bolt de-identifies opted-in Forge sessions inside its infrastructure and licenses resulting datasets to AI developers under a data license agreement. Arcee AI is the first named recipient; Bolt says the datasets may also be licensed to other AI developers.",
      "asset_disposition": "Embedded / Continuing",
      "corpus_renewal": "Renewable Corpus",
      "source_code_software_asset": "yes",
      "disclosed_consideration": {
        "status": "disclosed_non_cash_user_consideration",
        "amount": null,
        "currency": null,
        "description": "Individual Pro users receive up to 50× additional Forge usage during the preview; Bolt describes the allocation as payment for sharing. Bolt may also receive cash for licensed datasets, but no amount is disclosed."
      },
      "estimated_economics": {
        "status": "not_estimated",
        "amount": null,
        "currency": null,
        "method": null
      },
      "economics_scope": "In-kind usage allocation to contributing users; any Arcee-to-Bolt license payment is undisclosed.",
      "rights": {
        "training": "Explicit: the licensed sessions feed Arcee AI's first training run for a trillion-parameter-class model.",
        "post_training": "not_disclosed",
        "inference_retrieval": "not_disclosed",
        "ownership_transfer": "No ownership transfer is disclosed; the announced structure is a data license agreement.",
        "exclusivity": "Non-exclusive at Bolt level: Bolt says datasets may be licensed to other AI developers. Arcee-specific exclusivity terms are not disclosed.",
        "retention": "Users can request that Bolt stop using already-collected Forge content; deletion mechanics and treatment of trained-model effects are not disclosed.",
        "derived_model": "Arcee will train a trillion-parameter-class model; ownership and allocation of derived weights are not disclosed.",
        "sublicensing": "Bolt may license the datasets to additional AI developers; Arcee's sublicensing rights are not disclosed."
      },
      "restrictions": {
        "governed_access": "One-tap opt-in is required before collection; switching away from Forge stops new collection.",
        "geographic_sovereignty": "Sessions from the EEA, United Kingdom and Switzerland are excluded from training and dataset licensing.",
        "privacy_pii": "Bolt says personal information is stripped and the de-identification pipeline is tested with seeded data.",
        "trade_secret": "Bolt says secrets and sensitive data are stripped before material leaves its infrastructure; the effectiveness and contractual remedies are not independently verified.",
        "de_identification": "Explicitly required and performed by Bolt before transfer.",
        "employee_customer_data": "Team and Enterprise workspaces are excluded; the disclosed corpus comes from opted-in individual Pro users."
      },
      "refreshability": "Renewable while users opt into Forge; the first disclosed Arcee training window is September 14–October 14, 2026.",
      "historical_depth": "Newly collected preview corpus beginning September 14, 2026; no long operating history is disclosed.",
      "decision_outcome_richness": "The corpus records software-building sequences including attempts, edits, tool calls, errors and fixes, creating observable short-cycle action and outcome traces.",
      "likely_model_use": "Training open-weight coding and software-building models to complete real application-development workflows in fewer attempts.",
      "likely_strategic_value": "Captures real developer interaction and repair traces that cannot be reconstructed reliably from static public code repositories alone.",
      "m_and_a_implications": "Shows that a software platform can monetize consented workflow telemetry separately from subscriptions and can compensate contributors with compute or usage rather than cash.",
      "analysis_boundary": "The data-license structure, corpus fields, consent controls, initial Arcee recipient and model-training purpose are disclosed by Bolt. Cash economics, raw-data retention, derived-weight ownership, Arcee sublicensing and measured model uplift are not disclosed.",
      "confidence": "high",
      "scores": {
        "methodology_version": "1.0.0",
        "asset": {
          "score": 72,
          "coverage_pct": 100,
          "rationale": "Strategic Signal analysis. The workflow traces are unusually useful for software-agent learning, but the corpus is new, excludes enterprise workspaces and has no independent uplift evidence.",
          "factor_ratings": {
            "uniqueness_scarcity": 4,
            "historical_depth": 0.5,
            "decision_outcome_richness": 4,
            "decision_value_density": 2.5,
            "real_world_grounding": 4,
            "domain_value": 4,
            "refreshability": 4.5,
            "proprietary_advantage": 4,
            "model_learning_usefulness": 4.5,
            "non_replicability": 3.5,
            "rights_usability": 3.5
          },
          "factor_evidence": {
            "uniqueness_scarcity": {
              "basis": "analysis",
              "source_ids": [
                "E1"
              ],
              "rationale": "Static code is common, but consented end-to-end build, error and fix traces are difficult to scrape or reconstruct."
            },
            "historical_depth": {
              "basis": "disclosed_fact",
              "source_ids": [
                "E1"
              ],
              "rationale": "The disclosed corpus begins with a one-month preview window in September 2026, so historical depth is presently limited."
            },
            "decision_outcome_richness": {
              "basis": "analysis",
              "source_ids": [
                "E1"
              ],
              "rationale": "Edit histories, tool calls and fix traces expose repeated action-and-result sequences within software-building workflows."
            },
            "decision_value_density": {
              "basis": "analysis",
              "source_ids": [
                "E1"
              ],
              "rationale": "Individual coding decisions are usually lower-value than clinical or capital decisions, though successful repairs carry useful technical information."
            },
            "real_world_grounding": {
              "basis": "analysis",
              "source_ids": [
                "E1"
              ],
              "rationale": "The sessions arise from real user projects and tool interactions rather than a synthetic-only benchmark corpus."
            },
            "domain_value": {
              "basis": "analysis",
              "source_ids": [
                "E1"
              ],
              "rationale": "Coding-agent capability has broad economic value across software development, even though the disclosed users are not enterprise workspaces."
            },
            "refreshability": {
              "basis": "disclosed_fact",
              "source_ids": [
                "E1"
              ],
              "rationale": "Forge can continue collecting new opted-in sessions as users build, subject to the stated program and geography limits."
            },
            "proprietary_advantage": {
              "basis": "analysis",
              "source_ids": [
                "E1"
              ],
              "rationale": "Bolt controls the platform telemetry and de-identification pipeline that create a differentiated corpus beyond public repositories."
            },
            "model_learning_usefulness": {
              "basis": "analysis",
              "source_ids": [
                "E1"
              ],
              "rationale": "Bolt directly links the sessions to training a new model and explains that repair traces can reduce the number of attempts required."
            },
            "non_replicability": {
              "basis": "analysis",
              "source_ids": [
                "E1"
              ],
              "rationale": "Another builder could collect similar telemetry, but reproducing Bolt's exact user interactions and repair sequences would require comparable scale and consent."
            },
            "rights_usability": {
              "basis": "analysis",
              "source_ids": [
                "E1"
              ],
              "rationale": "Opt-in collection, de-identification and an explicit data license support usability, while retention and downstream model terms remain undisclosed."
            }
          }
        },
        "transaction_signal": {
          "score": 66,
          "coverage_pct": 85,
          "rationale": "Strategic Signal analysis. The arrangement cleanly proves training use and non-cash contributor consideration, but provides weak cash price discovery and no measured uplift.",
          "factor_ratings": {
            "disclosed_economics": 3,
            "clean_price_discovery": 1,
            "data_consideration_separability": 3.5,
            "explicit_model_use": 5,
            "rights_clarity": 4,
            "competing_bids": null,
            "strategic_buyer_quality": 3.5,
            "precedent_value": 4.5,
            "independent_model_uplift": null
          },
          "factor_evidence": {
            "disclosed_economics": {
              "basis": "disclosed_fact",
              "source_ids": [
                "E1"
              ],
              "rationale": "Bolt discloses up to 50× usage as contributor payment and says it may receive payment, but gives no cash value for the dataset license."
            },
            "clean_price_discovery": {
              "basis": "analysis",
              "source_ids": [
                "E1"
              ],
              "rationale": "The usage benefit is bundled with a preview program and no buyer payment amount is disclosed, limiting clean price discovery."
            },
            "data_consideration_separability": {
              "basis": "analysis",
              "source_ids": [
                "E1"
              ],
              "rationale": "Bolt explicitly describes the usage allocation as payment for sharing, but no cash-equivalent valuation is provided."
            },
            "explicit_model_use": {
              "basis": "disclosed_fact",
              "source_ids": [
                "E1"
              ],
              "rationale": "The announcement directly states that shared sessions feed Arcee's first training run for a trillion-parameter-class model."
            },
            "rights_clarity": {
              "basis": "analysis",
              "source_ids": [
                "E1"
              ],
              "rationale": "Training, opt-in, exclusions and Bolt's ability to license other developers are clear; retention and derived-model allocation are not."
            },
            "strategic_buyer_quality": {
              "basis": "analysis",
              "source_ids": [
                "E1"
              ],
              "rationale": "Arcee is a named model developer with a concrete disclosed training run, though it is not a frontier lab of the largest scale."
            },
            "precedent_value": {
              "basis": "analysis",
              "source_ids": [
                "E1"
              ],
              "rationale": "The data-for-usage exchange is a replicable template for software platforms monetizing consented workflow traces."
            }
          }
        }
      },
      "sources": [
        {
          "type": "primary",
          "publisher": "Bolt",
          "title": "What is Bolt Forge?",
          "url": "https://bolt.new/blog/what-is-bolt-forge"
        }
      ],
      "date_added": "2026-09-17",
      "date_last_reviewed": "2026-09-17",
      "publication_review": {
        "status": "human_reviewed",
        "approval_channel": "github_workflow",
        "reviewer": "Trailgenic",
        "reviewed_at": "2026-09-17T17:36:14.674Z",
        "decision_id": "35253728428-1",
        "packet_sha256": "e1891b409964b996e4b464fd5436cf7ef8ae43599ca7f8153171c06be5ce1ff9",
        "record_sha256": "c5a10eeece7f3e720d635e409999286246ebbd48d54636a21ffa50017c641d5e"
      }
    },
    {
      "transaction_id": "SS-AIDT-2026-0009",
      "slug": "silencio-tagalog-cebuano-voice-data-contract",
      "canonical_path": "/ai-data-transactions/deals/ss-aidt-2026-0009/",
      "announcement_date": "2026-09-07",
      "buyer_ai_developer": [
        "Undisclosed leading voice-AI company"
      ],
      "seller_data_owner": [
        "Silencio"
      ],
      "transaction_status": "contract_signed",
      "industry": "Voice AI & training data",
      "data_category": [
        "Human speech recordings",
        "Low-resource language data",
        "Single-speaker voice data",
        "Multi-speaker voice data"
      ],
      "corpus_description": "A contracted corpus of native-speaker Tagalog and Cebuano single- and multi-speaker voice recordings assembled through Silencio's distributed contributor network for a leading voice-AI company.",
      "primary_transaction_structure": "Proprietary Corpus License",
      "access_structure": "Dataset contract and delivery following procurement, legal, security and sample evaluation. The buyer, delivery volume, term, exclusivity and detailed license language are not disclosed.",
      "asset_disposition": "Stranded / Separable",
      "corpus_renewal": "Finite Corpus",
      "source_code_software_asset": "no",
      "disclosed_consideration": {
        "status": "disclosed_contract_value",
        "amount": 250000,
        "currency": "USD",
        "description": "Dataset-specific contract value disclosed as $250,000; the buyer is unnamed."
      },
      "estimated_economics": {
        "status": "not_estimated",
        "amount": null,
        "currency": null,
        "method": null
      },
      "economics_scope": "Specific Tagalog and Cebuano single- and multi-speaker voice-data contract.",
      "rights": {
        "training": "The buyer is a voice-AI company and Silencio states AI labs train on its data; the exact contract grant is not disclosed.",
        "post_training": "not_disclosed",
        "inference_retrieval": "not_disclosed",
        "ownership_transfer": "not_disclosed",
        "exclusivity": "not_disclosed",
        "retention": "not_disclosed",
        "derived_model": "not_disclosed",
        "sublicensing": "not_disclosed"
      },
      "restrictions": {
        "governed_access": "Procurement, legal and security review are disclosed; binding access controls are not.",
        "geographic_sovereignty": "not_disclosed",
        "privacy_pii": "Voice recordings can carry biometric and identity risk; contract-specific privacy terms are not disclosed.",
        "trade_secret": "not_disclosed",
        "de_identification": "not_disclosed",
        "employee_customer_data": "not_applicable; contributors record data for compensation rather than supplying enterprise employee or customer records."
      },
      "refreshability": "The contracted delivery is finite, while Silencio's contributor network can produce follow-on recordings and adjacent-language corpora.",
      "historical_depth": "Silencio began gathering data in November 2025; the age and duration represented in this specific delivery are not disclosed.",
      "decision_outcome_richness": "The corpus provides speech examples rather than a state–decision–action–outcome history; decision richness is limited.",
      "likely_model_use": "Training or improving speech recognition, multilingual voice generation or voice-agent performance in Tagalog and Cebuano.",
      "likely_strategic_value": "Low-resource native-speaker recordings are difficult to scrape and can fill coverage gaps that synthetic or high-resource-language data does not resolve.",
      "m_and_a_implications": "The disclosed $250,000 contract provides a rare observable price point for a defined low-resource-language voice corpus, though volume, hours and exact rights are unknown.",
      "analysis_boundary": "Contract value, languages, speaker formats and buyer type are disclosed by an investor in Silencio. Buyer identity, dataset size, ownership, exclusivity, retention, sublicensing, derived-model rights and independent model uplift are not disclosed.",
      "confidence": "medium_high",
      "scores": {
        "methodology_version": "1.0.0",
        "asset": {
          "score": 63,
          "coverage_pct": 100,
          "rationale": "Strategic Signal analysis. The recordings are scarce, consent-oriented and useful for model coverage, but they contain little decision-outcome structure and have limited disclosed historical depth.",
          "factor_ratings": {
            "uniqueness_scarcity": 4.5,
            "historical_depth": 1,
            "decision_outcome_richness": 0.5,
            "decision_value_density": 1,
            "real_world_grounding": 5,
            "domain_value": 4,
            "refreshability": 4,
            "proprietary_advantage": 4.5,
            "model_learning_usefulness": 4.5,
            "non_replicability": 4,
            "rights_usability": 3.5
          },
          "factor_evidence": {
            "uniqueness_scarcity": {
              "basis": "analysis",
              "source_ids": [
                "E1",
                "E2"
              ],
              "rationale": "Native-speaker Tagalog and Cebuano recordings cannot be assembled reliably from the open web at the same coverage and consent standard."
            },
            "historical_depth": {
              "basis": "disclosed_fact",
              "source_ids": [
                "E1"
              ],
              "rationale": "Silencio says collection began in November 2025, so the disclosed business has limited temporal depth relative to longitudinal operating datasets."
            },
            "decision_outcome_richness": {
              "basis": "analysis",
              "source_ids": [
                "E1"
              ],
              "rationale": "Speech recordings contain linguistic observations but not a disclosed sequence of economically consequential decisions and outcomes."
            },
            "decision_value_density": {
              "basis": "analysis",
              "source_ids": [
                "E1"
              ],
              "rationale": "The corpus improves language coverage but does not encode high-value underwriting, clinical, industrial or capital-allocation decisions."
            },
            "real_world_grounding": {
              "basis": "disclosed_fact",
              "source_ids": [
                "E1",
                "E2"
              ],
              "rationale": "The recordings are contributed by native speakers in relevant markets rather than described as synthetic speech."
            },
            "domain_value": {
              "basis": "analysis",
              "source_ids": [
                "E1",
                "E2"
              ],
              "rationale": "Multilingual speech capability is commercially valuable for voice agents and speech systems, especially in underserved languages."
            },
            "refreshability": {
              "basis": "analysis",
              "source_ids": [
                "E1",
                "E2"
              ],
              "rationale": "Silencio's distributed network can collect follow-on speech and related languages, though this contract is a finite delivery."
            },
            "proprietary_advantage": {
              "basis": "analysis",
              "source_ids": [
                "E1",
                "E2"
              ],
              "rationale": "Silencio controls consented contributor relationships and collection operations that are unavailable through ordinary web scraping."
            },
            "model_learning_usefulness": {
              "basis": "analysis",
              "source_ids": [
                "E1",
                "E2"
              ],
              "rationale": "The corpus directly addresses model coverage in languages with little public training data; the magnitude of improvement is unmeasured."
            },
            "non_replicability": {
              "basis": "analysis",
              "source_ids": [
                "E1",
                "E2"
              ],
              "rationale": "A competitor could recruit speakers, but reproducing geographic coverage, consent and collection at comparable speed requires substantial network infrastructure."
            },
            "rights_usability": {
              "basis": "analysis",
              "source_ids": [
                "E1",
                "E2"
              ],
              "rationale": "The data is marketed as consent-cleared and licensing-ready, while contract-specific retention, exclusivity and downstream rights remain unknown."
            }
          }
        },
        "transaction_signal": {
          "score": 89,
          "coverage_pct": 79,
          "rationale": "Strategic Signal analysis. The contract supplies unusually clean dataset-specific cash economics for a defined language corpus, but detailed rights and the buyer identity remain undisclosed.",
          "factor_ratings": {
            "disclosed_economics": 5,
            "clean_price_discovery": 4.5,
            "data_consideration_separability": 5,
            "explicit_model_use": 4.5,
            "rights_clarity": 3,
            "competing_bids": null,
            "strategic_buyer_quality": null,
            "precedent_value": 4,
            "independent_model_uplift": null
          },
          "factor_evidence": {
            "disclosed_economics": {
              "basis": "disclosed_fact",
              "source_ids": [
                "E1"
              ],
              "rationale": "The issuer-responsible announcement states a $250,000 contract value for the defined Tagalog and Cebuano corpus."
            },
            "clean_price_discovery": {
              "basis": "analysis",
              "source_ids": [
                "E1"
              ],
              "rationale": "The cash amount is tied to a specific corpus, although undisclosed volume and contract term prevent a per-hour benchmark."
            },
            "data_consideration_separability": {
              "basis": "disclosed_fact",
              "source_ids": [
                "E1"
              ],
              "rationale": "The announcement presents the $250,000 amount as a voice-data contract rather than a bundled software deployment."
            },
            "explicit_model_use": {
              "basis": "analysis",
              "source_ids": [
                "E1",
                "E2"
              ],
              "rationale": "The buyer is identified as a leading voice-AI company and Silencio says AI labs train on its network data; exact training language is not quoted from the contract."
            },
            "rights_clarity": {
              "basis": "analysis",
              "source_ids": [
                "E1",
                "E2"
              ],
              "rationale": "A data contract and consent-oriented supply are clear, but ownership, exclusivity, retention and sublicensing are undisclosed."
            },
            "precedent_value": {
              "basis": "analysis",
              "source_ids": [
                "E1"
              ],
              "rationale": "A dataset-specific cash price for named low-resource languages can inform future speech-data transactions, subject to missing volume normalization."
            }
          }
        }
      },
      "sources": [
        {
          "type": "primary_syndicated",
          "publisher": "Advanced Blockchain AG via EQS News",
          "title": "Portfolio company Silencio signs record voice-data contract as AI labs turn to low-resource languages",
          "url": "https://www.eqs-news.com/news/corporate/advanced-blockchain-ag-portfolio-company-silencio-signs-record-voice-data-contract-as-ai-labs-turn-to-low-resource-languages/87e56ca5-9883-4f82-8b2e-d265764a5dc7_en"
        },
        {
          "type": "primary",
          "publisher": "Silencio",
          "title": "Silencio AI",
          "url": "https://www.silencio.network/"
        }
      ],
      "date_added": "2026-09-17",
      "date_last_reviewed": "2026-09-17",
      "publication_review": {
        "status": "human_reviewed",
        "approval_channel": "github_workflow",
        "reviewer": "Trailgenic",
        "reviewed_at": "2026-09-17T17:37:52.520Z",
        "decision_id": "35253885258-1",
        "packet_sha256": "b084f0f7a0ae807c847525e517d07eaafda15abc1df076c4c07397f0a2558baf",
        "record_sha256": "165b8c0418f97c60c4332558ccb0a13968fb6c5ecb635a68041e380d662f75a0"
      }
    },
    {
      "transaction_id": "SS-AIDT-2026-0010",
      "slug": "silencio-african-language-voice-data-license",
      "canonical_path": "/ai-data-transactions/deals/ss-aidt-2026-0010/",
      "announcement_date": "2026-09-07",
      "buyer_ai_developer": [
        "Undisclosed leading AI developer"
      ],
      "seller_data_owner": [
        "Silencio"
      ],
      "transaction_status": "license_signed",
      "industry": "Voice AI & training data",
      "data_category": [
        "Human speech recordings",
        "Low-resource language data",
        "African language datasets",
        "Transcription-ready voice data"
      ],
      "corpus_description": "Native-speaker voice datasets covering Swahili, Amharic, Hausa, Wolof and Yoruba, assembled through Silencio's distributed contributor network under a separate license to an unnamed leading AI developer.",
      "primary_transaction_structure": "Proprietary Corpus License",
      "access_structure": "A separate data licensing agreement for five named African-language datasets. Economics, volume, delivery term, exclusivity and detailed license language are not disclosed.",
      "asset_disposition": "Stranded / Separable",
      "corpus_renewal": "Finite Corpus",
      "source_code_software_asset": "no",
      "disclosed_consideration": {
        "status": "not_disclosed",
        "amount": null,
        "currency": null,
        "description": "The announcement discloses the license but not its consideration."
      },
      "estimated_economics": {
        "status": "not_estimated",
        "amount": null,
        "currency": null,
        "method": null
      },
      "economics_scope": "Separate license for five African-language datasets; no transaction economics are disclosed.",
      "rights": {
        "training": "Silencio states that AI labs train on its data and identifies the counterparty as an AI developer; the exact contract grant is not disclosed.",
        "post_training": "not_disclosed",
        "inference_retrieval": "not_disclosed",
        "ownership_transfer": "not_disclosed",
        "exclusivity": "not_disclosed",
        "retention": "not_disclosed",
        "derived_model": "not_disclosed",
        "sublicensing": "not_disclosed"
      },
      "restrictions": {
        "governed_access": "A commercial data license is disclosed; binding access controls are not.",
        "geographic_sovereignty": "not_disclosed",
        "privacy_pii": "Voice recordings can carry biometric and identity risk; contract-specific privacy terms are not disclosed.",
        "trade_secret": "not_disclosed",
        "de_identification": "not_disclosed",
        "employee_customer_data": "not_applicable; contributors record data for compensation rather than supplying enterprise employee or customer records."
      },
      "refreshability": "The initial five-language delivery is finite; the contributor network supports follow-on collection, paid transcription and additional languages.",
      "historical_depth": "Silencio began gathering data in November 2025; the age and duration represented in these specific datasets are not disclosed.",
      "decision_outcome_richness": "The corpus provides speech examples rather than a state–decision–action–outcome history; decision richness is limited.",
      "likely_model_use": "Training or improving speech recognition, multilingual voice generation or voice-agent performance across five underrepresented African languages.",
      "likely_strategic_value": "Native-speaker recordings in languages with little public data can expand model coverage and reduce dependence on synthetic or translated substitutes.",
      "m_and_a_implications": "Confirms that one proprietary collection network can support repeat, multi-language AI-data licenses; undisclosed economics prevent valuation benchmarking for this agreement.",
      "analysis_boundary": "The separate license, five initial languages and buyer type are disclosed by an investor in Silencio. Buyer identity, consideration, dataset size, ownership, exclusivity, retention, sublicensing, derived-model rights and independent uplift are not disclosed.",
      "confidence": "medium_high",
      "scores": {
        "methodology_version": "1.0.0",
        "asset": {
          "score": 63,
          "coverage_pct": 100,
          "rationale": "Strategic Signal analysis. The recordings are scarce, consent-oriented and useful for model coverage, but they contain little decision-outcome structure and have limited disclosed historical depth.",
          "factor_ratings": {
            "uniqueness_scarcity": 4.5,
            "historical_depth": 1,
            "decision_outcome_richness": 0.5,
            "decision_value_density": 1,
            "real_world_grounding": 5,
            "domain_value": 4,
            "refreshability": 4,
            "proprietary_advantage": 4.5,
            "model_learning_usefulness": 4.5,
            "non_replicability": 4,
            "rights_usability": 3.5
          },
          "factor_evidence": {
            "uniqueness_scarcity": {
              "basis": "analysis",
              "source_ids": [
                "E1",
                "E2"
              ],
              "rationale": "Native-speaker recordings across the five named languages are difficult to source at scale from public web material."
            },
            "historical_depth": {
              "basis": "disclosed_fact",
              "source_ids": [
                "E1"
              ],
              "rationale": "Silencio says collection began in November 2025, so the disclosed business has limited temporal depth relative to longitudinal operating datasets."
            },
            "decision_outcome_richness": {
              "basis": "analysis",
              "source_ids": [
                "E1"
              ],
              "rationale": "Speech recordings contain linguistic observations but not a disclosed sequence of economically consequential decisions and outcomes."
            },
            "decision_value_density": {
              "basis": "analysis",
              "source_ids": [
                "E1"
              ],
              "rationale": "The corpus improves language coverage but does not encode high-value underwriting, clinical, industrial or capital-allocation decisions."
            },
            "real_world_grounding": {
              "basis": "disclosed_fact",
              "source_ids": [
                "E1",
                "E2"
              ],
              "rationale": "The recordings are contributed by speakers in relevant markets rather than described as synthetic speech."
            },
            "domain_value": {
              "basis": "analysis",
              "source_ids": [
                "E1",
                "E2"
              ],
              "rationale": "Multilingual speech capability is commercially valuable for voice agents and speech systems, especially in underserved languages."
            },
            "refreshability": {
              "basis": "analysis",
              "source_ids": [
                "E1",
                "E2"
              ],
              "rationale": "The network can collect follow-on speech and additional languages even though the initial five-language delivery is finite."
            },
            "proprietary_advantage": {
              "basis": "analysis",
              "source_ids": [
                "E1",
                "E2"
              ],
              "rationale": "Silencio controls consented contributor relationships and collection operations that are unavailable through ordinary web scraping."
            },
            "model_learning_usefulness": {
              "basis": "analysis",
              "source_ids": [
                "E1",
                "E2"
              ],
              "rationale": "The datasets directly address model coverage in languages with little public training data; improvement magnitude is unmeasured."
            },
            "non_replicability": {
              "basis": "analysis",
              "source_ids": [
                "E1",
                "E2"
              ],
              "rationale": "A competitor could recruit speakers, but reproducing multi-country coverage, consent and collection speed requires substantial network infrastructure."
            },
            "rights_usability": {
              "basis": "analysis",
              "source_ids": [
                "E1",
                "E2"
              ],
              "rationale": "The data is marketed as consent-cleared and licensing-ready, while contract-specific retention, exclusivity and downstream rights remain unknown."
            }
          }
        },
        "transaction_signal": {
          "score": null,
          "coverage_pct": 47,
          "rationale": "Strategic Signal analysis. A separable AI-data license is explicit, but undisclosed economics, buyer identity and most contract rights leave evidence coverage below the methodology's 70% threshold.",
          "factor_ratings": {
            "disclosed_economics": null,
            "clean_price_discovery": null,
            "data_consideration_separability": 4.5,
            "explicit_model_use": 4.5,
            "rights_clarity": 3,
            "competing_bids": null,
            "strategic_buyer_quality": null,
            "precedent_value": 4,
            "independent_model_uplift": null
          },
          "factor_evidence": {
            "data_consideration_separability": {
              "basis": "disclosed_fact",
              "source_ids": [
                "E1"
              ],
              "rationale": "The announcement explicitly describes a separate data licensing agreement for five named language datasets."
            },
            "explicit_model_use": {
              "basis": "analysis",
              "source_ids": [
                "E1",
                "E2"
              ],
              "rationale": "The counterparty is described as a leading AI developer and Silencio says AI labs train on its network data; contract wording is not public."
            },
            "rights_clarity": {
              "basis": "analysis",
              "source_ids": [
                "E1",
                "E2"
              ],
              "rationale": "The existence and scope languages of the license are clear, but ownership, exclusivity, retention and sublicensing are undisclosed."
            },
            "precedent_value": {
              "basis": "analysis",
              "source_ids": [
                "E1"
              ],
              "rationale": "The agreement demonstrates repeat multi-language licensing from a renewable contributor network, even though it offers no price benchmark."
            }
          }
        }
      },
      "sources": [
        {
          "type": "primary_syndicated",
          "publisher": "Advanced Blockchain AG via EQS News",
          "title": "Portfolio company Silencio signs record voice-data contract as AI labs turn to low-resource languages",
          "url": "https://www.eqs-news.com/news/corporate/advanced-blockchain-ag-portfolio-company-silencio-signs-record-voice-data-contract-as-ai-labs-turn-to-low-resource-languages/87e56ca5-9883-4f82-8b2e-d265764a5dc7_en"
        },
        {
          "type": "primary",
          "publisher": "Silencio",
          "title": "Silencio AI",
          "url": "https://www.silencio.network/"
        }
      ],
      "date_added": "2026-09-17",
      "date_last_reviewed": "2026-09-17",
      "publication_review": {
        "status": "human_reviewed",
        "approval_channel": "github_workflow",
        "reviewer": "Trailgenic",
        "reviewed_at": "2026-09-17T17:38:43.226Z",
        "decision_id": "35253967579-1",
        "packet_sha256": "edd54770424ddcc1d311f83df2a2a5fce9d23adc3996636c30b8222479b3b11b",
        "record_sha256": "4b81ae9d51db5e8e069e1517003afc4e36600684cd8c67cc4e403d839067a698"
      }
    },
    {
      "transaction_id": "SS-AIDT-2026-0011",
      "slug": "harell-data-controlled-biotech-model-training",
      "canonical_path": "/ai-data-transactions/deals/ss-aidt-2026-0011/",
      "announcement_date": "2026-09-15",
      "buyer_ai_developer": [
        "Authorized Harell Data model builders and scientific research teams"
      ],
      "seller_data_owner": [
        "A-Alpha Bio",
        "Adaptive Biotechnologies"
      ],
      "transaction_status": "active_multi_year_controlled_learning_platform",
      "industry": "Biotechnology, scientific AI, and AI infrastructure",
      "data_category": [
        "Antibody-antigen affinity and structural data",
        "T-cell-receptor binding data",
        "Hidden scientific validation datasets"
      ],
      "corpus_description": "Proprietary scientific datasets hosted in Harell Data's controlled environment, initially including A-Alpha Bio antibody-antigen affinity and structural data and Adaptive Biotechnologies T-cell-receptor binding data, plus hidden validation questions retained by data partners.",
      "primary_transaction_structure": "Federated / Controlled Learning Rights",
      "access_structure": "Authorized model builders bring models into Harell Data's CoreWeave-powered environment for training, fine-tuning, validation, and inference. Raw datasets remain in place and cannot be viewed or extracted. Builders receive trained models and model IP; data owners receive revenue share from each metered training run, while downstream models may be listed as Models as a Service on Harell's marketplace.",
      "asset_disposition": "Embedded / Continuing",
      "corpus_renewal": "Renewable Corpus",
      "source_code_software_asset": "no",
      "disclosed_consideration": {
        "status": "disclosed_revenue_share_structure_amount_not_disclosed",
        "amount": null,
        "currency": null,
        "description": "Harell states that data owners receive a revenue share on every training job executed against their dataset; the percentage, minimum commitment, and cash amount are not disclosed. Model builders can also collect a fee when users run marketplace-listed models."
      },
      "estimated_economics": {
        "status": "not_estimated",
        "amount": null,
        "currency": null,
        "method": null
      },
      "economics_scope": "Per-training-run data-owner revenue share and usage fees for downstream Models as a Service; CoreWeave contract value and exact revenue-share percentages are not disclosed.",
      "rights": {
        "training": "Conditional and permitted inside Harell's controlled environment for authorized builders; raw data may not be viewed or extracted.",
        "post_training": "Fine-tuning is expressly permitted inside the controlled platform; other post-training rights are not disclosed.",
        "inference_retrieval": "Inference is expressly supported inside Harell; raw-record retrieval and extraction are prohibited by the disclosed architecture.",
        "ownership_transfer": "No dataset ownership transfer is disclosed. Raw datasets remain with data providers and do not leave the Harell environment.",
        "exclusivity": "not_disclosed",
        "retention": "Raw data remains inside the Harell environment; contract duration, deletion timing, and trained-model memorization controls are not disclosed.",
        "derived_model": "Model builders keep their trained models and associated intellectual property and may list models under a Models as a Service arrangement.",
        "sublicensing": "Data sublicensing is not disclosed. Marketplace use of trained models is disclosed, but this does not establish a right to sublicense the underlying datasets."
      },
      "restrictions": {
        "governed_access": "Training, fine-tuning, validation, and inference execute inside Harell's controlled platform; raw datasets cannot be viewed or extracted, and compute is attributable and metered per run.",
        "geographic_sovereignty": "not_disclosed",
        "privacy_pii": "The disclosed initial assets are scientific binding datasets. Patient-level content, consent terms, and privacy controls are not disclosed in the reviewed sources.",
        "trade_secret": "Raw proprietary datasets remain inside the platform and are not delivered to model builders; contractual remedies and technical exfiltration testing are not disclosed.",
        "de_identification": "not_disclosed",
        "employee_customer_data": "No employee or ordinary customer-operational corpus is identified; the assets are proprietary experimental scientific datasets from named data partners."
      },
      "refreshability": "Harell describes a growing partner and dataset ecosystem and per-run marketplace model; the cadence and exact update process for each initial dataset are not disclosed.",
      "historical_depth": "The data partners generated substantial proprietary experimental datasets, but the reviewed sources do not quantify collection start dates or longitudinal depth.",
      "decision_outcome_richness": "The initial datasets link biological sequences or structures to measured antibody-antigen affinity and T-cell-receptor binding outcomes, supporting model learning and blinded validation against experimental results.",
      "likely_model_use": "Training, fine-tuning, validating, and serving protein-language, structure-prediction, antibody-engineering, and immune-receptor models without transferring raw proprietary records.",
      "likely_strategic_value": "Provides governed access to scarce experimental biology data whose scale and measured binding outcomes cannot be reproduced from public literature alone, while preserving data-provider control and builder ownership of resulting models.",
      "m_and_a_implications": "Establishes a repeatable data-market structure in which scientific asset owners monetize controlled learning runs without selling or delivering the corpus, potentially increasing strategic value for experimental-data generators and secure model-training platforms.",
      "analysis_boundary": "The named data partners, initial dataset categories, controlled in-platform training and inference, raw-data non-extraction, builder model ownership, multi-year infrastructure agreement, and per-run data-owner revenue share are disclosed. Exact corpus sizes, dataset-specific update cadence, access prices, revenue-share percentages, exclusivity, geography, retention periods, patient-level content, downstream memorization controls, and measured model uplift are not disclosed. Chronology correction: the founder post is dated September 15, 2026 and already describes the same named datasets and learning-rights architecture; the September 23 release concerns supporting infrastructure. The first contract execution date, original provider agreement dates and historical page publication/version timestamps are not independently established. This record is excluded from the September 17, 2026 forecast cohort under the conservative first-disclosure treatment; the correction does not create or remove a registry event.",
      "confidence": "high",
      "scores": {
        "methodology_version": "1.0.0",
        "asset": {
          "score": 84,
          "coverage_pct": 92,
          "rationale": "Strategic Signal analysis. The named scientific corpora combine scarce proprietary experimental measurements with high-value biological outcomes and direct model-training utility; exact historical depth and corpus scale remain partly undisclosed.",
          "factor_ratings": {
            "uniqueness_scarcity": 4,
            "historical_depth": null,
            "decision_outcome_richness": 4,
            "decision_value_density": 4,
            "real_world_grounding": 5,
            "domain_value": 5,
            "refreshability": 3,
            "proprietary_advantage": 4.5,
            "model_learning_usefulness": 4.5,
            "non_replicability": 4,
            "rights_usability": 4
          },
          "factor_evidence": {
            "uniqueness_scarcity": {
              "basis": "analysis",
              "source_ids": [
                "E2",
                "E3"
              ],
              "rationale": "The platform exposes named proprietary antibody and immune-receptor datasets, and Harell states that A-Alpha Bio's scale exceeds publicly available alternatives."
            },
            "decision_outcome_richness": {
              "basis": "analysis",
              "source_ids": [
                "E2",
                "E3"
              ],
              "rationale": "Binding-affinity and receptor-binding measurements connect biological inputs to experimentally observed outcomes suitable for learning and blinded validation."
            },
            "decision_value_density": {
              "basis": "analysis",
              "source_ids": [
                "E2",
                "E3"
              ],
              "rationale": "Individual experimental outcomes can shape high-value protein engineering, immune medicine, and drug-discovery decisions."
            },
            "real_world_grounding": {
              "basis": "disclosed_fact",
              "source_ids": [
                "E2",
                "E3"
              ],
              "rationale": "The assets are experimentally generated affinity, structural, and immune-receptor data rather than synthetic-only benchmarks."
            },
            "domain_value": {
              "basis": "analysis",
              "source_ids": [
                "E1",
                "E3"
              ],
              "rationale": "Protein engineering and immune-driven medicine are economically consequential domains with direct scientific and therapeutic applications."
            },
            "refreshability": {
              "basis": "analysis",
              "source_ids": [
                "E1",
                "E3"
              ],
              "rationale": "Harell describes a growing ecosystem and future datasets, but does not disclose a contractual refresh cadence for the initial assets."
            },
            "proprietary_advantage": {
              "basis": "analysis",
              "source_ids": [
                "E1",
                "E2"
              ],
              "rationale": "Named data partners control experimental assets that builders cannot view or extract and that exceed public alternatives in scale or specificity."
            },
            "model_learning_usefulness": {
              "basis": "disclosed_fact",
              "source_ids": [
                "E1",
                "E3"
              ],
              "rationale": "The arrangement expressly supports model training, fine-tuning, inference, and blinded validation on the proprietary scientific data."
            },
            "non_replicability": {
              "basis": "analysis",
              "source_ids": [
                "E2",
                "E3"
              ],
              "rationale": "Reproducing the precise experimental measurements would require comparable biological platforms, samples, protocols, time, and cost."
            },
            "rights_usability": {
              "basis": "analysis",
              "source_ids": [
                "E1"
              ],
              "rationale": "Controlled in-platform execution makes the data usable for training while preserving raw-data boundaries, though detailed contract terms remain undisclosed."
            }
          }
        },
        "transaction_signal": {
          "score": 71,
          "coverage_pct": 85,
          "rationale": "Strategic Signal analysis. The arrangement cleanly separates controlled data use from raw-data transfer and ties data-owner compensation to training runs, but provides no dollar value, share percentage, bid process, or independent uplift result.",
          "factor_ratings": {
            "disclosed_economics": 2.5,
            "clean_price_discovery": 1.5,
            "data_consideration_separability": 4.5,
            "explicit_model_use": 5,
            "rights_clarity": 4.5,
            "competing_bids": null,
            "strategic_buyer_quality": 3,
            "precedent_value": 5,
            "independent_model_uplift": null
          },
          "factor_evidence": {
            "disclosed_economics": {
              "basis": "disclosed_fact",
              "source_ids": [
                "E1",
                "E4"
              ],
              "rationale": "A per-training-run revenue share and downstream model fees are disclosed, but no dollar amount, percentage, or minimum commitment is provided."
            },
            "clean_price_discovery": {
              "basis": "analysis",
              "source_ids": [
                "E1"
              ],
              "rationale": "Metered runs create an attributable usage base, but the private CoreWeave agreement and undisclosed share terms do not reveal a clean market price."
            },
            "data_consideration_separability": {
              "basis": "analysis",
              "source_ids": [
                "E1",
                "E4"
              ],
              "rationale": "Data-owner compensation is expressly linked to each training job against a dataset, making the data consideration unusually separable from general software subscriptions."
            },
            "explicit_model_use": {
              "basis": "disclosed_fact",
              "source_ids": [
                "E1"
              ],
              "rationale": "Training, fine-tuning, validation, and inference are all expressly identified as supported workloads."
            },
            "rights_clarity": {
              "basis": "analysis",
              "source_ids": [
                "E1"
              ],
              "rationale": "Raw-data non-extraction, controlled execution, builder model ownership, and marketplace use are clear; retention, sublicensing, and exclusivity remain undisclosed."
            },
            "strategic_buyer_quality": {
              "basis": "analysis",
              "source_ids": [
                "E1",
                "E4"
              ],
              "rationale": "The platform targets serious scientific model builders and runs on a scaled AI cloud, but no named frontier model developer is disclosed as a customer."
            },
            "precedent_value": {
              "basis": "analysis",
              "source_ids": [
                "E1",
                "E4"
              ],
              "rationale": "Per-run compensation for in-place proprietary data use is a repeatable template for scientific and other high-value data markets."
            }
          }
        }
      },
      "sources": [
        {
          "type": "primary",
          "publisher": "CoreWeave and Harell Data",
          "title": "Harell Data Selects CoreWeave to Power Its Secure Platform for AI Model Training",
          "url": "https://investors.coreweave.com/news/news-details/2026/Harell-Data-Selects-CoreWeave-to-Power-Its-Secure-Platform-for-AI-Model-Training/default.aspx"
        },
        {
          "type": "primary",
          "publisher": "Harell Data",
          "title": "Data Partners",
          "url": "https://harelldata.com/data-partners/"
        },
        {
          "type": "primary",
          "publisher": "Harell Data",
          "title": "Why we started Harell Data",
          "url": "https://harelldata.com/why-we-started-harell-data/"
        },
        {
          "type": "secondary",
          "publisher": "GeekWire",
          "title": "Adaptive Biotech co-founder raises $15M for new startup to rethink how AI trains on science",
          "url": "https://www.geekwire.com/2026/adaptive-biotech-co-founder-raises-15m-for-new-startup-to-rethink-how-ai-trains-on-science/"
        }
      ],
      "date_added": "2026-09-26",
      "date_last_reviewed": "2026-09-30",
      "announcement_date_basis": "Date displayed on Harell's founder post describing the same two live datasets and governed learning arrangement. This is the earliest source-stated disclosure identified in this review, not a verified agreement execution date or proof of first-ever public availability. The September 23 CoreWeave release is recorded separately as infrastructure news.",
      "event_history": [
        {
          "date": "2026-09-15",
          "event": "Founder post describes the existing controlled learning platform, A-Alpha Bio and Adaptive Biotechnologies datasets, retained data, model IP and per-training compensation.",
          "count_as_new_transaction": true
        },
        {
          "date": "2026-09-23",
          "event": "CoreWeave announces a multi-year infrastructure agreement supporting Harell's platform; no separately evidenced new underlying data-rights arrangement is counted.",
          "count_as_new_transaction": false
        }
      ],
      "publication_review": {
        "status": "human_reviewed",
        "approval_channel": "owner_conversation",
        "reviewer": "Mike Ye",
        "reviewed_at": "2026-09-30T19:34:37.142Z",
        "decision_id": "conversation-20260930-2AEBFCD969B7",
        "packet_sha256": "56a04ca0b550881fe5f41794246d96f8b36ed91f5268a6fc2d7e7ba0c2ba3724",
        "record_sha256": "d4ef8da3e3de774ff279956f7058b3a294d0b60b6726fdc91d8baee8a3799611"
      }
    },
    {
      "transaction_id": "SS-AIDT-2026-0012",
      "slug": "tempus-recursion-oncology-data-license",
      "canonical_path": "/ai-data-transactions/deals/ss-aidt-2026-0012/",
      "announcement_date": "2023-11-03",
      "buyer_ai_developer": [
        "Recursion Pharmaceuticals"
      ],
      "seller_data_owner": [
        "Tempus AI"
      ],
      "transaction_status": "active_amended_six_year_license",
      "industry": "Healthcare, oncology, biotechnology and AI drug discovery",
      "data_category": [
        "De-identified clinical oncology data",
        "Molecular and genomic oncology data",
        "Patient-centric multimodal records"
      ],
      "corpus_description": "Tempus's proprietary de-identified patient-centric multimodal oncology database of clinical and molecular records, originally described by Recursion as more than 20 petabytes, licensed for therapeutic-development AI/ML use subject to unique-record caps and retention limits.",
      "primary_transaction_structure": "Training Rights License",
      "access_structure": "Recursion receives limited licensed access to Tempus's proprietary de-identified clinical and molecular data for therapeutic product development, including model training, improvement, modification and derivative works. Downloaded records are subject to per-time and aggregate unique-record caps and historically a 180-day retention limit. The September 2026 amendment extends the term to six years, removes convenience termination and lowers the remaining unique-record cap.",
      "asset_disposition": "Embedded / Continuing",
      "corpus_renewal": "Renewable Corpus",
      "source_code_software_asset": "no",
      "disclosed_consideration": {
        "status": "disclosed_contractual_license_fees",
        "amount": 42000000,
        "currency": "USD",
        "description": "The 2026 amendment sets $14 million annual license fees on each of the third, fourth and fifth anniversaries, at least $4 million of each payable in cash and the remainder in cash and/or Recursion shares. The original 2023 deal disclosed up to $160 million over five years."
      },
      "estimated_economics": {
        "status": "not_estimated",
        "amount": null,
        "currency": null,
        "method": null
      },
      "economics_scope": "$42 million of amended third-through-fifth anniversary license fees is explicitly disclosed; amounts already paid under the original schedule and the economic effect of the reduced unique-record cap are not restated here.",
      "rights": {
        "training": "Permitted for Recursion and affiliates to develop and train AI/ML models solely for therapeutic product development, subject to the agreement's limits.",
        "post_training": "Permitted to improve and modify Recursion AI/ML models and develop embeddings, training and test data, subject to contractual restrictions.",
        "inference_retrieval": "Licensed data may be accessed and downloaded within stated caps for permitted therapeutic-development uses; unrestricted retrieval or redistribution is not permitted.",
        "ownership_transfer": "No ownership transfer. Tempus retains ownership of Licensed Data and Tempus Materials.",
        "exclusivity": "not_disclosed",
        "retention": "The filed 2023 disclosure states downloaded records may be retained for 180 days; the 2026 amendment does not disclose a change to that limit.",
        "derived_model": "Recursion may create derivative works of its AI/ML models and use, deploy, commercialize or license those models subject to the agreement; underlying licensed data remain Tempus property.",
        "sublicensing": "Certain third-party therapeutic collaborations are allowed under defined conditions, but no general right to sublicense the underlying dataset is established."
      },
      "restrictions": {
        "governed_access": "Use is restricted to therapeutic product development within a secure Recursion environment and subject to download, aggregate unique-record and retention controls.",
        "geographic_sovereignty": "not_disclosed",
        "privacy_pii": "The licensed corpus is disclosed as de-identified clinical and molecular data and governed by healthcare-data restrictions.",
        "trade_secret": "Tempus retains ownership of its proprietary database and related materials; permitted uses do not authorize unrestricted reproduction or transfer.",
        "de_identification": "Licensed records are de-identified; the agreement references HIPAA-aligned de-identification treatment for relevant data.",
        "employee_customer_data": "The asset contains patient-centric clinical and molecular oncology records rather than ordinary employee or customer-operational data."
      },
      "refreshability": "The arrangement spans six years and grants continuing licensed access subject to aggregate unique-record limits; the exact refresh cadence and amended unique-record count are not disclosed.",
      "historical_depth": "Tempus's oncology database aggregates clinical and molecular patient records accumulated over time; the exact date range and longitudinal completeness are not disclosed in the amendment.",
      "decision_outcome_richness": "The corpus supports therapeutic hypotheses, biomarker strategies, patient stratification and model learning from linked real-world clinical and molecular observations.",
      "likely_model_use": "Training, improving, modifying and creating derivative works of Recursion causal AI/ML models for therapeutic product development, plus biomarker and patient-stratification work.",
      "likely_strategic_value": "Provides Recursion with large-scale patient-centric multimodal oncology data that complements its proprietary interventional biology and chemistry data and directly supports AI-driven drug discovery.",
      "m_and_a_implications": "Demonstrates that a proprietary healthcare-data asset can command large, separately stated multi-year licensing economics when explicit AI-training and derivative-model rights are granted without selling the underlying corpus.",
      "analysis_boundary": "Primary SEC filings establish the proprietary corpus, explicit AI/ML training and derivative-work rights, original and amended economics, six-year term, removal of convenience termination, lower aggregate unique-record cap, ownership boundaries and de-identification. The amended exact record cap, exclusivity, current refresh cadence, geographic restrictions and independent model-uplift evidence are not disclosed.",
      "confidence": "high",
      "scores": {
        "methodology_version": "1.0.0",
        "asset": {
          "score": 94,
          "coverage_pct": 100,
          "rationale": "Strategic Signal analysis. Tempus's large proprietary multimodal oncology corpus combines real-world clinical and molecular observations with explicit AI-training utility and strong therapeutic decision relevance.",
          "factor_ratings": {
            "uniqueness_scarcity": 4.5,
            "historical_depth": 4.5,
            "decision_outcome_richness": 4.5,
            "decision_value_density": 5,
            "real_world_grounding": 5,
            "domain_value": 5,
            "refreshability": 4.5,
            "proprietary_advantage": 4.5,
            "model_learning_usefulness": 5,
            "non_replicability": 4,
            "rights_usability": 4.5
          },
          "factor_evidence": {
            "uniqueness_scarcity": {
              "basis": "analysis",
              "source_ids": [
                "E2",
                "E3"
              ],
              "rationale": "Evidence supports uniqueness scarcity as a material characteristic of the Tempus oncology corpus and its AI-training use, with remaining limits preserved explicitly."
            },
            "historical_depth": {
              "basis": "analysis",
              "source_ids": [
                "E2",
                "E3"
              ],
              "rationale": "Evidence supports historical depth as a material characteristic of the Tempus oncology corpus and its AI-training use, with remaining limits preserved explicitly."
            },
            "decision_outcome_richness": {
              "basis": "analysis",
              "source_ids": [
                "E2",
                "E3"
              ],
              "rationale": "Evidence supports decision outcome richness as a material characteristic of the Tempus oncology corpus and its AI-training use, with remaining limits preserved explicitly."
            },
            "decision_value_density": {
              "basis": "analysis",
              "source_ids": [
                "E2",
                "E3"
              ],
              "rationale": "Evidence supports decision value density as a material characteristic of the Tempus oncology corpus and its AI-training use, with remaining limits preserved explicitly."
            },
            "real_world_grounding": {
              "basis": "disclosed_fact",
              "source_ids": [
                "E2",
                "E3"
              ],
              "rationale": "Evidence supports real world grounding as a material characteristic of the Tempus oncology corpus and its AI-training use, with remaining limits preserved explicitly."
            },
            "domain_value": {
              "basis": "analysis",
              "source_ids": [
                "E2",
                "E3"
              ],
              "rationale": "Evidence supports domain value as a material characteristic of the Tempus oncology corpus and its AI-training use, with remaining limits preserved explicitly."
            },
            "refreshability": {
              "basis": "analysis",
              "source_ids": [
                "E2",
                "E3"
              ],
              "rationale": "Evidence supports refreshability as a material characteristic of the Tempus oncology corpus and its AI-training use, with remaining limits preserved explicitly."
            },
            "proprietary_advantage": {
              "basis": "analysis",
              "source_ids": [
                "E2",
                "E3"
              ],
              "rationale": "Evidence supports proprietary advantage as a material characteristic of the Tempus oncology corpus and its AI-training use, with remaining limits preserved explicitly."
            },
            "model_learning_usefulness": {
              "basis": "disclosed_fact",
              "source_ids": [
                "E2",
                "E3"
              ],
              "rationale": "Evidence supports model learning usefulness as a material characteristic of the Tempus oncology corpus and its AI-training use, with remaining limits preserved explicitly."
            },
            "non_replicability": {
              "basis": "analysis",
              "source_ids": [
                "E2",
                "E3"
              ],
              "rationale": "Evidence supports non replicability as a material characteristic of the Tempus oncology corpus and its AI-training use, with remaining limits preserved explicitly."
            },
            "rights_usability": {
              "basis": "analysis",
              "source_ids": [
                "E2",
                "E3"
              ],
              "rationale": "Evidence supports rights usability as a material characteristic of the Tempus oncology corpus and its AI-training use, with remaining limits preserved explicitly."
            }
          }
        },
        "transaction_signal": {
          "score": 96,
          "coverage_pct": 85,
          "rationale": "Strategic Signal analysis. The agreement has unusually explicit model-use rights and separately disclosed multi-year license economics, strengthened by a filed 2026 amendment; no competitive process or independent uplift study is disclosed.",
          "factor_ratings": {
            "disclosed_economics": 5,
            "clean_price_discovery": 4,
            "data_consideration_separability": 5,
            "explicit_model_use": 5,
            "rights_clarity": 5,
            "competing_bids": null,
            "strategic_buyer_quality": 4.5,
            "precedent_value": 5,
            "independent_model_uplift": null
          },
          "factor_evidence": {
            "disclosed_economics": {
              "basis": "disclosed_fact",
              "source_ids": [
                "E1",
                "E2",
                "E3"
              ],
              "rationale": "The filed agreement and amendment provide sufficient evidence to rate disclosed economics for this separately priced AI data-rights arrangement."
            },
            "clean_price_discovery": {
              "basis": "analysis",
              "source_ids": [
                "E1",
                "E2",
                "E3"
              ],
              "rationale": "The filed agreement and amendment provide sufficient evidence to rate clean price discovery for this separately priced AI data-rights arrangement."
            },
            "data_consideration_separability": {
              "basis": "analysis",
              "source_ids": [
                "E1",
                "E2",
                "E3"
              ],
              "rationale": "The filed agreement and amendment provide sufficient evidence to rate data consideration separability for this separately priced AI data-rights arrangement."
            },
            "explicit_model_use": {
              "basis": "disclosed_fact",
              "source_ids": [
                "E1",
                "E2",
                "E3"
              ],
              "rationale": "The filed agreement and amendment provide sufficient evidence to rate explicit model use for this separately priced AI data-rights arrangement."
            },
            "rights_clarity": {
              "basis": "disclosed_fact",
              "source_ids": [
                "E1",
                "E2",
                "E3"
              ],
              "rationale": "The filed agreement and amendment provide sufficient evidence to rate rights clarity for this separately priced AI data-rights arrangement."
            },
            "strategic_buyer_quality": {
              "basis": "analysis",
              "source_ids": [
                "E1",
                "E2",
                "E3"
              ],
              "rationale": "The filed agreement and amendment provide sufficient evidence to rate strategic buyer quality for this separately priced AI data-rights arrangement."
            },
            "precedent_value": {
              "basis": "analysis",
              "source_ids": [
                "E1",
                "E2",
                "E3"
              ],
              "rationale": "The filed agreement and amendment provide sufficient evidence to rate precedent value for this separately priced AI data-rights arrangement."
            }
          }
        }
      },
      "sources": [
        {
          "type": "primary",
          "publisher": "Recursion Pharmaceuticals / SEC",
          "title": "September 15, 2026 Form 8-K — Tempus Master Agreement Amendment",
          "url": "https://www.sec.gov/Archives/edgar/data/1601830/000160183026000112/rxrx-20260915.htm"
        },
        {
          "type": "primary",
          "publisher": "Recursion Pharmaceuticals / SEC",
          "title": "November 9, 2023 Form 8-K — Tempus Master Agreement",
          "url": "https://www.sec.gov/Archives/edgar/data/1601830/000160183023000068/rxrx-20231109.htm"
        },
        {
          "type": "primary",
          "publisher": "Recursion Pharmaceuticals / SEC",
          "title": "Tempus Master Agreement Exhibit",
          "url": "https://www.sec.gov/Archives/edgar/data/1601830/000160183023000069/exhibit104-masteragreement.htm"
        },
        {
          "type": "primary",
          "publisher": "Recursion Pharmaceuticals",
          "title": "Recursion Announces Data Collaboration Deal with Tempus",
          "url": "https://www.sec.gov/Archives/edgar/data/1601830/000160183023000068/exhibit992-q0323.htm"
        }
      ],
      "date_added": "2026-09-29",
      "date_last_reviewed": "2026-09-29",
      "publication_review": {
        "status": "human_reviewed",
        "approval_channel": "owner_conversation",
        "reviewer": "Mike Ye",
        "reviewed_at": "2026-09-29T18:02:23.696Z",
        "decision_id": "conversation-20260929-1C674D15A8B4",
        "packet_sha256": "23a5d5b381767ab928eeb336bd5bad568134f4a95c5968247b856aa36360be0e",
        "record_sha256": "c4b83271b0cdc8d5f6cbf96f6f12176eaaba7ff0ca039c4dd314f7dd504b3461"
      }
    },
    {
      "transaction_id": "SS-AIDT-2026-0013",
      "slug": "guideai-vrad-exclusive-imaging-ai-rights",
      "canonical_path": "/ai-data-transactions/deals/ss-aidt-2026-0013/",
      "announcement_date": "2026-09-28",
      "buyer_ai_developer": [
        "GuideAI Health"
      ],
      "seller_data_owner": [
        "vRad (Virtual Radiologic)"
      ],
      "transaction_status": "active_renewed_exclusive_data_agreement",
      "industry": "Healthcare, medical imaging and clinical AI",
      "data_category": [
        "De-identified radiology imaging data",
        "Routine CT imaging",
        "Vascular disease imaging corpus"
      ],
      "corpus_description": "De-identified imaging data generated across vRad's national network of more than 2,100 U.S. hospitals and healthcare facilities, made exclusively available to GuideAI for development, training and validation of AI models for vascular-disease detection and characterization.",
      "primary_transaction_structure": "Training Rights License",
      "access_structure": "Under the renewed data exclusivity agreement, GuideAI retains exclusive rights to use de-identified imaging generated across vRad's network to develop AI models. The issuer states that the dataset provides the foundation to train and validate GuideAI algorithms; detailed delivery, retention and downstream model-rights mechanics are not disclosed.",
      "asset_disposition": "Embedded / Continuing",
      "corpus_renewal": "Renewable Corpus",
      "source_code_software_asset": "no",
      "disclosed_consideration": {
        "status": "not_disclosed",
        "amount": null,
        "currency": null,
        "description": "The renewed agreement's financial consideration, minimum commitments and pricing formula are not disclosed."
      },
      "estimated_economics": {
        "status": "not_estimated",
        "amount": null,
        "currency": null,
        "method": null
      },
      "economics_scope": "No agreement value, per-study pricing, revenue share, equity component or other economic terms are disclosed.",
      "rights": {
        "training": "Permitted and expressly contemplated: GuideAI states that the imaging corpus provides a foundation to train its vascular-disease algorithms.",
        "post_training": "Validation of GuideAI algorithms is expressly contemplated; fine-tuning and other post-training rights are not separately described.",
        "inference_retrieval": "GuideAI has exclusive access to use the de-identified imaging data for AI-model development; record-level retrieval mechanics are not disclosed.",
        "ownership_transfer": "No ownership transfer of vRad imaging data is disclosed.",
        "exclusivity": "GuideAI states that it retains exclusive rights to develop AI models using the de-identified imaging data generated across vRad's network.",
        "retention": "not_disclosed",
        "derived_model": "GuideAI is permitted to develop AI models from the licensed imaging corpus; detailed ownership language for resulting models is not disclosed.",
        "sublicensing": "not_disclosed"
      },
      "restrictions": {
        "governed_access": "Access is limited to the renewed vRad data partnership and AI-model development scope; detailed technical access controls are not disclosed.",
        "geographic_sovereignty": "The disclosed source network consists of U.S. hospitals and healthcare facilities; processing-location restrictions are not disclosed.",
        "privacy_pii": "The licensed imaging data are disclosed as de-identified, with healthcare privacy and data-protection compliance identified as relevant risks.",
        "trade_secret": "The dataset is treated as an exclusive commercial data asset; technical confidentiality controls are not disclosed.",
        "de_identification": "The announcement expressly describes the imaging data as de-identified.",
        "employee_customer_data": "The asset consists of de-identified patient imaging generated in clinical care rather than employee or ordinary commercial customer data."
      },
      "refreshability": "The renewed partnership covers imaging generated across an ongoing network of more than 2,100 hospitals and healthcare facilities, indicating a renewable flow; exact refresh cadence is not disclosed.",
      "historical_depth": "The arrangement is a renewal of an existing data partnership, but the reviewed source does not disclose the original start date, number of studies or years of imaging history.",
      "decision_outcome_richness": "The imaging corpus contains clinically meaningful disease signals used for detection and characterization, but the disclosure does not establish linked treatment interventions and outcomes.",
      "likely_model_use": "Training, validating and expanding GuideAI models for detection and characterization of peripheral arterial and other vascular diseases from routine CT imaging.",
      "likely_strategic_value": "Exclusive access to a large, diverse, real-world national imaging corpus can improve model coverage and create a proprietary data advantage in vascular-disease AI.",
      "m_and_a_implications": "Shows how a large clinical-network data owner can grant exclusive AI-model development rights without transferring ownership of the underlying corpus, potentially creating strategic scarcity for specialized healthcare AI developers.",
      "analysis_boundary": "The issuer release establishes the renewed exclusive rights, de-identified imaging scope, more than 2,100-hospital network, approximately 500-radiologist workflow, and training/validation purpose. Economics, term length, exact corpus size, delivery architecture, retention, derived-model ownership, sublicensing, geographic processing controls and independent model-uplift evidence are not disclosed.",
      "confidence": "high",
      "scores": {
        "methodology_version": "1.0.0",
        "asset": {
          "score": 93,
          "coverage_pct": 100,
          "rationale": "Strategic Signal analysis. The exclusive national imaging corpus is large, real-world, renewable and directly useful for clinical-model training, though exact longitudinal depth and patient-level outcome linkage are not disclosed.",
          "factor_ratings": {
            "uniqueness_scarcity": 4.5,
            "historical_depth": 4,
            "decision_outcome_richness": 4,
            "decision_value_density": 5,
            "real_world_grounding": 5,
            "domain_value": 5,
            "refreshability": 5,
            "proprietary_advantage": 5,
            "model_learning_usefulness": 5,
            "non_replicability": 4.5,
            "rights_usability": 4
          },
          "factor_evidence": {
            "uniqueness_scarcity": {
              "basis": "analysis",
              "source_ids": [
                "E1"
              ],
              "rationale": "The issuer disclosure provides sufficient evidence to rate uniqueness scarcity for the exclusive vRad imaging corpus while preserving undisclosed limits."
            },
            "historical_depth": {
              "basis": "analysis",
              "source_ids": [
                "E1"
              ],
              "rationale": "The issuer disclosure provides sufficient evidence to rate historical depth for the exclusive vRad imaging corpus while preserving undisclosed limits."
            },
            "decision_outcome_richness": {
              "basis": "analysis",
              "source_ids": [
                "E1"
              ],
              "rationale": "The issuer disclosure provides sufficient evidence to rate decision outcome richness for the exclusive vRad imaging corpus while preserving undisclosed limits."
            },
            "decision_value_density": {
              "basis": "analysis",
              "source_ids": [
                "E1"
              ],
              "rationale": "The issuer disclosure provides sufficient evidence to rate decision value density for the exclusive vRad imaging corpus while preserving undisclosed limits."
            },
            "real_world_grounding": {
              "basis": "disclosed_fact",
              "source_ids": [
                "E1"
              ],
              "rationale": "The issuer disclosure provides sufficient evidence to rate real world grounding for the exclusive vRad imaging corpus while preserving undisclosed limits."
            },
            "domain_value": {
              "basis": "analysis",
              "source_ids": [
                "E1"
              ],
              "rationale": "The issuer disclosure provides sufficient evidence to rate domain value for the exclusive vRad imaging corpus while preserving undisclosed limits."
            },
            "refreshability": {
              "basis": "disclosed_fact",
              "source_ids": [
                "E1"
              ],
              "rationale": "The issuer disclosure provides sufficient evidence to rate refreshability for the exclusive vRad imaging corpus while preserving undisclosed limits."
            },
            "proprietary_advantage": {
              "basis": "disclosed_fact",
              "source_ids": [
                "E1"
              ],
              "rationale": "The issuer disclosure provides sufficient evidence to rate proprietary advantage for the exclusive vRad imaging corpus while preserving undisclosed limits."
            },
            "model_learning_usefulness": {
              "basis": "disclosed_fact",
              "source_ids": [
                "E1"
              ],
              "rationale": "The issuer disclosure provides sufficient evidence to rate model learning usefulness for the exclusive vRad imaging corpus while preserving undisclosed limits."
            },
            "non_replicability": {
              "basis": "analysis",
              "source_ids": [
                "E1"
              ],
              "rationale": "The issuer disclosure provides sufficient evidence to rate non replicability for the exclusive vRad imaging corpus while preserving undisclosed limits."
            },
            "rights_usability": {
              "basis": "analysis",
              "source_ids": [
                "E1"
              ],
              "rationale": "The issuer disclosure provides sufficient evidence to rate rights usability for the exclusive vRad imaging corpus while preserving undisclosed limits."
            }
          }
        },
        "transaction_signal": {
          "score": 55,
          "coverage_pct": 85,
          "rationale": "Strategic Signal analysis. Exclusive model-development rights are unusually explicit, but economics and market price discovery are absent, materially limiting the transaction signal despite strong AI-use and rights clarity.",
          "factor_ratings": {
            "disclosed_economics": 0,
            "clean_price_discovery": 0,
            "data_consideration_separability": 4,
            "explicit_model_use": 5,
            "rights_clarity": 4.5,
            "competing_bids": null,
            "strategic_buyer_quality": 3.5,
            "precedent_value": 4.5,
            "independent_model_uplift": null
          },
          "factor_evidence": {
            "disclosed_economics": {
              "basis": "disclosed_fact",
              "source_ids": [
                "E1"
              ],
              "rationale": "The renewed agreement provides sufficient evidence to rate disclosed economics for this exclusive AI-model development data-rights arrangement."
            },
            "clean_price_discovery": {
              "basis": "analysis",
              "source_ids": [
                "E1"
              ],
              "rationale": "The renewed agreement provides sufficient evidence to rate clean price discovery for this exclusive AI-model development data-rights arrangement."
            },
            "data_consideration_separability": {
              "basis": "analysis",
              "source_ids": [
                "E1"
              ],
              "rationale": "The renewed agreement provides sufficient evidence to rate data consideration separability for this exclusive AI-model development data-rights arrangement."
            },
            "explicit_model_use": {
              "basis": "disclosed_fact",
              "source_ids": [
                "E1"
              ],
              "rationale": "The renewed agreement provides sufficient evidence to rate explicit model use for this exclusive AI-model development data-rights arrangement."
            },
            "rights_clarity": {
              "basis": "disclosed_fact",
              "source_ids": [
                "E1"
              ],
              "rationale": "The renewed agreement provides sufficient evidence to rate rights clarity for this exclusive AI-model development data-rights arrangement."
            },
            "strategic_buyer_quality": {
              "basis": "analysis",
              "source_ids": [
                "E1"
              ],
              "rationale": "The renewed agreement provides sufficient evidence to rate strategic buyer quality for this exclusive AI-model development data-rights arrangement."
            },
            "precedent_value": {
              "basis": "analysis",
              "source_ids": [
                "E1"
              ],
              "rationale": "The renewed agreement provides sufficient evidence to rate precedent value for this exclusive AI-model development data-rights arrangement."
            }
          }
        }
      },
      "sources": [
        {
          "type": "primary_syndicated",
          "publisher": "GuideAI Health",
          "title": "GuideAI Health Corp. and vRad Renew Data Exclusivity Agreement to Develop AI Models",
          "url": "https://www.globenewswire.com/news-release/2026/09/28/3369827/0/en/guideai-health-corp-and-vrad-renew-data-exclusivity-agreement-to-develop-ai-models-for-the-detection-and-characterization-of-multiple-vascular-diseases.html"
        }
      ],
      "date_added": "2026-09-29",
      "date_last_reviewed": "2026-09-29",
      "publication_review": {
        "status": "human_reviewed",
        "approval_channel": "owner_conversation",
        "reviewer": "Mike Ye",
        "reviewed_at": "2026-09-29T18:02:23.716Z",
        "decision_id": "conversation-20260929-22F00D6C2ABE",
        "packet_sha256": "8534eef99edbad6cbffa7a0a9b29177561506eb77207ffc61f8d68ac30f125f4",
        "record_sha256": "498d385cb70c4250f97bb7a0856be2a422c066b9f4d49cd94f96f875fa453cc3"
      }
    },
    {
      "transaction_id": "SS-AIDT-2026-0014",
      "slug": "onemednet-precision-medicine-rwd-training-license",
      "canonical_path": "/ai-data-transactions/deals/ss-aidt-2026-0014/",
      "announcement_date": "2026-09-30",
      "buyer_ai_developer": [
        "Undisclosed commercially scaled AI-enabled precision-medicine company"
      ],
      "seller_data_owner": [
        "OneMedNet"
      ],
      "transaction_status": "active_multiyear_data_license",
      "industry": "Healthcare, precision medicine and clinical AI",
      "data_category": [
        "De-identified multimodal clinical real-world data",
        "Medical imaging",
        "Longitudinal clinical records and patient journeys"
      ],
      "corpus_description": "De-identified, multimodal, longitudinal clinical real-world data sourced directly through OneMedNet's network of more than 2,300 healthcare partner sites, including medical imaging and longitudinal clinical journeys, licensed for AI model training and incorporation into a commercial derivative dataset.",
      "primary_transaction_structure": "Training Rights License",
      "access_structure": "Under a seven-figure multiyear agreement, OneMedNet will deliver de-identified multimodal longitudinal clinical RWD through its iRWD platform to an undisclosed commercially scaled AI-enabled precision-medicine company. The customer intends to train clinical and diagnostic AI models and create a licensable derivative dataset combining its molecular data with OneMedNet imaging and longitudinal clinical journeys. OneMedNet retains source-data rights and receives incremental annual sublicensing revenue when the derivative solution is licensed to end customers.",
      "asset_disposition": "Embedded / Continuing",
      "corpus_renewal": "Renewable Corpus",
      "source_code_software_asset": "no",
      "disclosed_consideration": {
        "status": "reported_range_only",
        "amount": null,
        "currency": null,
        "description": "Seven-figure multiyear agreement, plus incremental annual sublicensing revenue to OneMedNet for each downstream end-customer license; exact contract value, currency, payment form, annual schedule and per-license revenue share are not disclosed in the reviewed issuer announcement."
      },
      "estimated_economics": {
        "status": "not_estimated",
        "amount": null,
        "currency": null,
        "method": null
      },
      "economics_scope": "The disclosed seven-figure amount applies to the multiyear data license. OneMedNet separately states that it receives recurring sublicensing revenue on each downstream license of the derivative product and that the agreement includes an option to expand data volume and contract value; exact values are not disclosed.",
      "rights": {
        "training": "Permitted and explicit: the customer intends to use OneMedNet data to train AI-enabled clinical and diagnostic models.",
        "post_training": "not_disclosed",
        "inference_retrieval": "The customer receives licensed access to the data through OneMedNet's iRWD platform; detailed retrieval mechanics and inference-only permissions are not separately disclosed.",
        "ownership_transfer": "Prohibited for the source data: OneMedNet expressly states that it retains source-data rights.",
        "exclusivity": "not_disclosed",
        "retention": "not_disclosed",
        "derived_model": "Permitted in substance: the customer may train clinical and diagnostic AI models and create a licensable derivative dataset combining its molecular data with OneMedNet imaging and longitudinal clinical journeys; ownership terms for resulting models are not separately disclosed.",
        "sublicensing": "Conditional and explicit for the derivative product: the customer may license the derivative dataset to end customers, OneMedNet receives incremental annual sublicensing revenue, and end customers may not relicense or resell the underlying OneMedNet data."
      },
      "restrictions": {
        "governed_access": "Data are delivered under a multiyear commercial license through OneMedNet's iRWD platform; OneMedNet retains source-data rights and derivative end customers may not relicense or resell the underlying OneMedNet data.",
        "geographic_sovereignty": "not_disclosed",
        "privacy_pii": "The licensed clinical data are disclosed as de-identified; the release does not disclose patient-level consent terms or the detailed privacy architecture.",
        "trade_secret": "The source corpus and derivative commercialization rights are governed by the commercial license; additional confidentiality terms are not disclosed.",
        "de_identification": "OneMedNet states that data are de-identified at source and designed to meet standards for AI training and life-sciences commercial use.",
        "employee_customer_data": "The asset consists of de-identified healthcare patient data and clinical records rather than employee or ordinary commercial customer data."
      },
      "refreshability": "The arrangement is multiyear, uses a live network of more than 2,300 healthcare partner sites, includes recurring downstream licensing and provides an option to expand data volume, supporting a renewable rather than one-time corpus.",
      "historical_depth": "OneMedNet describes longitudinal clinical journeys and a network encompassing more than 90 million patient journeys and 270 million studies, but the specific licensed subset's years of history and patient count are not disclosed.",
      "decision_outcome_richness": "The licensed corpus links imaging and longitudinal clinical records across patient journeys and is intended for clinical and diagnostic model development. The reviewed disclosure does not establish accessible context-intervention-outcome linkage sufficient to infer canonical Human Response H2 or H3.",
      "likely_model_use": "Training clinical and diagnostic AI models and constructing a commercial derivative dataset that combines OneMedNet imaging and longitudinal clinical journeys with the customer's molecular datasets.",
      "likely_strategic_value": "The agreement gives the AI customer recurring access to direct-from-source multimodal clinical data for model learning while creating a derivative-data product with recurring downstream monetization, making both the corpus and the rights architecture strategically valuable.",
      "m_and_a_implications": "The transaction provides observable evidence that proprietary healthcare data can support a layered monetization model: multiyear training-license economics plus participation in downstream derivative-data revenue while the source owner retains the underlying data rights.",
      "analysis_boundary": "The issuer release establishes a seven-figure multiyear license, explicit AI-model training, derivative-dataset commercialization, recurring downstream sublicensing revenue, retained source-data rights, de-identification, more than 2,300 healthcare partner sites, and an option to expand volume and value. The counterparty name, exact contract value, annual payment schedule, sublicensing rate, licensed subset size, historical years, exclusivity, retention, post-training and evaluation rights, geographic processing terms, patient consent details, model ownership and independent model-uplift evidence are not disclosed. Longitudinal patient journeys do not by themselves establish Human Response H2/H3 linkage. Currency correction: the reviewed issuer announcement uses 'seven figure' without explicitly identifying a currency or settlement form. The earlier USD field is withdrawn in this successor record; no numeric range, cash realization or currency conversion is inferred.",
      "confidence": "high",
      "scores": {
        "methodology_version": "1.0.0",
        "asset": {
          "score": 94,
          "coverage_pct": 100,
          "rationale": "Strategic Signal analysis. OneMedNet's direct-from-source multimodal clinical corpus is large, longitudinal, renewable, real-world and explicitly useful for AI training, with strong scarcity and derivative-product utility; exact licensed-subset depth and intervention-to-outcome linkage remain undisclosed.",
          "factor_ratings": {
            "uniqueness_scarcity": 4.5,
            "historical_depth": 4,
            "decision_outcome_richness": 4,
            "decision_value_density": 5,
            "real_world_grounding": 5,
            "domain_value": 5,
            "refreshability": 5,
            "proprietary_advantage": 5,
            "model_learning_usefulness": 5,
            "non_replicability": 4.5,
            "rights_usability": 4.5
          },
          "factor_evidence": {
            "uniqueness_scarcity": {
              "basis": "analysis",
              "source_ids": [
                "E1"
              ],
              "rationale": "The direct-from-source network, multimodal imaging plus clinical journeys, and scale across more than 2,300 healthcare sites support a high scarcity rating while the exact licensed subset remains undisclosed."
            },
            "historical_depth": {
              "basis": "analysis",
              "source_ids": [
                "E1"
              ],
              "rationale": "Longitudinal clinical journeys and the broader platform scale support substantial history, but the announcement does not disclose the licensed subset's exact years of coverage."
            },
            "decision_outcome_richness": {
              "basis": "analysis",
              "source_ids": [
                "E1"
              ],
              "rationale": "The corpus includes longitudinal clinical journeys and imaging relevant to diagnosis, but the disclosure does not establish treatment-intervention and outcome linkage for canonical Human Response staging."
            },
            "decision_value_density": {
              "basis": "analysis",
              "source_ids": [
                "E1"
              ],
              "rationale": "Clinical and diagnostic model development operates in a high-consequence healthcare domain where the licensed data can influence valuable model-development decisions."
            },
            "real_world_grounding": {
              "basis": "disclosed_fact",
              "source_ids": [
                "E1"
              ],
              "rationale": "OneMedNet states the data are first-party, direct-from-source real-world clinical data drawn from a live network of more than 2,300 healthcare partner sites."
            },
            "domain_value": {
              "basis": "analysis",
              "source_ids": [
                "E1"
              ],
              "rationale": "Precision medicine, clinical diagnostics and life-sciences data products are economically valuable applications, supporting a maximum domain-value rating."
            },
            "refreshability": {
              "basis": "disclosed_fact",
              "source_ids": [
                "E1"
              ],
              "rationale": "The multiyear agreement, live provider network, expansion option and recurring downstream licenses establish a renewable data relationship rather than a one-time corpus."
            },
            "proprietary_advantage": {
              "basis": "disclosed_fact",
              "source_ids": [
                "E1"
              ],
              "rationale": "OneMedNet retains source-data rights while licensing a direct-from-source corpus that the customer will combine with proprietary molecular data into a derivative product."
            },
            "model_learning_usefulness": {
              "basis": "disclosed_fact",
              "source_ids": [
                "E1"
              ],
              "rationale": "The customer explicitly intends to use the licensed data to train AI-enabled clinical and diagnostic models."
            },
            "non_replicability": {
              "basis": "analysis",
              "source_ids": [
                "E1"
              ],
              "rationale": "Replicating a de-identified multimodal corpus across more than 2,300 direct healthcare sites would require substantial access, integration and compliance infrastructure."
            },
            "rights_usability": {
              "basis": "analysis",
              "source_ids": [
                "E1"
              ],
              "rationale": "Training and derivative commercialization are expressly permitted and downstream sublicensing economics are defined, although retention, exclusivity and model-ownership details remain undisclosed."
            }
          }
        },
        "transaction_signal": {
          "score": 78,
          "coverage_pct": 85,
          "rationale": "Strategic Signal analysis. The transaction discloses a seven-figure multiyear data license, explicit model-training use, cleanly separable data consideration, derivative commercialization and recurring sublicensing economics. The exact price, counterparty identity, competitive process and independent model-uplift evidence remain undisclosed.",
          "factor_ratings": {
            "disclosed_economics": 3,
            "clean_price_discovery": 2,
            "data_consideration_separability": 5,
            "explicit_model_use": 5,
            "rights_clarity": 4.5,
            "competing_bids": null,
            "strategic_buyer_quality": 3.5,
            "precedent_value": 5,
            "independent_model_uplift": null
          },
          "factor_evidence": {
            "disclosed_economics": {
              "basis": "disclosed_fact",
              "source_ids": [
                "E1"
              ],
              "rationale": "The issuer discloses a seven-figure multiyear data license plus recurring annual sublicensing revenue, but not an exact contract amount or payment schedule."
            },
            "clean_price_discovery": {
              "basis": "analysis",
              "source_ids": [
                "E1"
              ],
              "rationale": "The disclosure isolates the data-license economics from the derivative sublicensing stream, but gives only a seven-figure range and no competitive bidding evidence."
            },
            "data_consideration_separability": {
              "basis": "disclosed_fact",
              "source_ids": [
                "E1"
              ],
              "rationale": "The announced consideration is explicitly for the multiyear data license, with a separately described downstream sublicensing revenue stream."
            },
            "explicit_model_use": {
              "basis": "disclosed_fact",
              "source_ids": [
                "E1"
              ],
              "rationale": "The customer explicitly intends to train AI-enabled clinical and diagnostic models using the licensed OneMedNet data."
            },
            "rights_clarity": {
              "basis": "disclosed_fact",
              "source_ids": [
                "E1"
              ],
              "rationale": "The announcement clearly establishes training, derivative commercialization, retained source-data rights and downstream restrictions, while several secondary rights remain undisclosed."
            },
            "strategic_buyer_quality": {
              "basis": "analysis",
              "source_ids": [
                "E1"
              ],
              "rationale": "The counterparty is unnamed but described as a commercially scaled AI-enabled precision-medicine company with an established life-sciences customer base."
            },
            "precedent_value": {
              "basis": "analysis",
              "source_ids": [
                "E1"
              ],
              "rationale": "A multiyear training license plus recurring participation in derivative-data sublicensing provides a repeatable precedent for source-data owners monetizing downstream AI data products."
            }
          }
        }
      },
      "sources": [
        {
          "type": "primary_syndicated",
          "publisher": "OneMedNet",
          "title": "OneMedNet Signs Seven Figure Multiyear Agreement to Power a Partner Sold Derivative Data Solution",
          "url": "https://www.globenewswire.com/news-release/2026/09/30/3371902/0/en/onemednet-signs-seven-figure-multiyear-agreement-to-power-a-partner-sold-derivative-data-solution.html"
        }
      ],
      "date_added": "2026-09-30",
      "date_last_reviewed": "2026-09-30",
      "publication_review": {
        "status": "human_reviewed",
        "approval_channel": "owner_conversation",
        "reviewer": "Mike Ye",
        "reviewed_at": "2026-09-30T19:34:37.164Z",
        "decision_id": "conversation-20260930-1BE00DE10512",
        "packet_sha256": "ab15069afc2980e4e693ed1554c3ee7c9091af7fc4fa43e2021a8c482500df80",
        "record_sha256": "780e019ff810f2e9758e4fa33c443b9018cd250c1647c418e50dc08d66ce1c57"
      }
    },
    {
      "transaction_id": "SS-AIDT-2026-0015",
      "slug": "antibody-developability-consortium-federated-data-model",
      "canonical_path": "/ai-data-transactions/deals/ss-aidt-2026-0015/",
      "announcement_date": "2026-09-29",
      "buyer_ai_developer": [
        "Ginkgo Datapoints",
        "Apheris",
        "AbbVie",
        "argenx",
        "Lundbeck",
        "Takeda"
      ],
      "seller_data_owner": [
        "AbbVie",
        "argenx",
        "Lundbeck",
        "Takeda"
      ],
      "transaction_status": "active_consortium_initial_dataset_expected_early_2027",
      "industry": "Biopharmaceutical research, antibody engineering, and scientific AI",
      "data_category": [
        "Proprietary antibody sequences",
        "Standardized antibody developability assay data",
        "Federated member model outputs"
      ],
      "corpus_description": "A planned standardized dataset targeting 10,000 antibodies, combining proprietary antibody sequences contributed by AbbVie, argenx, Lundbeck, and Takeda with Ginkgo-generated developability assay measurements. Raw member sequences remain protected through Apheris federated infrastructure.",
      "primary_transaction_structure": "Federated / Controlled Learning Rights",
      "access_structure": "Members contribute proprietary sequences; Ginkgo generates standardized assay data and trains a foundation model. Apheris federated infrastructure lets members train, benchmark, and refine models across consortium data without exposing raw proprietary sequences, and members can fine-tune models in their own environments.",
      "asset_disposition": "Embedded / Continuing",
      "corpus_renewal": "Renewable Corpus",
      "source_code_software_asset": "no",
      "disclosed_consideration": {
        "status": "not_disclosed",
        "amount": null,
        "currency": null,
        "description": "No membership fee, cash consideration, revenue share, minimum commitment, or asset valuation is disclosed. Members contribute proprietary sequences and receive governed access to consortium data, models, and derivatives for internal use."
      },
      "estimated_economics": {
        "status": "not_estimated",
        "amount": null,
        "currency": null,
        "method": null
      },
      "economics_scope": "Member contributions, assay generation, model-development services, and internal-use benefits are disclosed; transaction value and the allocation of consideration among data, laboratory work, infrastructure, and models are not disclosed.",
      "rights": {
        "training": "Permitted within the consortium's federated structure; Ginkgo trains a foundation model and members can train and refine models across consortium data.",
        "post_training": "Members may fine-tune models in their own environments; exact portability and continuing-access terms are not disclosed.",
        "inference_retrieval": "Members may use resulting models and derivatives internally; raw proprietary sequences of other members are not exposed.",
        "ownership_transfer": "Members retain ownership of their contributed sequences and corresponding assay data; no corpus ownership transfer is disclosed.",
        "exclusivity": "No exclusivity is disclosed.",
        "retention": "The initial dataset is expected in early 2027; retention, deletion, withdrawal, and post-membership access terms are not disclosed.",
        "derived_model": "Members may use models and derivatives internally and may fine-tune within their own environments; ownership allocation for shared and member-specific models is not fully disclosed.",
        "sublicensing": "No right to sublicense underlying member sequences, assay data, or shared models is disclosed."
      },
      "restrictions": {
        "governed_access": "Apheris provides privacy-preserving federated infrastructure so members can train, benchmark, and refine models without exposing raw proprietary sequences.",
        "geographic_sovereignty": "not_disclosed",
        "privacy_pii": "The disclosed assets are antibody sequences and assay data; patient-level or personal data is not identified.",
        "trade_secret": "Raw proprietary member sequences are not exposed to other members, and ownership remains with the contributing member.",
        "de_identification": "not_applicable_to_disclosed_antibody_assets",
        "employee_customer_data": "No employee or ordinary customer data is identified; the external assets are proprietary scientific research data."
      },
      "refreshability": "The consortium targets an initial 10,000-antibody dataset with early-2027 delivery and may add member contributions, but no fixed refresh cadence is disclosed.",
      "historical_depth": "The announcement does not disclose collection start dates, vintages, or longitudinal depth of contributed sequence portfolios.",
      "decision_outcome_richness": "Standardized developability assays connect antibody sequences to measured properties used to prioritize, optimize, or reject therapeutic candidates.",
      "likely_model_use": "Foundation-model training, federated member training, benchmarking, refinement, and member-local fine-tuning for antibody developability prediction and biologics R&D.",
      "likely_strategic_value": "Pools scarce, competitively sensitive antibody sequences and harmonized experimental outcomes across multiple drug developers while preserving member control, creating a learning asset broader than any one participant's isolated corpus.",
      "m_and_a_implications": "Demonstrates a governed consortium alternative to outright acquisition: strategic firms can pool learning rights across proprietary scientific assets without transferring raw data ownership.",
      "analysis_boundary": "Canonical Knowledge facet is true. Professional Workflow is false because the arrangement supports scientific learning but does not grant execution authority over a member operating workflow. Human Response is false and response subject is scientific/biological nonhuman. The strict data-rights derived flag is true because named proprietary corporate corpora, separately material federated learning rights, explicit model training, and a non-ordinary consortium structure are all disclosed. No ownership, sublicensing, exclusivity, economics, or raw-sequence access is inferred beyond the announcement.",
      "confidence": "high",
      "scores": {
        "methodology_version": "1.0.0",
        "asset": {
          "score": 90,
          "coverage_pct": 92,
          "rationale": "Strategic Signal analysis. The pooled proprietary sequences, standardized experimental outcomes, cross-company scope, and explicit model-learning rights create an unusually strong scientific asset; historical depth and continuing refresh cadence remain partly undisclosed.",
          "factor_ratings": {
            "uniqueness_scarcity": 4.5,
            "historical_depth": null,
            "decision_outcome_richness": 4,
            "decision_value_density": 4.5,
            "real_world_grounding": 5,
            "domain_value": 5,
            "refreshability": 3,
            "proprietary_advantage": 5,
            "model_learning_usefulness": 5,
            "non_replicability": 4.5,
            "rights_usability": 4.5
          },
          "factor_evidence": {
            "uniqueness_scarcity": {
              "basis": "analysis",
              "source_ids": [
                "E1"
              ],
              "rationale": "The consortium pools otherwise siloed proprietary antibody sequences and harmonized assay outputs across four large drug developers."
            },
            "decision_outcome_richness": {
              "basis": "analysis",
              "source_ids": [
                "E1"
              ],
              "rationale": "The standardized assays connect antibody sequences to measured developability properties that guide candidate-selection decisions."
            },
            "decision_value_density": {
              "basis": "analysis",
              "source_ids": [
                "E1"
              ],
              "rationale": "Each experimentally measured sequence can inform costly and high-value biologic discovery and development decisions."
            },
            "real_world_grounding": {
              "basis": "disclosed_fact",
              "source_ids": [
                "E1"
              ],
              "rationale": "Ginkgo will generate standardized physical assay data rather than relying only on synthetic or literature-derived examples."
            },
            "domain_value": {
              "basis": "analysis",
              "source_ids": [
                "E1"
              ],
              "rationale": "Antibody developability is central to biopharmaceutical R&D, where better early selection can avoid expensive late-stage failure."
            },
            "refreshability": {
              "basis": "analysis",
              "source_ids": [
                "E1"
              ],
              "rationale": "The initial target is 10,000 antibodies with a first dataset expected in early 2027; a continuing refresh cadence is not disclosed."
            },
            "proprietary_advantage": {
              "basis": "disclosed_fact",
              "source_ids": [
                "E1"
              ],
              "rationale": "Each pharmaceutical member contributes proprietary sequences while federated infrastructure prevents exposure of raw member data."
            },
            "model_learning_usefulness": {
              "basis": "disclosed_fact",
              "source_ids": [
                "E1"
              ],
              "rationale": "Ginkgo will train a foundation model and members may train, benchmark, refine and fine-tune models against the consortium data."
            },
            "non_replicability": {
              "basis": "analysis",
              "source_ids": [
                "E1"
              ],
              "rationale": "Reproducing the pooled sequences and standardized assays would require access to multiple proprietary portfolios and substantial laboratory work."
            },
            "rights_usability": {
              "basis": "analysis",
              "source_ids": [
                "E1"
              ],
              "rationale": "Federated training and member-local fine-tuning make the asset usable for learning while preserving raw-sequence boundaries."
            }
          }
        },
        "transaction_signal": {
          "score": 51,
          "coverage_pct": 85,
          "rationale": "Strategic Signal analysis. The consortium provides explicit and well-governed training rights with strong precedent value, but no disclosed economics or clean price discovery.",
          "factor_ratings": {
            "disclosed_economics": 0,
            "clean_price_discovery": 0,
            "data_consideration_separability": 2.5,
            "explicit_model_use": 5,
            "rights_clarity": 4.5,
            "competing_bids": null,
            "strategic_buyer_quality": 4.5,
            "precedent_value": 5,
            "independent_model_uplift": null
          },
          "factor_evidence": {
            "disclosed_economics": {
              "basis": "disclosed_fact",
              "source_ids": [
                "E1"
              ],
              "rationale": "The announcement does not disclose membership fees, cash consideration, revenue share, or minimum commitments."
            },
            "clean_price_discovery": {
              "basis": "analysis",
              "source_ids": [
                "E1"
              ],
              "rationale": "No transaction value, auction, bid process, or separable market price for the contributed rights is disclosed."
            },
            "data_consideration_separability": {
              "basis": "analysis",
              "source_ids": [
                "E1"
              ],
              "rationale": "Members expressly contribute proprietary sequences and receive model and dataset-use benefits, but monetary consideration is not separated."
            },
            "explicit_model_use": {
              "basis": "disclosed_fact",
              "source_ids": [
                "E1"
              ],
              "rationale": "Foundation-model training, benchmarking, refinement, and member-local fine-tuning are all expressly disclosed."
            },
            "rights_clarity": {
              "basis": "analysis",
              "source_ids": [
                "E1"
              ],
              "rationale": "Contribution ownership, raw-data non-exposure, federated access, internal model use, and local fine-tuning are described, while sublicensing remains undisclosed."
            },
            "strategic_buyer_quality": {
              "basis": "analysis",
              "source_ids": [
                "E1"
              ],
              "rationale": "AbbVie, argenx, Lundbeck, and Takeda are major strategic biopharma participants working with Ginkgo and Apheris."
            },
            "precedent_value": {
              "basis": "analysis",
              "source_ids": [
                "E1"
              ],
              "rationale": "The federated consortium structure is a repeatable template for pooling competitively sensitive scientific data without central raw-data transfer."
            }
          }
        }
      },
      "sources": [
        {
          "type": "primary_syndicated",
          "publisher": "Ginkgo Datapoints and Apheris via Business Wire",
          "title": "Ginkgo Datapoints and Apheris Welcome AbbVie, argenx, Lundbeck and Takeda as Founding Members of the Antibody Developability Consortium",
          "url": "https://www.businesswire.com/news/home/20260929556449/en/Ginkgo-Datapoints-and-Apheris-Welcome-AbbVie-argenx-Lundbeck-and-Takeda-as-Founding-Members-of-the-Antibody-Developability-Consortium"
        }
      ],
      "date_added": "2026-10-01",
      "date_last_reviewed": "2026-10-01",
      "publication_review": {
        "status": "human_reviewed",
        "approval_channel": "owner_conversation",
        "reviewer": "Mike Ye",
        "reviewed_at": "2026-10-01T15:39:17.459Z",
        "decision_id": "conversation-20261001-D4406416282E",
        "packet_sha256": "330c0cd6e2b7b761bddc7a509c54ad9a2d86c2519f50cf3c2efe6d9412bdcfeb",
        "record_sha256": "355e673232bf563ed8484247753d67ff0f4c353fdafa2850263693087b157fc1"
      }
    },
    {
      "transaction_id": "SS-AIDT-2026-0016",
      "slug": "openai-synopsys-gpt-synopsys-eda-training-license",
      "canonical_path": "/ai-data-transactions/deals/ss-aidt-2026-0016/",
      "announcement_date": "2026-09-30",
      "buyer_ai_developer": [
        "OpenAI",
        "GPT-Synopsys"
      ],
      "seller_data_owner": [
        "Synopsys"
      ],
      "transaction_status": "active_multi_year_preferred_partnership",
      "industry": "Semiconductor design, electronic design automation, and frontier AI",
      "data_category": [
        "Proprietary electronic design automation tools",
        "Verified chip-design workflows and tool outputs",
        "Synopsys engineering expertise"
      ],
      "corpus_description": "Synopsys' proprietary EDA toolchain and expert engineering workflows, including the execution behavior, outputs, constraints, and iterative optimization patterns needed to design and verify chips. The reviewed sources do not identify a delivered static corpus, and customer data is excluded from training.",
      "primary_transaction_structure": "Training Rights License",
      "access_structure": "Under a multi-year preferred partnership, OpenAI licenses Synopsys EDA tools to develop and train the OpenAI-hosted GPT-Synopsys model to run expert tools, interpret results, and iteratively optimize designs. Synopsys contributes engineering expertise; customer data is not used for training.",
      "asset_disposition": "Embedded / Continuing",
      "corpus_renewal": "Renewable Corpus",
      "source_code_software_asset": "yes",
      "disclosed_consideration": {
        "status": "economic_structure_disclosed_amount_not_disclosed",
        "amount": null,
        "currency": null,
        "description": "Reuters reports that OpenAI pays Synopsys a training subscription fee and that the parties share downstream GPT-Synopsys revenue. Dollar amounts, revenue-share percentages, minimum commitments, and term-by-term allocation are not disclosed."
      },
      "estimated_economics": {
        "status": "not_estimated",
        "amount": null,
        "currency": null,
        "method": null
      },
      "economics_scope": "The reported training subscription covers development access, while downstream revenue sharing covers commercial use. The reviewed sources do not allocate value among tools, expertise, workflow access, model hosting, or go-to-market rights.",
      "rights": {
        "training": "Permitted for GPT-Synopsys development: OpenAI licenses Synopsys EDA tools so the model can learn tool execution, output interpretation, and iterative design optimization.",
        "post_training": "Commercial model deployment is contemplated through GPT-Synopsys, but separate fine-tuning, distillation, and adaptation rights are not disclosed.",
        "inference_retrieval": "GPT-Synopsys will be OpenAI-hosted and available for design tasks; exact retrieval access to Synopsys source assets is not disclosed.",
        "ownership_transfer": "No ownership transfer of Synopsys tools, workflows, or customer data is disclosed.",
        "exclusivity": "OpenAI is described as Synopsys' preferred model-development partner; the scope and legal exclusivity of that status are not disclosed.",
        "retention": "The agreement is multi-year, but training-data, tool-output, log, and post-termination retention periods are not disclosed.",
        "derived_model": "GPT-Synopsys is a jointly developed domain model hosted by OpenAI; intellectual-property allocation and rights in intermediate or derivative models are not fully disclosed.",
        "sublicensing": "No right to sublicense Synopsys tools or underlying workflow assets is disclosed; downstream customer access to GPT-Synopsys does not establish sublicensing of those assets."
      },
      "restrictions": {
        "governed_access": "OpenAI receives licensed tool access for specified model development; GPT-Synopsys is OpenAI-hosted and customer data is explicitly not used for training.",
        "geographic_sovereignty": "not_disclosed",
        "privacy_pii": "The disclosed asset is engineering software and workflow knowledge. Personal data is not identified.",
        "trade_secret": "Synopsys retains ownership of its tools; detailed safeguards for proprietary algorithms, tool outputs, and workflow traces are not disclosed.",
        "de_identification": "not_applicable_to_disclosed_engineering_assets",
        "employee_customer_data": "Synopsys and OpenAI state that customer data will not be used to train GPT-Synopsys."
      },
      "refreshability": "The multi-year development relationship can incorporate continuing tool and engineering-workflow evolution, but no fixed refresh or retraining cadence is disclosed.",
      "historical_depth": "The sources do not quantify the vintage or historical depth of the tool execution and workflow knowledge available for training.",
      "decision_outcome_richness": "The model learns to execute expert tools, interpret outputs, and iteratively optimize designs, linking design decisions to verification and optimization feedback.",
      "likely_model_use": "Domain-model training and post-training for chip-design planning, tool orchestration, output interpretation, iterative optimization, and engineering assistance.",
      "likely_strategic_value": "Provides OpenAI governed access to a leading proprietary EDA environment and expert workflow structure that public text alone cannot reproduce, enabling frontier models to perform consequential semiconductor engineering tasks.",
      "m_and_a_implications": "Establishes a high-value vertical-AI partnership template in which a domain-software leader monetizes tool and workflow learning access through a training subscription plus downstream revenue share without selling the underlying platform.",
      "analysis_boundary": "Canonical Knowledge is true. Professional Workflow remains unknown in the canonical registry because the strict legacy adapter has no separately approved capability-flow classification for this arrangement; executable tool access is not converted into that facet by narrative inference. Human Response is likewise not separately classified in this strict record. The strict data-rights derived flag remains true because a named proprietary operating asset is licensed for model development, the training subscription makes those rights separately material, model use is explicit, and the arrangement is not ordinary enterprise deployment. Customer data is excluded from training. Ownership, raw data delivery, retention, sublicensing, and model-IP allocation are not inferred.",
      "confidence": "high",
      "scores": {
        "methodology_version": "1.0.0",
        "asset": {
          "score": 94,
          "coverage_pct": 92,
          "rationale": "Strategic Signal analysis. Synopsys' proprietary EDA environment and verified expert workflows are scarce, economically consequential, highly grounded, and explicitly useful for model learning; historical depth is not disclosed.",
          "factor_ratings": {
            "uniqueness_scarcity": 5,
            "historical_depth": null,
            "decision_outcome_richness": 4,
            "decision_value_density": 5,
            "real_world_grounding": 4.5,
            "domain_value": 5,
            "refreshability": 4,
            "proprietary_advantage": 5,
            "model_learning_usefulness": 5,
            "non_replicability": 5,
            "rights_usability": 4.5
          },
          "factor_evidence": {
            "uniqueness_scarcity": {
              "basis": "analysis",
              "source_ids": [
                "E1"
              ],
              "rationale": "Synopsys' production EDA toolchain, engineering expertise, and verified design workflows are difficult to substitute with public chip-design data."
            },
            "decision_outcome_richness": {
              "basis": "analysis",
              "source_ids": [
                "E1"
              ],
              "rationale": "Tool execution and iterative optimization expose relationships between design choices, tool outputs, constraints, and improved chip designs."
            },
            "decision_value_density": {
              "basis": "analysis",
              "source_ids": [
                "E1",
                "E2"
              ],
              "rationale": "Semiconductor design decisions carry extremely high engineering, tape-out, manufacturing, schedule, and product-value consequences."
            },
            "real_world_grounding": {
              "basis": "disclosed_fact",
              "source_ids": [
                "E1"
              ],
              "rationale": "GPT-Synopsys is trained to operate production engineering tools and interpret their outputs, grounding learning in real design processes."
            },
            "domain_value": {
              "basis": "analysis",
              "source_ids": [
                "E1",
                "E2"
              ],
              "rationale": "Electronic design automation is strategic infrastructure for the semiconductor industry and directly shapes costly chip development."
            },
            "refreshability": {
              "basis": "analysis",
              "source_ids": [
                "E1"
              ],
              "rationale": "A multi-year preferred partnership can expose continuing tool and workflow evolution, although no specific data-refresh cadence is disclosed."
            },
            "proprietary_advantage": {
              "basis": "disclosed_fact",
              "source_ids": [
                "E1"
              ],
              "rationale": "The licensed asset is Synopsys' proprietary EDA environment and engineering expertise, not a readily available public corpus."
            },
            "model_learning_usefulness": {
              "basis": "disclosed_fact",
              "source_ids": [
                "E1"
              ],
              "rationale": "The tools and workflows are licensed expressly so the model can learn to run tools, interpret outputs, and optimize designs."
            },
            "non_replicability": {
              "basis": "analysis",
              "source_ids": [
                "E1"
              ],
              "rationale": "Recreating the full toolchain, verification behavior, and accumulated engineering workflow knowledge would require exceptional time, capital, and domain access."
            },
            "rights_usability": {
              "basis": "analysis",
              "source_ids": [
                "E1"
              ],
              "rationale": "The multi-year license and hosted product path support model development and commercial deployment, while detailed retention and derivative-right terms remain undisclosed."
            }
          }
        },
        "transaction_signal": {
          "score": 74,
          "coverage_pct": 85,
          "rationale": "Strategic Signal analysis. The training subscription, downstream revenue share, explicit model-development license, and strong counterparties create a high-value signal, while amounts, bid dynamics, and independent uplift remain undisclosed.",
          "factor_ratings": {
            "disclosed_economics": 2.5,
            "clean_price_discovery": 2,
            "data_consideration_separability": 4,
            "explicit_model_use": 5,
            "rights_clarity": 4.5,
            "competing_bids": null,
            "strategic_buyer_quality": 5,
            "precedent_value": 5,
            "independent_model_uplift": null
          },
          "factor_evidence": {
            "disclosed_economics": {
              "basis": "disclosed_fact",
              "source_ids": [
                "E2"
              ],
              "rationale": "Reuters reports a training subscription fee and downstream revenue sharing, but no dollar amount, percentage, or minimum commitment."
            },
            "clean_price_discovery": {
              "basis": "analysis",
              "source_ids": [
                "E1",
                "E2"
              ],
              "rationale": "The bilateral multi-year partnership reveals an economic structure but not a public price, competitive process, or asset valuation."
            },
            "data_consideration_separability": {
              "basis": "analysis",
              "source_ids": [
                "E1",
                "E2"
              ],
              "rationale": "Reuters identifies a training subscription distinct from downstream revenue sharing, making model-development access partly separable from product commercialization."
            },
            "explicit_model_use": {
              "basis": "disclosed_fact",
              "source_ids": [
                "E1"
              ],
              "rationale": "The agreement expressly licenses Synopsys tools for GPT-Synopsys development and training to execute and optimize chip-design workflows."
            },
            "rights_clarity": {
              "basis": "analysis",
              "source_ids": [
                "E1",
                "E2"
              ],
              "rationale": "The licensed development purpose, OpenAI hosting, preferred-partner structure, customer-data training prohibition, subscription, and revenue share are disclosed; retention and sublicensing are not."
            },
            "strategic_buyer_quality": {
              "basis": "analysis",
              "source_ids": [
                "E1",
                "E2"
              ],
              "rationale": "OpenAI is a leading frontier-model developer and Synopsys is a leading electronic-design-automation provider."
            },
            "precedent_value": {
              "basis": "analysis",
              "source_ids": [
                "E1",
                "E2"
              ],
              "rationale": "A domain-software provider licensing expert tools and workflows for frontier-model training with subscription and revenue share is a repeatable vertical-AI template."
            }
          }
        }
      },
      "sources": [
        {
          "type": "primary",
          "publisher": "Synopsys and OpenAI",
          "title": "OpenAI and Synopsys Announce GPT-Synopsys, Frontier Intelligence to Revolutionize Chip Design",
          "url": "https://investor.synopsys.com/news/news-details/2026/OpenAI-and-Synopsys-Announce-GPT-Synopsys-Frontier-Intelligence-to-Revolutionize-Chip-Design"
        },
        {
          "type": "secondary",
          "publisher": "Reuters",
          "title": "Synopsys, OpenAI strike deal to develop AI model for chip design work",
          "url": "https://www.reuters.com/business/synopsys-openai-strike-deal-develop-ai-model-chip-design-work-2026-09-30/"
        }
      ],
      "date_added": "2026-10-01",
      "date_last_reviewed": "2026-10-07",
      "publication_review": {
        "status": "human_reviewed",
        "approval_channel": "owner_conversation",
        "reviewer": "Mike Ye",
        "reviewed_at": "2026-10-07T18:10:22.357Z",
        "decision_id": "conversation-20261007-6C2EB1F82CB3",
        "packet_sha256": "dc08a78d4b713c345b02753095ca2d3602418e64ff7c76ac043c057fa0dfb7aa",
        "record_sha256": "b77584df90656b3d44bc2277aea8702327497348bc2cfdc73dd5fd58b02fd465"
      }
    },
    {
      "transaction_id": "SS-AIDT-2026-0018",
      "slug": "biohub-virtual-biology-ai-data-commercial-embargo",
      "canonical_path": "/ai-data-transactions/deals/ss-aidt-2026-0018/",
      "announcement_date": "2026-10-07",
      "buyer_ai_developer": [
        "Meta",
        "Google DeepMind",
        "Isomorphic Labs",
        "Biohub predictive biology models"
      ],
      "seller_data_owner": [
        "Biohub",
        "U.S. Department of Energy",
        "National Institutes of Health",
        "Participating scientific organizations"
      ],
      "transaction_status": "announced_funded_program",
      "industry": "Biomedical research, life sciences, drug discovery, scientific AI, and public research infrastructure",
      "data_category": [
        "Large-scale cellular response measurements",
        "Spatial transcriptomics and intact-tissue molecular maps",
        "Standardized federal biomedical datasets and repositories",
        "AI-ready biology training data"
      ],
      "corpus_description": "New, coordinated measurements of how cells and tissues respond across conditions, including spatial transcriptomics and environmental-response screens, combined with standardized NIH and DOE biomedical resources. Commercially funded data is initially embargoed for participating funders and later released as an open scientific resource; government-funded work has no comparable restriction.",
      "primary_transaction_structure": "Strategic Model Co-Development",
      "access_structure": "Biohub coordinates data generation and standardization with DOE and NIH. Meta, Google DeepMind and Isomorphic Labs jointly invest $300 million and receive a one-year commercial embargo period for data they help develop, enabling early model training and experimentation before public release.",
      "asset_disposition": "Embedded / Continuing",
      "corpus_renewal": "Renewable Corpus",
      "source_code_software_asset": "no",
      "disclosed_consideration": {
        "status": "aggregate_commitments_disclosed_allocation_partial",
        "amount": 1800000000,
        "currency": "USD",
        "description": "The initiative totals $1.8 billion in funding, data, compute and measurement technology: $300 million jointly from Meta, Google DeepMind and Isomorphic Labs, more than $500 million from DOE over five years, more than $500 million in prior NIH-funded resources, and Biohub’s earlier $500 million commitment."
      },
      "estimated_economics": {
        "status": "not_estimated",
        "amount": null,
        "currency": null,
        "method": null
      },
      "economics_scope": "Only the $300 million commercial contribution is directly associated with the time-limited embargo advantage; the disclosed $1.8 billion total also includes public and philanthropic investments and in-kind resources.",
      "rights": {
        "training": "Permitted: the initiative is expressly building datasets for training predictive AI models of biology.",
        "post_training": "Model evaluation and iterative improvement are contemplated; detailed fine-tuning and distillation terms are not disclosed.",
        "inference_retrieval": "Commercial funders receive early access to data they help develop during an embargo period; government-funded resources are intended to be openly accessible.",
        "ownership_transfer": "No transfer of ownership in underlying biological samples, federal repositories or resulting public datasets is disclosed.",
        "exclusivity": "Commercial partners receive a one-year exclusive-access embargo for the data they develop; the data is then released broadly. Government-funded work has no such restriction.",
        "retention": "Post-embargo retention and derivative-use limits are not disclosed.",
        "derived_model": "Partners may train predictive biology models; ownership and commercialization rights in derived models are not fully disclosed.",
        "sublicensing": "No sublicensing terms are disclosed; eventual public release is not inferred to grant commercial sublicensing before the embargo ends."
      },
      "restrictions": {
        "governed_access": "Commercially funded data is subject to a one-year embargo, while government-funded data carries no such restriction and the long-term resource is intended to be open.",
        "geographic_sovereignty": "not_disclosed",
        "privacy_pii": "The initiative concerns cellular and tissue data; human-subject privacy, consent and re-identification controls are not detailed in the reviewed sources.",
        "trade_secret": "The embargo creates a temporary proprietary advantage; later public release limits permanent exclusivity.",
        "de_identification": "not_disclosed",
        "employee_customer_data": "not_applicable"
      },
      "refreshability": "The program will generate new biological measurements over five years, with the first major dataset expected in about one year and continuing coordinated expansion afterward.",
      "historical_depth": "The asset combines existing NIH-funded resources with newly generated measurements; comparable historical depth and longitudinal sample coverage are not quantified.",
      "decision_outcome_richness": "Screens measure cellular responses to environmental changes, linking biological context and perturbation to observed molecular outcomes for predictive modeling.",
      "likely_model_use": "Training and evaluation of predictive biology and virtual-cell models for digital experiments, disease understanding and drug-development prioritization.",
      "likely_strategic_value": "Creates a coordinated, model-ready biological measurement layer at a scale unavailable to any single laboratory, with a temporary commercial head start for major AI and drug-discovery funders.",
      "m_and_a_implications": "Shows that commercial AI firms will finance open-science data generation in exchange for a time-limited access advantage, creating a new benchmark for separately material data rights without permanent privatization.",
      "analysis_boundary": "Canonical Knowledge is true and Professional Workflow is true for scientific measurement and virtual experimentation. Human Response is not applicable because the modeled response subject is cellular or tissue biology, not repeated human behavior-response episodes. Strict data-rights qualification is true only for the separable commercial-funded tranche with a one-year embargo and explicit model training; government-funded open work is part of the same arrangement but is not itself scored as a private rights transfer.",
      "confidence": "high",
      "scores": {
        "methodology_version": "1.0.0",
        "asset": {
          "score": 94,
          "coverage_pct": 100,
          "rationale": "The coordinated cell-response data is exceptionally scarce, grounded, renewable and useful for predictive-model learning; exact historical depth and post-embargo rights remain less clear.",
          "factor_ratings": {
            "uniqueness_scarcity": 5,
            "historical_depth": 3,
            "decision_outcome_richness": 5,
            "decision_value_density": 5,
            "real_world_grounding": 5,
            "domain_value": 5,
            "refreshability": 5,
            "proprietary_advantage": 4,
            "model_learning_usefulness": 5,
            "non_replicability": 5,
            "rights_usability": 4
          },
          "factor_evidence": {
            "uniqueness_scarcity": {
              "basis": "analysis",
              "source_ids": [
                "E1",
                "E2",
                "E3"
              ],
              "rationale": "Strategic Signal analysis: reviewed evidence supports a 5 of 5 rating for uniqueness scarcity under methodology v1.0.0."
            },
            "historical_depth": {
              "basis": "analysis",
              "source_ids": [
                "E1",
                "E2",
                "E3"
              ],
              "rationale": "Strategic Signal analysis: reviewed evidence supports a 3 of 5 rating for historical depth under methodology v1.0.0."
            },
            "decision_outcome_richness": {
              "basis": "analysis",
              "source_ids": [
                "E1",
                "E2",
                "E3"
              ],
              "rationale": "Strategic Signal analysis: reviewed evidence supports a 5 of 5 rating for decision outcome richness under methodology v1.0.0."
            },
            "decision_value_density": {
              "basis": "analysis",
              "source_ids": [
                "E1",
                "E2",
                "E3"
              ],
              "rationale": "Strategic Signal analysis: reviewed evidence supports a 5 of 5 rating for decision value density under methodology v1.0.0."
            },
            "real_world_grounding": {
              "basis": "disclosed_fact",
              "source_ids": [
                "E1",
                "E2",
                "E3"
              ],
              "rationale": "Strategic Signal analysis: reviewed evidence supports a 5 of 5 rating for real world grounding under methodology v1.0.0."
            },
            "domain_value": {
              "basis": "analysis",
              "source_ids": [
                "E1",
                "E2",
                "E3"
              ],
              "rationale": "Strategic Signal analysis: reviewed evidence supports a 5 of 5 rating for domain value under methodology v1.0.0."
            },
            "refreshability": {
              "basis": "analysis",
              "source_ids": [
                "E1",
                "E2",
                "E3"
              ],
              "rationale": "Strategic Signal analysis: reviewed evidence supports a 5 of 5 rating for refreshability under methodology v1.0.0."
            },
            "proprietary_advantage": {
              "basis": "analysis",
              "source_ids": [
                "E1",
                "E2",
                "E3"
              ],
              "rationale": "Strategic Signal analysis: reviewed evidence supports a 4 of 5 rating for proprietary advantage under methodology v1.0.0."
            },
            "model_learning_usefulness": {
              "basis": "disclosed_fact",
              "source_ids": [
                "E1",
                "E2",
                "E3"
              ],
              "rationale": "Strategic Signal analysis: reviewed evidence supports a 5 of 5 rating for model learning usefulness under methodology v1.0.0."
            },
            "non_replicability": {
              "basis": "analysis",
              "source_ids": [
                "E1",
                "E2",
                "E3"
              ],
              "rationale": "Strategic Signal analysis: reviewed evidence supports a 5 of 5 rating for non replicability under methodology v1.0.0."
            },
            "rights_usability": {
              "basis": "analysis",
              "source_ids": [
                "E1",
                "E2",
                "E3"
              ],
              "rationale": "Strategic Signal analysis: reviewed evidence supports a 4 of 5 rating for rights usability under methodology v1.0.0."
            }
          }
        },
        "transaction_signal": {
          "score": 84,
          "coverage_pct": 85,
          "rationale": "The disclosed commercial funding and one-year access advantage strongly signal standalone data value, while price allocation, competition and independent uplift are not disclosed.",
          "factor_ratings": {
            "disclosed_economics": 4.5,
            "clean_price_discovery": 2.5,
            "data_consideration_separability": 4.5,
            "explicit_model_use": 5,
            "rights_clarity": 4,
            "competing_bids": null,
            "strategic_buyer_quality": 5,
            "precedent_value": 5,
            "independent_model_uplift": null
          },
          "factor_evidence": {
            "disclosed_economics": {
              "basis": "disclosed_fact",
              "source_ids": [
                "E1",
                "E2",
                "E3"
              ],
              "rationale": "Strategic Signal analysis: reviewed evidence supports a 4.5 of 5 rating for disclosed economics under methodology v1.0.0."
            },
            "clean_price_discovery": {
              "basis": "analysis",
              "source_ids": [
                "E1",
                "E2",
                "E3"
              ],
              "rationale": "Strategic Signal analysis: reviewed evidence supports a 2.5 of 5 rating for clean price discovery under methodology v1.0.0."
            },
            "data_consideration_separability": {
              "basis": "analysis",
              "source_ids": [
                "E1",
                "E2",
                "E3"
              ],
              "rationale": "Strategic Signal analysis: reviewed evidence supports a 4.5 of 5 rating for data consideration separability under methodology v1.0.0."
            },
            "explicit_model_use": {
              "basis": "disclosed_fact",
              "source_ids": [
                "E1",
                "E2",
                "E3"
              ],
              "rationale": "Strategic Signal analysis: reviewed evidence supports a 5 of 5 rating for explicit model use under methodology v1.0.0."
            },
            "rights_clarity": {
              "basis": "analysis",
              "source_ids": [
                "E1",
                "E2",
                "E3"
              ],
              "rationale": "Strategic Signal analysis: reviewed evidence supports a 4 of 5 rating for rights clarity under methodology v1.0.0."
            },
            "strategic_buyer_quality": {
              "basis": "analysis",
              "source_ids": [
                "E1",
                "E2",
                "E3"
              ],
              "rationale": "Strategic Signal analysis: reviewed evidence supports a 5 of 5 rating for strategic buyer quality under methodology v1.0.0."
            },
            "precedent_value": {
              "basis": "analysis",
              "source_ids": [
                "E1",
                "E2",
                "E3"
              ],
              "rationale": "Strategic Signal analysis: reviewed evidence supports a 5 of 5 rating for precedent value under methodology v1.0.0."
            }
          }
        }
      },
      "sources": [
        {
          "type": "primary_syndicated",
          "publisher": "Biohub via EurekAlert",
          "title": "International, cross-sector collaboration commits nearly $2 billion to build foundational data for AI models",
          "url": "https://www.eurekalert.org/news-releases/1146765"
        },
        {
          "type": "secondary",
          "publisher": "Reuters",
          "title": "US government, Google join Zuckerberg-backed Biohub in $1.8 billion push for AI biology data",
          "url": "https://www.reuters.com/business/healthcare-pharmaceuticals/us-government-google-join-zuckerberg-backed-biohub-18-billion-push-ai-biology-2026-10-07/"
        },
        {
          "type": "secondary",
          "publisher": "Axios",
          "title": "Zuckerberg teams with Google, U.S. in push to map cells",
          "url": "https://www.axios.com/2026/10/07/zuckerberg-biohub-ai-biology-data"
        }
      ],
      "date_added": "2026-10-07",
      "date_last_reviewed": "2026-10-07",
      "publication_review": {
        "status": "human_reviewed",
        "approval_channel": "owner_conversation",
        "reviewer": "Mike Ye",
        "reviewed_at": "2026-10-07T16:31:20.531Z",
        "decision_id": "conversation-20261007-26D1A0A725B3",
        "packet_sha256": "32194e93d8bd8652fcdd9fff90cb88e0bc7ddb5b2628c9f28e1b867797b67118",
        "record_sha256": "dec3313484a6e685f8931fef4e7804b35609a4914c42530eaa8dd2dc7a8ed008"
      }
    },
    {
      "transaction_id": "SS-AIDT-2026-0017",
      "slug": "openai-ironclad-contracting-workflows-training-evaluation",
      "canonical_path": "/ai-data-transactions/deals/ss-aidt-2026-0017/",
      "announcement_date": "2026-10-06",
      "buyer_ai_developer": [
        "OpenAI",
        "GPT-6 Astra"
      ],
      "seller_data_owner": [
        "Ironclad"
      ],
      "transaction_status": "active_research_collaboration",
      "industry": "Legal technology, enterprise contracting, procurement, and frontier AI",
      "data_category": [
        "Expert contracting workflows and rubrics",
        "Hosted Ironclad software environments",
        "Synthetic legal, commercial, and procurement tasks"
      ],
      "corpus_description": "Eleven high-value contracting tasks selected with Ironclad experts across legal, commercial and procurement work, each evaluated against detailed rubrics and practiced in hosted Ironclad product environments. Training inputs were synthetic and based on filtered public SEC contracts; nonpublic customer contracts were excluded.",
      "primary_transaction_structure": "Training Rights License",
      "access_structure": "OpenAI researchers worked with Ironclad employees and experienced users to define tasks and success criteria, received hosted Ironclad software environments for practice, created synthetic training tasks and applied reinforcement learning and evaluation to GPT-6 Astra.",
      "asset_disposition": "Embedded / Continuing",
      "corpus_renewal": "Renewable Corpus",
      "source_code_software_asset": "yes",
      "disclosed_consideration": {
        "status": "not_disclosed",
        "amount": null,
        "currency": null,
        "description": "No cash consideration or commercial rights allocation is disclosed; the material consideration is expert workflow design, hosted product access and model-development collaboration."
      },
      "estimated_economics": {
        "status": "not_estimated",
        "amount": null,
        "currency": null,
        "method": null
      },
      "economics_scope": "The announcement does not allocate value among expert time, hosted environments, synthetic task creation, evaluation design or prospective product integration.",
      "rights": {
        "training": "Permitted for GPT-6 Astra and internal model development using synthetic tasks representing Ironclad workflows in hosted environments.",
        "post_training": "Reinforcement learning and model improvement are expressly disclosed; broader fine-tuning and distillation terms are not.",
        "inference_retrieval": "Hosted environments were used for research practice and evaluation; production customer retrieval rights are not part of the disclosed research arrangement.",
        "ownership_transfer": "No ownership transfer of Ironclad software, customer contracts or OpenAI models is disclosed.",
        "exclusivity": "No exclusivity is disclosed; OpenAI invites a small number of additional software companies into similar collaborations.",
        "retention": "Retention terms for synthetic tasks, model traces and hosted-environment logs are not disclosed.",
        "derived_model": "GPT-6 Astra is expressly trained on Ironclad tasks; ownership and reuse rights in intermediate artifacts are not fully disclosed.",
        "sublicensing": "No sublicensing of Ironclad software or underlying customer assets is disclosed."
      },
      "restrictions": {
        "governed_access": "Research used hosted Ironclad environments, expert-defined tasks and detailed rubrics; production customer access is not established.",
        "geographic_sovereignty": "not_disclosed",
        "privacy_pii": "OpenAI states tasks were created from filtered public SEC contracts with personal-information removal.",
        "trade_secret": "Ironclad retains its product and workflow know-how; detailed confidentiality terms are not disclosed.",
        "de_identification": "Synthetic tasks and public-contract filters were used instead of nonpublic customer contracts.",
        "employee_customer_data": "OpenAI states it did not use OpenAI customer data, OpenAI internal contracts, or nonpublic Ironclad customer data or contracts."
      },
      "refreshability": "The collaboration model can add new workflows, tasks, rubrics and hosted product changes, but no formal refresh cadence is disclosed.",
      "historical_depth": "The research used representative contemporary workflows; historical customer contract depth was excluded and is not disclosed.",
      "decision_outcome_richness": "Tasks encode requirements, configuration actions, approval routes and rubric-based verification of completed contracting processes.",
      "likely_model_use": "Frontier-model training, reinforcement learning and evaluation for complex computer-use agents performing legal, procurement and commercial workflows.",
      "likely_strategic_value": "Provides OpenAI with expert-defined, executable professional workflows and verifiable task environments that public text alone cannot reproduce.",
      "m_and_a_implications": "Establishes a repeatable partnership template in which vertical software companies supply expert tasks, secure environments and evaluation criteria directly to frontier-model developers.",
      "analysis_boundary": "Canonical Knowledge and Professional Workflow facets are true. Human Response is not applicable. The strict data-rights derived flag is true because Ironclad grants identifiable proprietary workflow and software-environment access for training and evaluation outside ordinary deployment. Training is explicit; customer contracts are excluded; ownership, sublicensing and retention are not inferred.",
      "confidence": "high",
      "scores": {
        "methodology_version": "1.0.0",
        "asset": {
          "score": 93,
          "coverage_pct": 92,
          "rationale": "Ironclad provides scarce expert workflow structure, executable environments and rubric-based feedback with direct model-learning value; historical depth is intentionally excluded.",
          "factor_ratings": {
            "uniqueness_scarcity": 4.5,
            "historical_depth": null,
            "decision_outcome_richness": 5,
            "decision_value_density": 4.5,
            "real_world_grounding": 5,
            "domain_value": 4.5,
            "refreshability": 4,
            "proprietary_advantage": 4.5,
            "model_learning_usefulness": 5,
            "non_replicability": 4.5,
            "rights_usability": 4.5
          },
          "factor_evidence": {
            "uniqueness_scarcity": {
              "basis": "analysis",
              "source_ids": [
                "E1"
              ],
              "rationale": "Strategic Signal analysis: reviewed evidence supports a 4.5 of 5 rating for uniqueness scarcity under methodology v1.0.0."
            },
            "decision_outcome_richness": {
              "basis": "analysis",
              "source_ids": [
                "E1"
              ],
              "rationale": "Strategic Signal analysis: reviewed evidence supports a 5 of 5 rating for decision outcome richness under methodology v1.0.0."
            },
            "decision_value_density": {
              "basis": "analysis",
              "source_ids": [
                "E1"
              ],
              "rationale": "Strategic Signal analysis: reviewed evidence supports a 4.5 of 5 rating for decision value density under methodology v1.0.0."
            },
            "real_world_grounding": {
              "basis": "disclosed_fact",
              "source_ids": [
                "E1"
              ],
              "rationale": "Strategic Signal analysis: reviewed evidence supports a 5 of 5 rating for real world grounding under methodology v1.0.0."
            },
            "domain_value": {
              "basis": "analysis",
              "source_ids": [
                "E1"
              ],
              "rationale": "Strategic Signal analysis: reviewed evidence supports a 4.5 of 5 rating for domain value under methodology v1.0.0."
            },
            "refreshability": {
              "basis": "analysis",
              "source_ids": [
                "E1"
              ],
              "rationale": "Strategic Signal analysis: reviewed evidence supports a 4 of 5 rating for refreshability under methodology v1.0.0."
            },
            "proprietary_advantage": {
              "basis": "analysis",
              "source_ids": [
                "E1"
              ],
              "rationale": "Strategic Signal analysis: reviewed evidence supports a 4.5 of 5 rating for proprietary advantage under methodology v1.0.0."
            },
            "model_learning_usefulness": {
              "basis": "disclosed_fact",
              "source_ids": [
                "E1"
              ],
              "rationale": "Strategic Signal analysis: reviewed evidence supports a 5 of 5 rating for model learning usefulness under methodology v1.0.0."
            },
            "non_replicability": {
              "basis": "analysis",
              "source_ids": [
                "E1"
              ],
              "rationale": "Strategic Signal analysis: reviewed evidence supports a 4.5 of 5 rating for non replicability under methodology v1.0.0."
            },
            "rights_usability": {
              "basis": "analysis",
              "source_ids": [
                "E1"
              ],
              "rationale": "Strategic Signal analysis: reviewed evidence supports a 4.5 of 5 rating for rights usability under methodology v1.0.0."
            }
          }
        },
        "transaction_signal": {
          "score": 63,
          "coverage_pct": 85,
          "rationale": "The arrangement clearly grants model-development access and sets a strong vertical-software precedent, but no price, competitive process or independent uplift evidence is disclosed.",
          "factor_ratings": {
            "disclosed_economics": 1,
            "clean_price_discovery": 1,
            "data_consideration_separability": 3.5,
            "explicit_model_use": 5,
            "rights_clarity": 4.5,
            "competing_bids": null,
            "strategic_buyer_quality": 5,
            "precedent_value": 5,
            "independent_model_uplift": null
          },
          "factor_evidence": {
            "disclosed_economics": {
              "basis": "disclosed_fact",
              "source_ids": [
                "E1"
              ],
              "rationale": "Strategic Signal analysis: reviewed evidence supports a 1 of 5 rating for disclosed economics under methodology v1.0.0."
            },
            "clean_price_discovery": {
              "basis": "analysis",
              "source_ids": [
                "E1"
              ],
              "rationale": "Strategic Signal analysis: reviewed evidence supports a 1 of 5 rating for clean price discovery under methodology v1.0.0."
            },
            "data_consideration_separability": {
              "basis": "analysis",
              "source_ids": [
                "E1"
              ],
              "rationale": "Strategic Signal analysis: reviewed evidence supports a 3.5 of 5 rating for data consideration separability under methodology v1.0.0."
            },
            "explicit_model_use": {
              "basis": "disclosed_fact",
              "source_ids": [
                "E1"
              ],
              "rationale": "Strategic Signal analysis: reviewed evidence supports a 5 of 5 rating for explicit model use under methodology v1.0.0."
            },
            "rights_clarity": {
              "basis": "analysis",
              "source_ids": [
                "E1"
              ],
              "rationale": "Strategic Signal analysis: reviewed evidence supports a 4.5 of 5 rating for rights clarity under methodology v1.0.0."
            },
            "strategic_buyer_quality": {
              "basis": "analysis",
              "source_ids": [
                "E1"
              ],
              "rationale": "Strategic Signal analysis: reviewed evidence supports a 5 of 5 rating for strategic buyer quality under methodology v1.0.0."
            },
            "precedent_value": {
              "basis": "analysis",
              "source_ids": [
                "E1"
              ],
              "rationale": "Strategic Signal analysis: reviewed evidence supports a 5 of 5 rating for precedent value under methodology v1.0.0."
            }
          }
        }
      },
      "sources": [
        {
          "type": "primary",
          "publisher": "OpenAI",
          "title": "Advancing computer use with Ironclad",
          "url": "https://openai.com/index/advancing-computer-use-with-ironclad/"
        }
      ],
      "date_added": "2026-10-07",
      "date_last_reviewed": "2026-10-07",
      "publication_review": {
        "status": "human_reviewed",
        "approval_channel": "owner_conversation",
        "reviewer": "Mike Ye",
        "reviewed_at": "2026-10-07T16:31:20.553Z",
        "decision_id": "conversation-20261007-5332CD80C5A8",
        "packet_sha256": "25ef205031d894d3cf255e175595c7dd0dcac5947126991e7d9622f768a28677",
        "record_sha256": "7dd356d0763c504ba5ebe2128ad2c6fdc153e628731223874a25c5c649fd790a"
      }
    }
  ]
}
