Files
OSGKeyboard/ModelTraining/ClipboardSemantics/open-training-sources.json
T
Rocky d23a3275a8 docs(ai): record AI and clipboard semantics updates
- Document the smart reply center, clipboard semantic v6 specialization,
  personal style generation fix, and durable Apple account session
  work in both English and Simplified Chinese.
- Update clipboard semantics README and open-training sources to
  reflect the v6 boundary, blessing, and consensus-adjudication
  corpora that the new release-gate pipeline consumes.
2026-08-29 11:51:45 +08:00

241 lines
7.9 KiB
JSON

{
"baseCorpusRecords": 17692,
"combinedCorpusSHA256": "24bbec3fb9b3cc6434582d660b03b37401abe1339185405115646675ec3b654a",
"combinedRecords": 48779,
"excludedBaseOverlap": 6,
"excludedHoldoutOverlap": 425,
"excludedSources": [
{
"dataset": "MIDAS",
"reason": "Upstream deleted da_data in commit 509eb7e78b6661e54883fe3522ba60aaddcf69f2; deleted data is not resurrected for training."
},
{
"dataset": "DailyDialog / EmpatheticDialogues / Switchboard",
"reason": "Non-commercial license restriction."
},
{
"dataset": "NormDial / xLLMs Dialogue Greetings",
"reason": "No clear standard dataset license."
},
{
"dataset": "Schema-Guided Dialogue",
"reason": "CC-BY-SA-4.0 compatibility requires legal review."
},
{
"dataset": "Stanford Politeness Corpus",
"reason": "The downloadable archive does not include an explicit corpus license; source comments may carry ShareAlike obligations."
},
{
"dataset": "CPED",
"reason": "The Apache repository license does not establish commercial training rights for dialogue transcribed from copyrighted TV shows."
},
{
"dataset": "Tianji Wishes / Birthday Quotes",
"reason": "Dataset-level licenses do not provide a sufficiently clear per-record source or synthetic-generation rights chain."
},
{
"dataset": "Verified Enron Intent / GitHub issue holdout source",
"reason": "No clean official train split independent from the frozen holdout."
},
{
"dataset": "CFPB",
"reason": "No official train split; the source is already represented in frozen evaluation data and contains privacy-sensitive narratives."
},
{
"dataset": "CLINC150",
"reason": "Kept isolated from product training because its assistant intents are not aligned with the product-policy taxonomy."
},
{
"dataset": "openclaw-zh-greetings / WeChat-AutoSendBless",
"reason": "Repository licenses do not establish a sufficiently clear, versioned rights chain for the underlying message templates."
}
],
"openTrainingRecords": 31087,
"policy": "Licensed official training splits only. Per-record knownLabels prevent unannotated intents from becoming false negatives. Exact normalized text overlap with every local *holdout-corpus.jsonl file is excluded.",
"schemaVersion": 1,
"seed": 20260827,
"sources": [
{
"dataset": "AmazonScience/MASSIVE en-US",
"knownLabelCounts": {
"assistantCommand": 3658,
"domain": 9824,
"informationQuery": 5307,
"sentiment": 126,
"task": 358
},
"languages": {
"en": 10810
},
"license": "CC-BY-4.0",
"records": 10810,
"revision": "1.1",
"url": "https://huggingface.co/datasets/AmazonScience/massive"
},
{
"dataset": "AmazonScience/MASSIVE zh-CN",
"knownLabelCounts": {
"assistantCommand": 3272,
"domain": 9166,
"informationQuery": 5025,
"sentiment": 120,
"task": 348
},
"languages": {
"zh-Hans": 10099
},
"license": "CC-BY-4.0",
"records": 10099,
"revision": "1.1",
"url": "https://huggingface.co/datasets/AmazonScience/massive"
},
{
"dataset": "CrossWOZ",
"knownLabelCounts": {
"domain": 476,
"informationQuery": 393
},
"languages": {
"zh-Hans": 599
},
"license": "Apache-2.0",
"records": 599,
"revision": "df82c9fdff91b9b130f2d6b89110d3870ba6260e",
"url": "https://github.com/thu-coai/CrossWOZ/blob/df82c9fdff91b9b130f2d6b89110d3870ba6260e/data/crosswoz/train.json.zip"
},
{
"dataset": "FormosaNLU Synth v1",
"knownLabelCounts": {
"assistantCommand": 839,
"domain": 1747,
"informationQuery": 817,
"task": 119
},
"languages": {
"zh-Hans": 1988
},
"license": "CC-BY-4.0",
"records": 1988,
"revision": "03a337b61a200ab690994dca4dc31aa7f209800e",
"url": "https://huggingface.co/datasets/steven0226/formosa-nlu-synth-v1/resolve/03a337b61a200ab690994dca4dc31aa7f209800e/data/train.jsonl"
},
{
"dataset": "Google Research GoEmotions",
"knownLabelCounts": {
"sentiment": 799
},
"languages": {
"en": 1199
},
"license": "Apache-2.0",
"records": 1199,
"revision": "5d8f4ac97c873bde3a792ba4628f00bb9103d3e6",
"url": "https://github.com/google-research/google-research/blob/5d8f4ac97c873bde3a792ba4628f00bb9103d3e6/goemotions/data/train.tsv"
},
{
"dataset": "Google Taskmaster-1",
"knownLabelCounts": {
"domain": 631
},
"languages": {
"en": 631
},
"license": "CC-BY-4.0",
"records": 631,
"revision": "d92cb6af3005f1dc09c39e75e7daf4a04905e00b",
"url": "https://github.com/google-research-datasets/Taskmaster/tree/d92cb6af3005f1dc09c39e75e7daf4a04905e00b/TM-1-2019"
},
{
"dataset": "HLTCHKUST/BiToD",
"knownLabelCounts": {
"domain": 1339,
"informationQuery": 885,
"task": 355
},
"languages": {
"en": 681,
"zh-Hans": 658
},
"license": "Apache-2.0",
"records": 1339,
"revision": "a9bd74de9eecdc3d875cb4ebf6a6beaf9c30c2ff",
"url": "https://raw.githubusercontent.com/HLTCHKUST/BiToD/a9bd74de9eecdc3d875cb4ebf6a6beaf9c30c2ff/data/zh_train.json"
},
{
"dataset": "Meituan-Dianping/ASAP",
"knownLabelCounts": {
"complaint": 500,
"domain": 1000,
"sentiment": 1000
},
"languages": {
"zh-Hans": 1000
},
"license": "Apache-2.0",
"records": 1000,
"revision": "975122a60065240124df62cb4d5dbfd19ed9ef2c",
"url": "https://github.com/Meituan-Dianping/ASAP/blob/975122a60065240124df62cb4d5dbfd19ed9ef2c/data/train.csv"
},
{
"dataset": "MultiDoGO",
"knownLabelCounts": {
"domain": 464,
"informationQuery": 61,
"task": 83
},
"languages": {
"en": 464
},
"license": "CDLA-Permissive-1.0",
"records": 464,
"revision": "baa30639c4b271f394b81443c842193407cdf26d",
"url": "https://github.com/awslabs/multi-domain-goal-oriented-dialogues-dataset/blob/baa30639c4b271f394b81443c842193407cdf26d/data/paper_splits/splits_annotated_at_turn_level/airline/train.tsv"
},
{
"dataset": "PolyAI MInDS-14 zh-CN",
"knownLabelCounts": {
"assistantCommand": 163,
"domain": 480,
"informationQuery": 179
},
"languages": {
"zh-Hans": 480
},
"license": "CC-BY-4.0",
"records": 480,
"revision": "40ce77cb32a384e4d50a568e1ec39ac804019d33",
"url": "https://huggingface.co/datasets/PolyAI/minds14/resolve/40ce77cb32a384e4d50a568e1ec39ac804019d33/zh-CN/train-00000-of-00001.parquet"
},
{
"dataset": "PolyAI RESTAURANTS-8K",
"knownLabelCounts": {
"domain": 986
},
"languages": {
"en": 986
},
"license": "CC-BY-4.0",
"records": 986,
"revision": "57ec275d8078af65b7731c2a98be812d844a6d6b",
"url": "https://raw.githubusercontent.com/PolyAI-LDN/task-specific-datasets/57ec275d8078af65b7731c2a98be812d844a6d6b/span_extraction/restaurant8k/train_0.json"
},
{
"dataset": "SNIPS NLU Benchmark",
"knownLabelCounts": {
"assistantCommand": 499,
"domain": 1492,
"informationQuery": 743,
"task": 250
},
"languages": {
"en": 1492
},
"license": "CC0-1.0",
"records": 1492,
"revision": "b86ac7f1577868c42158d0dec77db50956046696",
"url": "https://raw.githubusercontent.com/sonos/nlu-benchmark/b86ac7f1577868c42158d0dec77db50956046696/2017-06-custom-intent-engines/BookRestaurant/train_BookRestaurant_full.json"
}
],
"unavailableSources": []
}