{
  "story_id": "f0cfe2d5a3daac798a05d4212e57cc0c",
  "desk": "drm3",
  "revision": 1,
  "published_at": "2026-09-03T11:19:19.784Z",
  "content_hash": "c60d65591a8de5f660b1256c69a580060ad833f8490c93840dd26bb0b28de65f",
  "hash_basis": "sha256 over `headline\\ndek\\nprose`, plus `\\n` + the canonical citations JSON when any source is placed, plus `\\n#blog` for blogs",
  "basis": {
    "headline": "Perplexity to Open Source Faster Lily AI Engine for Apple Silicon",
    "dek": "Perplexity plans to release its specialized Lily AI engine for Apple Silicon as open source.",
    "prose": "Perplexity has built a local artificial intelligence engine designed specifically for Apple silicon and the Qwen3.6-35B-A3B model. [^1]\n\nPerplexity says it plans to release Lily as open source, but the code is not available yet. [^2]\n\nPerplexity says Lily averaged 23 percent faster prompt processing and 35 percent faster token generation than MLX-LM on an M5 Max MacBook Pro with 128GB of unified memory. [^3]\n\nThe engine, called Lily, uses a Rust runtime and custom Metal kernels, with neither PyTorch nor MLX in its execution path. [^4]\n\nFor a 203 GiB model (Llama-4-Scout), weights loading from S3 dominates startup time at approximately 423 seconds (92%), while torch.compile takes only 34 seconds (8%). [^5]\n\nSubsequent launches on the same node for a 64 GiB model were reduced from 82 seconds to 16 seconds after configuration changes. [^6]",
    "cited": "[{\"statement\":\"Perplexity has built a local artificial intelligence engine designed specifically for Apple silicon and the Qwen3.6-35B-A3B model.\",\"source\":\"slashdot.org\",\"instrument\":\"News\",\"claim_key\":null,\"published_at\":\"2026-09-03T11:19:19.784Z\",\"publisher_count\":1,\"sources\":[\"slashdot.org\"]},{\"statement\":\"Perplexity says it plans to release Lily as open source, but the code is not available yet.\",\"source\":\"slashdot.org\",\"instrument\":\"News\",\"claim_key\":null,\"published_at\":\"2026-09-03T11:19:19.784Z\",\"publisher_count\":1,\"sources\":[\"slashdot.org\"]},{\"statement\":\"Perplexity says Lily averaged 23 percent faster prompt processing and 35 percent faster token generation than MLX-LM on an M5 Max MacBook Pro with 128GB of unified memory.\",\"source\":\"slashdot.org\",\"instrument\":\"News\",\"claim_key\":null,\"published_at\":\"2026-09-03T11:19:19.784Z\",\"publisher_count\":1,\"sources\":[\"slashdot.org\"]},{\"statement\":\"The engine, called Lily, uses a Rust runtime and custom Metal kernels, with neither PyTorch nor MLX in its execution path.\",\"source\":\"slashdot.org\",\"instrument\":\"News\",\"claim_key\":null,\"published_at\":\"2026-09-03T11:19:19.784Z\",\"publisher_count\":1,\"sources\":[\"slashdot.org\"]},{\"statement\":\"For a 203 GiB model (Llama-4-Scout), weights loading from S3 dominates startup time at approximately 423 seconds (92%), while torch.compile takes only 34 seconds (8%).\",\"source\":\"Amazon Web Services\",\"instrument\":\"News\",\"claim_key\":null,\"published_at\":\"2026-09-01T15:48:15.000Z\",\"publisher_count\":1,\"sources\":[\"Amazon Web Services\"]},{\"statement\":\"Subsequent launches on the same node for a 64 GiB model were reduced from 82 seconds to 16 seconds after configuration changes.\",\"source\":\"Amazon Web Services\",\"instrument\":\"News\",\"claim_key\":null,\"published_at\":\"2026-09-01T15:48:15.000Z\",\"publisher_count\":1,\"sources\":[\"Amazon Web Services\"]}]",
    "kind": "news"
  },
  "receipt_verify": "Ed25519 over the dot-joined string `slice_hash.cursor_from.cursor_to.view.view_version.row_count`; public_key and sig are base64url of the raw 32-byte key / 64-byte signature",
  "receipt": null,
  "receipt_note": "this revision predates receipt-keeping (before v0.37.0); the filed row lives in the record",
  "generation_chain": {
    "wire": {
      "stream": "fountain_news",
      "story_id": "91b7152f1d76b8bde35abf3ea6858400",
      "thread_id": "55cd5a465c77287cbe3934b79d70d937",
      "thread_label": "Qwen3.6-35B-A3B",
      "novelty": "UPDATE",
      "content_hash": "6d2ee9835f28b3eb827bcf8b14547b57d7a34bea403606e9e48f7142276f5842",
      "last_published_at": "2026-09-03T11:19:19.784Z",
      "read_receipt": {
        "slice_hash": "256dcdd60dfda8cfb8e689b27a44613fca55ab97f7151dbaf5c2c233deded06f",
        "cursor_from": "eyJ0cyI6IjIwMjYtMDktMDNUMTA6NDc6NTkuMDAwMDAwWiIsImlkIjoiNDQxNWY3MmVjNjI3YWFmZmYyOWNkMjU2NGE5MWIxMDAiLCJ2IjoiMSJ9",
        "cursor_to": "eyJ0cyI6IjIwMjYtMDktMDNUMTE6Mzg6NTYuMDAwMDAwWiIsImlkIjoiYTVkZWJkMWExN2NhY2MyZjlkZjJkMzZhOTk0OGNhODkiLCJ2IjoiMSJ9",
        "view": "v_fountain_news",
        "view_version": "1",
        "row_count": 100,
        "window_days": 3,
        "bytes_scanned": 11903793,
        "credits": 8,
        "price_per_100_rows": 8,
        "sig": "n-y37kFH7D_G1v511Ny_UEmV9ikJegt_-kDvpQVgnaqt6CypMMLVCaNja31RAB87Tt_8TcM82LQAx4bSO1o2Aw",
        "public_key": "bMUigy8O0jOnBxQ4Sc-5lwhIZ8LQVAhxMbR7qESVuUE",
        "signer_path": "lakehouse/data-extract/v1",
        "alg": "Ed25519",
        "signed": true
      }
    },
    "written_at": "2026-09-03T12:21:15.008Z"
  },
  "cited_facts": [
    {
      "statement": "Perplexity has built a local artificial intelligence engine designed specifically for Apple silicon and the Qwen3.6-35B-A3B model.",
      "source": "slashdot.org",
      "instrument": "News",
      "claim_key": null,
      "published_at": "2026-09-03T11:19:19.784Z",
      "publisher_count": 1,
      "sources": [
        "slashdot.org"
      ]
    },
    {
      "statement": "Perplexity says it plans to release Lily as open source, but the code is not available yet.",
      "source": "slashdot.org",
      "instrument": "News",
      "claim_key": null,
      "published_at": "2026-09-03T11:19:19.784Z",
      "publisher_count": 1,
      "sources": [
        "slashdot.org"
      ]
    },
    {
      "statement": "Perplexity says Lily averaged 23 percent faster prompt processing and 35 percent faster token generation than MLX-LM on an M5 Max MacBook Pro with 128GB of unified memory.",
      "source": "slashdot.org",
      "instrument": "News",
      "claim_key": null,
      "published_at": "2026-09-03T11:19:19.784Z",
      "publisher_count": 1,
      "sources": [
        "slashdot.org"
      ]
    },
    {
      "statement": "The engine, called Lily, uses a Rust runtime and custom Metal kernels, with neither PyTorch nor MLX in its execution path.",
      "source": "slashdot.org",
      "instrument": "News",
      "claim_key": null,
      "published_at": "2026-09-03T11:19:19.784Z",
      "publisher_count": 1,
      "sources": [
        "slashdot.org"
      ]
    },
    {
      "statement": "For a 203 GiB model (Llama-4-Scout), weights loading from S3 dominates startup time at approximately 423 seconds (92%), while torch.compile takes only 34 seconds (8%).",
      "source": "Amazon Web Services",
      "instrument": "News",
      "claim_key": null,
      "published_at": "2026-09-01T15:48:15.000Z",
      "publisher_count": 1,
      "sources": [
        "Amazon Web Services"
      ]
    },
    {
      "statement": "Subsequent launches on the same node for a 64 GiB model were reduced from 82 seconds to 16 seconds after configuration changes.",
      "source": "Amazon Web Services",
      "instrument": "News",
      "claim_key": null,
      "published_at": "2026-09-01T15:48:15.000Z",
      "publisher_count": 1,
      "sources": [
        "Amazon Web Services"
      ]
    }
  ],
  "note": "A signature proves who filed this and that it has not changed since. It never makes a claim true."
}