{"release":{"schemaVersion":"maha-epistemic-release/1.0","releaseId":"epirelease_c5d8cc9f35ca441d8bac4c9254775f82","releaseKind":"initial","status":"active","recordId":"urn:maha:record:mechanistic-interpretability-sae-encoder-decoder","domainSlug":"mechanistic-interpretability","targetSha256":"sha256:369c160f224159778ec08f586366d8c163e81afe67c769a0c5b9db269a01d54c","canonicalPath":"/knowledge/mechanistic-interpretability/mechanisms/mechanistic-interpretability-sae-encoder-decoder","canonicalVersion":"1.0.0","supersedesReleaseId":null,"approvals":[{"scope":"boundary-adequacy","reviewId":"epireview_736ab49dbc05451aa8422624397c21ba","reviewSha256":"sha256:99ffc5b549f2339c5dfc13f928fb6a008ed4be8fdecce49dbbf251409e780f95","reviewedAt":"2026-08-31T04:41:56.665Z","reviewerKind":"internal-editorial","reviewMethod":"Record-specific exact-revision review against the inspected source location, bounded claim, non-claims, rights basis, and source identity. No external reviewer participated."},{"scope":"domain-fidelity","reviewId":"epireview_736cbe6f7cb64d4b8dd7dea36c3783dd","reviewSha256":"sha256:829b601b8fc2a2cbbcd616cd6288b56248536e5f4db514057b5b1dd10561cbd4","reviewedAt":"2026-08-31T04:41:56.500Z","reviewerKind":"internal-editorial","reviewMethod":"Record-specific exact-revision review against the inspected source location, bounded claim, non-claims, rights basis, and source identity. No external reviewer participated."},{"scope":"rights-and-locator","reviewId":"epireview_b4fafaa1df9e46238dc2ad27883413a0","reviewSha256":"sha256:f1e5292ed0dc051b468f59c09021fdd6f277aa20a3097cb18136e0167dfc96f0","reviewedAt":"2026-08-31T04:41:56.923Z","reviewerKind":"internal-editorial","reviewMethod":"Record-specific exact-revision review against the inspected source location, bounded claim, non-claims, rights basis, and source identity. No external reviewer participated."},{"scope":"source-fidelity","reviewId":"epireview_2b020afefc17455785db3ae2a5a78f9b","reviewSha256":"sha256:8cc9fe14f341e10eecc10d9983effcbb56d3413b70fd8b6ccd2ab827caad4bfb","reviewedAt":"2026-08-31T04:41:56.303Z","reviewerKind":"internal-editorial","reviewMethod":"Record-specific exact-revision review against the inspected source location, bounded claim, non-claims, rights basis, and source identity. No external reviewer participated."}],"assuranceTier":"internally-reviewed-canonical","releaseAuthority":{"authoritySha256":"sha256:85146ab83a294271f963c43e4a7436cbc410d062e1238b141044553a508c8d81","attribution":"withheld-by-consent"},"publicChangeSummary":"Initial canonical release activates an already-compiled exact-revision substantial reference under the disclosed internal-review tier.","recordSha256":"sha256:79445a65081305ad694085c50fbdd859d0a04b3503ec55f5839a3e76d5576ee1","gateDecision":{"reasons":[],"recordId":"urn:maha:record:mechanistic-interpretability-sae-encoder-decoder","publicEligible":true,"evaluatedAgainst":"maha-epistemic/1.0"},"releasedAt":"2026-08-31T04:42:05.329Z","releaseSha256":"sha256:bc74fe4e9717b59eaffdfa8c93b4651ea8860b4bd19c50a295472e51a9897379","withdrawal":null},"provenance":{"schemaVersion":"maha-epistemic/1.0","evidencePolicyVersion":"mps/0.1","recordId":"urn:maha:record:mechanistic-interpretability-sae-encoder-decoder","canonicalPath":"/knowledge/mechanistic-interpretability/mechanisms/mechanistic-interpretability-sae-encoder-decoder","contentHash":"sha256:79445a65081305ad694085c50fbdd859d0a04b3503ec55f5839a3e76d5576ee1","generatedAt":"2026-08-31T04:42:05.329Z","publicationDecision":{"recordId":"urn:maha:record:mechanistic-interpretability-sae-encoder-decoder","publicEligible":true,"evaluatedAgainst":"maha-epistemic/1.0","reasons":[]},"claims":[{"id":"urn:maha:claim:mechanistic-interpretability-sae-encoder-decoder","scope":"Limited to Method, reconstruction and sparsity objectives, experiments, feature analysis, and limitations. in “Sparse Autoencoders Find Highly Interpretable Features in Language Models”; this candidate records the concept boundary and does not pool results from uncited systems or studies.","boundary":"SAE encoder decoder does not by itself establish system-level performance, safety, manufacturability, scalability, economic advantage, clinical benefit, or deployment readiness.","claimKind":"theoretical-model","sourceIds":["source-mechanistic-interpretability-sae"],"statement":"The cited source supports treating sae encoder decoder as a distinct mechanism within the stated mechanistic interpretability scope.","replication":{"asOfDate":"2026-08-24","assessment":"Independent replication and cross-platform transfer have not been compiled for this candidate; the evidence maturity refers only to the bounded source contract.","independentReplicationCount":null},"uncertainty":{"kind":"qualitative","statement":"No cross-source quantitative interval is asserted. Definitions, operating conditions, samples, instruments, and outcome measures must be checked against the exact cited locator during review."},"evidenceMaturity":"single-study"}],"sources":[{"id":"source-mechanistic-interpretability-sae","url":"https://arxiv.org/abs/2309.08600","title":"Sparse Autoencoders Find Highly Interpretable Features in Language Models","rights":{"note":"The candidate uses original boundary language and a short paraphrase linked to the cited source. No source passage, figure, or table is reproduced.","basis":"citation-with-paraphrase","quotationUsed":false},"authors":["Hoagy Cunningham","Aidan Ewart","Logan Riggs","Robert Huben","Lee Sharkey"],"boundary":"Sparse features and human labels do not establish completeness, unique decomposition, or causal faithfulness.","publisher":"arXiv","establishes":"The paper trains sparse autoencoders on language-model activations and evaluates specified reconstruction, sparsity, and interpretability properties.","identifiers":[{"value":"https://arxiv.org/abs/2309.08600","scheme":"url"}],"publishedAt":"2023-09-15","exactLocator":"Method, reconstruction and sparsity objectives, experiments, feature analysis, and limitations."}],"reviewEvents":[{"reviewId":"epireview_2b020afefc17455785db3ae2a5a78f9b","scope":"source-fidelity","targetSha256":"sha256:369c160f224159778ec08f586366d8c163e81afe67c769a0c5b9db269a01d54c","reviewedAt":"2026-08-31T04:41:56.303Z","verdict":"approve","supersedesReviewId":null},{"reviewId":"epireview_736cbe6f7cb64d4b8dd7dea36c3783dd","scope":"domain-fidelity","targetSha256":"sha256:369c160f224159778ec08f586366d8c163e81afe67c769a0c5b9db269a01d54c","reviewedAt":"2026-08-31T04:41:56.500Z","verdict":"approve","supersedesReviewId":null},{"reviewId":"epireview_736ab49dbc05451aa8422624397c21ba","scope":"boundary-adequacy","targetSha256":"sha256:369c160f224159778ec08f586366d8c163e81afe67c769a0c5b9db269a01d54c","reviewedAt":"2026-08-31T04:41:56.665Z","verdict":"approve","supersedesReviewId":null},{"reviewId":"epireview_b4fafaa1df9e46238dc2ad27883413a0","scope":"rights-and-locator","targetSha256":"sha256:369c160f224159778ec08f586366d8c163e81afe67c769a0c5b9db269a01d54c","reviewedAt":"2026-08-31T04:41:56.923Z","verdict":"approve","supersedesReviewId":null}]},"privacyBoundary":"Operational actor fingerprints, bearer credentials, private reviewer profiles, affiliations, conflicts, and non-consented authority identity fields are excluded."}