{ "scope": "052-080 current static checks; historical runtime and visual records retained, not all rerun", "articles": [ { "index": 52, "directory": "2026-08-24_052_transformer_encoder_minimal_pytorch", "title": "复现一个 Transformer Encoder:从论文公式到最小 PyTorch", "quality_status": "passed", "cjk": 3226, "images": [ "images/transformer_encoder_method.png", "images/cover.png" ], "python_files": 1, "metadata_sha256": "1d0bb8a300c155a0305e05ffa219d8703528efae0c6a345e0037a2a60995b427", "prior_verification": { "syntax_check": "passed", "runtime_check": "not run: PyTorch is not installed in the current workspace Python environment", "image_generation": "passed: generated with the Codex built-in GPT-Image-2 channel; the cover received one targeted edit to remove pseudo-text", "visual_quality_check": "passed: both 1672x941 PNG files were inspected at original detail; the cover is text-free and the method figure has legible labels, balanced composition, and correct Transformer Encoder flow" } }, { "index": 53, "directory": "2026-08-25_053_attention_visualization_boundaries", "title": "Attention 可视化:热力图与解释边界", "quality_status": "passed", "cjk": 3033, "images": [ "images/attention_visualization_method.png", "images/cover.png" ], "python_files": 1, "metadata_sha256": "b00b4b02a335b9a3cd0a5473381ba89044d34172eeaa19d0600b4f9a17331d92", "prior_verification": { "source_check": "passed: primary paper pages, official BERTViz repository, Hugging Face documentation, and Captum documentation were opened on 2026-08-25", "syntax_check": "passed: PYTHONPYCACHEPREFIX=/private/tmp/codex_pycache_ai python3 -m py_compile code/attention_visualization_toy.py", "runtime_check": "passed: standard-library --check-only smoke test returned shape [8, 8], row_sums_close_to_one=true, and masked_pad_column_max=0.0", "image_generation": "passed: generated with the Codex built-in GPT-Image-2 channel; the cover received one targeted edit to remove tiny pseudo-labels and the method figure received one targeted edit to remove extra small labels/numbers", "visual_quality_check": "passed: both 1672x941 PNG files were inspected at high detail; the cover is text-free and the method figure has legible labels, no乱码, and correct TOKENS -> ATTENTION -> HEATMAP -> CHECKS -> CLAIMS flow" } }, { "index": 54, "directory": "2026-08-26_054_lora_qlora_memory_budget", "title": "LoRA / QLoRA:低秩更新、量化和显存预算", "quality_status": "passed", "cjk": 3001, "images": [ "images/lora_qlora_method.png", "images/cover.png" ], "python_files": 1, "metadata_sha256": "cd4f679c9ae33d506ed094d4a76eadf4143e18dbaa0ad7588d14b0bb833fd53e", "prior_verification": { "source_check": "passed: primary paper pages, official repositories, and official Hugging Face documentation were opened on 2026-08-26", "syntax_check": "passed: PYTHONPYCACHEPREFIX=/private/tmp/codex_pycache_ai python3 -m py_compile code/minimal_lora_qlora_budget.py", "runtime_check": "partial: python3 code/minimal_lora_qlora_budget.py --check-only passed using only the standard library and printed LoRA shapes plus rough memory budget; PyTorch toy training was not run because torch is not installed", "image_generation": "passed: cover.png and lora_qlora_method.png were generated with the Codex built-in GPT-Image-2/imagegen channel and copied to the article images directory; the method figure succeeded on the follow-up pure-text generation attempt after prior network failures", "visual_quality_check": "passed: both 1672x941 PNG files were inspected with view_image; the cover is text-free and the method figure has legible labels, no乱码/watermark, and correct FROZEN BASE -> LOW RANK -> ADAPTER -> QUANTIZE -> BUDGET -> CHECKS flow" } }, { "index": 55, "directory": "2026-08-27_055_dpo_data_loss_reward_free", "title": "DPO:数据格式、Loss 和“Reward-free”训练", "quality_status": "passed", "cjk": 3012, "images": [ "images/dpo_method.png", "images/cover.png" ], "python_files": 1, "metadata_sha256": "82c2613ac69f1332aabad3d142c4757baa085026499445eeb1bee447cd6c58f5", "prior_verification": { "source_check": "passed: primary paper pages, NeurIPS proceedings, the authors' reference repository, and official Hugging Face TRL documentation/source were opened on 2026-08-27", "syntax_check": "passed: PYTHONPYCACHEPREFIX=/private/tmp/codex_pycache_ai python3 -m py_compile code/minimal_dpo.py", "runtime_check": "partial: python3 code/minimal_dpo.py --check-only passed using only the standard library; loss decreased from 0.693147 to 0.415586 and mean implicit reward margin increased from 0.000000 to 0.663086; PyTorch toy and real language-model DPO fine-tuning were not run because torch is not installed", "image_generation": "passed: cover.png and dpo_method.png were generated with the Codex built-in GPT-Image-2/imagegen channel and copied to the article directory; the method figure succeeded during the follow-up run after prior network failures and received one targeted edit to correct the DPO loss curve; no CLI, SVG, or alternate image channel was used", "visual_quality_check": "passed: both 1672x941 PNG files were inspected with view_image; the cover is text-free, and the method figure has legible labels, no pseudo-text or watermark, a correct PAIR DATA -> TOKENIZE -> POLICY + REF -> LOG RATIOS -> DPO LOSS -> CHECKS flow, distinct policy/reference paths, and a DPO negative-log-sigmoid loss curve that decreases as reward margin increases" } }, { "index": 56, "directory": "2026-08-28_056_rag_baseline_retrieval_rerank_citations", "title": "RAG Baseline:BM25、Embedding、Rerank 与引用评测", "quality_status": "passed", "cjk": 3176, "images": [ "images/rag_baseline_method.png", "images/cover.png" ], "python_files": 1, "metadata_sha256": "396f186a7f826e93e8cfecea01f9b0a82d99224343171342a8933922316806eb", "prior_verification": { "source_check": "passed: primary paper/conference pages, official framework documentation, and author repositories were opened on 2026-08-28", "article_check": "passed: article has 3176 CJK characters and includes reproduction value, formulas/modules, official code reading route, minimal experiment, evaluation protocol, failure analysis, research questions, summary, and primary references", "syntax_check": "passed: PYTHONPYCACHEPREFIX=/private/tmp/codex_pycache_ai python3 -m py_compile code/minimal_rag_baseline.py", "runtime_check": "passed for the standard-library protocol: --check-only verified BM25 top-1, a normalized 256-dimensional deterministic feature vector, RRF tie behavior, and deliberately flawed citation metrics (validity 2/3, oracle precision 1/3, oracle recall 1/2); the six-document toy pipeline also ran with Recall@3=1.0 and MRR=1.0; the optional neural backend was not run because its models and dependency are not installed", "image_generation": "passed: cover.png and rag_baseline_method.png were generated with Codex built-in GPT-Image-2/imagegen and copied to the article directory. The cover succeeded during the first user retry; the method figure succeeded on the first pure-text call during the second user retry after earlier network failures. No CLI, SVG, code-drawn, or alternate image channel was used", "visual_quality_check": "passed: both project PNGs are 1672x941 and were inspected with view_image at original detail. The cover is text-free and shows a complete corpus -> parallel sparse/dense retrieval -> reranking gate -> cited evidence apparatus. The method figure has exactly seven correctly spelled headings, no pseudo-text or watermark, and a correct CORPUS -> parallel BM25/EMBEDDING -> FUSION -> RERANK -> ANSWER -> EVAL relationship with citation checks", "final_acceptance": "passed: metadata parses; topic, title, and paths are consistent; article has 3176 CJK characters and required sections; both relative image links resolve to 1672x941 PNGs near 16:9 and passed visual QA; code syntax and dependency-free smoke test pass; image_prompts.md records prompts, generation paths, visual conclusions, and retry history" } }, { "index": 57, "directory": "2026-08-29_057_dense_retrieval_dpr_contriever", "title": "Dense Retrieval:DPR / Contriever 双塔检索最小复现", "quality_status": "passed", "cjk": 3227, "images": [ "images/dense_retrieval_method.png", "images/cover.png" ], "python_files": 1, "metadata_sha256": "ae06b304508c4acc4abbe3bdf07ab8217522f1924b838812eb4fab1c23c86cc0", "prior_verification": { "source_check": "passed: primary paper/conference pages, official author repositories, FAISS documentation, and BEIR source were opened on 2026-08-29", "article_check": "passed: article has 3227 CJK characters and includes reproduction value, formulas with shapes, data/negative contract, official code reading route, minimal experiment, staged reproduction plan, evaluation protocol, failure analysis, research questions, summary, and primary references", "syntax_check": "passed: PYTHONPYCACHEPREFIX=/private/tmp/codex_pycache_ai python3 -m py_compile code/minimal_dense_retrieval.py", "runtime_check": "passed: dependency-free --check-only printed query vectors (6,24), passage index (10,24), train score matrix (6,6), loss 1.7899 -> 0.0044, Recall@1=1.0, Recall@3=1.0, MRR=1.0, and verified all six top-1 IDs; this is a protocol-level toy result on seen queries, not a DPR/Contriever benchmark reproduction", "image_generation": "passed: two final PNGs generated with Codex built-in GPT-Image-2/imagegen and copied into the article directory; no CLI, API key, SVG, or code-drawn substitute used", "visual_quality_check": "passed: both project PNGs are 1672x941 and inspected with view_image at original detail; cover is text-free and method figure has eight exact labels with correct dual-tower, offline-index, online-query, dot-product, and top-k relationships", "final_acceptance": "passed: metadata parses; exactly one 057 directory exists; topic, title, slug, and paths are consistent; article has 3227 CJK characters and all required sections; both relative image links resolve to 1672x941 PNGs near 16:9 and passed original-detail visual QA; primary source links were opened; code syntax and dependency-free smoke test pass; image_prompts.md records full prompts, generation channel, visual conclusions, and retry history" } }, { "index": 58, "directory": "2026-08-30_058_reranker_cross_encoder_hard_negative", "title": "Reranker:Cross-Encoder 与 Hard Negative 最小复现", "quality_status": "passed", "cjk": 3167, "images": [ "images/reranker_method.png", "images/cover.png" ], "python_files": 1, "metadata_sha256": "40a3f3bb9843529214fd9f33c68c201680ddb16e243bd5879f6698ce6d63379b", "prior_verification": { "source_check": "passed: primary papers, original author repository, official MS MARCO project, official Sentence Transformers documentation/source, and RocketQA paper/repository were opened on 2026-08-30", "article_check": "passed: article has 3167 CJK characters and includes reproduction value, formulas with shapes and variable meanings, official code reading route, minimal experiment, staged reproduction plan, hard-negative audit contract, evaluation protocol, failure analysis, research questions, summary, and primary references", "syntax_check": "passed: PYTHONPYCACHEPREFIX=/private/tmp/codex_pycache_ai python3 -m py_compile code/minimal_reranker.py", "runtime_check": "passed: dependency-free --check-only printed pair_features=(batch,43); easy-only BCE 0.6939->0.0998 and MRR 0.750; mixed hard-negative BCE 0.6930->0.0878 and MRR/NDCG@3 1.000; removing one positive produced candidate Recall@3=MRR=NDCG@3=0.750; these are protocol-level toy results, not monoBERT/MS MARCO reproduction", "image_generation": "passed: two project PNGs generated with Codex built-in GPT-Image-2/imagegen and copied into the article directory; no CLI, API key, SVG, or code-drawn substitute used", "visual_quality_check": "passed: both project PNGs are 1672x941 and inspected with view_image at original detail; cover is text-free and method figure has exactly eight unique labels with correct hard-negative training, joint pair scoring, retrieval-before-rerank, candidate ceiling, and metric relationships", "final_acceptance": "passed: metadata parses; exactly one 058 directory exists; topic, title, slug, and paths are consistent; article has 3167 CJK characters and all required sections; both relative image links resolve to 1672x941 PNGs near 16:9 and passed original-detail visual QA; primary source links were opened; code syntax and dependency-free smoke test pass; image_prompts.md records full prompts, generation channel, visual conclusions, and edit history" } }, { "index": 59, "directory": "2026-08-31_059_self_consistency_cost_accuracy", "title": "Self-Consistency:采样、投票和成本—准确率", "quality_status": "passed", "cjk": 3252, "images": [ "images/self_consistency_method.png", "images/cover.png" ], "python_files": 1, "metadata_sha256": "164c62cbde326e3cbbeee24a6253ded003ceb4237756085b9bee4083c3d98b54", "prior_verification": { "source_check": "passed: primary paper/venue pages, Google Research publication, NeurIPS paper, official GSM8K paper/repository/source, Google DeepMind OneTwo repository/release notes, ACL Adaptive-Consistency, ICLR Early-Stopping Self-Consistency, and Google DeepMind Universal Self-Consistency were opened on 2026-08-31", "article_check": "passed: article has 3398 CJK characters and includes reproduction value, formulas with assumptions and shapes/counts, source/code reading route, minimal experiment, real-model upgrade path, cost-aware evaluation, failure analysis, research questions, summary, primary references, and two relative image links", "syntax_check": "passed: PYTHONPYCACHEPREFIX=/private/tmp/codex_pycache_ai python3 -m py_compile code/minimal_self_consistency.py", "runtime_check": "passed: dependency-free --check-only produced diverse-errors accuracy 0.5435 at m=1 and 0.9520 at m=17, systematic-error accuracy 0.4219 at m=1 and 0.3356 at m=17, and binary independent q=0.60 theory 0.8789 at m=33; parser and assertions passed; these are controlled protocol simulations, not LLM/GSM8K results", "image_generation": "passed: two project PNGs were generated with Codex built-in GPT-Image-2/imagegen and copied into the article images directory; cover succeeded on the second current-run attempt after one network error, method figure succeeded on the third current-run attempt after two network errors; no CLI, API key, SVG, or code-drawn substitute was used", "visual_quality_check": "passed: both final project PNGs are 1672x941 and were inspected with view_image at original detail; cover is text-free with complete multi-path apparatus, vote aggregation, green consensus, and resource cylinders; method figure contains exactly eight unique labels with correct sample, extract, normalize, majority-vote, consensus, token-cost, and systematic-error relationships; no pseudo-text, logo, watermark, or crossed connectors", "final_acceptance": "passed: metadata parses; exactly one 059 directory exists; topic, title, slug, and paths are consistent; article has 3398 CJK characters and all required sections; both relative image links resolve to 1672x941 PNGs near 16:9 and passed original-detail visual QA; primary source links were opened; code syntax and dependency-free smoke test pass; image_prompts.md records complete prompts, built-in generation channel, retry history, output files, and visual conclusions" } }, { "index": 60, "directory": "2026-09-01_060_tool_calling_agent_react_verifiable_execution", "title": "Tool Calling Agent:ReAct、工具错误和可验证执行", "quality_status": "passed", "cjk": 3077, "images": [ "images/tool_agent_method.png", "images/cover.png" ], "python_files": 1, "metadata_sha256": "14fd583f7650eeea856c8418f6fe4a394402010f68287defa1a5d66bb35ea3c1", "prior_verification": { "source_check": "passed: ReAct paper/project/official code, BFCL methodology/V3/code, tau-bench paper/code, ToolSandbox research page/code, and ToolEmu code were opened on 2026-09-01", "article_check": "passed: article has 3230 CJK characters and includes reproduction value, ReAct action-space and state-transition formulas with variable meanings and trajectory shape semantics, official code reading route, deterministic minimal experiment, error taxonomy, four-layer evaluation protocol, failure analysis, research questions, summary, primary references, and two relative image links", "syntax_check": "passed: PYTHONPYCACHEPREFIX=/private/tmp/codex_pycache_ai python3 -m py_compile code/minimal_tool_agent.py", "runtime_check": "passed: dependency-free --check-only; verified retry reused one idempotency key and passed with one side effect, blind retry returned the correct answer but failed state verification with two side effects, schema repair and four-step loop guard passed", "image_generation": "passed: two project PNGs were generated with Codex built-in GPT-Image-2/imagegen and copied into the article images directory; cover succeeded on the third total attempt after two generations-endpoint network errors, method figure succeeded on the first attempt; no CLI, API key, SVG, or code-drawn substitute was used", "visual_quality_check": "passed: both final project PNGs are 1672x941 and were inspected with view_image at original detail; cover has a complete photorealistic planning/tool/fault/idempotency/state-verifier/audit apparatus with no text or watermark; method figure has exactly eight unique labels and correct goal-plan-call-validate-execute-observe loop, typed retry branch, step barrier, and answer-plus-state verification, with no extra text or crossed connectors", "final_acceptance": "passed: metadata parses; exactly one 060 directory exists; topic, title, slug, and paths are consistent; article has 3230 CJK characters and all required sections; two relative image links resolve to 1672x941 PNGs near 16:9 and passed original-detail visual QA; primary source links were opened; code syntax and dependency-free smoke test pass; image_prompts.md records complete prompts, built-in generation channel, retry history, output files, and visual conclusions" } }, { "index": 61, "directory": "2026-09-02_061_llm_as_judge_bias_rubric_consistency", "title": "LLM-as-a-Judge:偏差、Rubric 和一致性", "quality_status": "passed", "cjk": 3137, "images": [ "images/cover.png", "images/llm_judge_method.png" ], "python_files": 1, "metadata_sha256": "356767c87fed7645b9475380ec2f5df402c544465267a6a22747852d1172b05f", "prior_verification": { "source_check": "passed: primary papers, conference pages, official repositories, FastChat prompt definitions, pairwise runner, order swap mapping, and result logging were opened on 2026-09-02", "article_check": "passed: article has 3137 CJK characters and includes reproduction value, formulas with variable and tensor-shape meanings, primary code reading route, deterministic minimal experiment, bias map, rubric design, evaluation protocol, failure analysis, research questions, summary, primary references, and two relative image links", "syntax_check": "passed: PYTHONPYCACHEPREFIX=/private/tmp/codex_pycache_ai python3 -m py_compile code/minimal_llm_judge.py", "runtime_check": "passed: dependency-free --check-only on CPU with seed 61 and 120 synthetic pairs; deterministic assertions passed", "image_generation": "passed: two project PNGs generated with Codex built-in GPT-Image-2/imagegen only and copied into images/; no CLI, API key, SVG, or code drawing used", "visual_quality_check": "passed: both project PNGs are 1672x941 and inspected with view_image at original detail; final detailed findings recorded in image_prompts.md", "limitations": "toy protocol simulator only; no real LLM, human-label, MT-Bench, LLMBar, or public leaderboard result was reproduced; dynamic API behavior and current rankings require manual verification" } }, { "index": 62, "directory": "2026-09-04_062_long_context_position_sensitivity", "title": "Long Context 位置敏感性:复现 Lost-in-the-Middle", "quality_status": "passed", "cjk": 3158, "images": [ "images/long_context_method.png", "images/cover.png" ], "python_files": 1, "metadata_sha256": "d2db9888058b99d47f474433655956bf21f19d1dfdd3fd240a7ab28909e29f01", "prior_verification": { "source_check": "passed: TACL/ACL pages, original official repository and experiment/data-generation/evaluation paths, LongBench paper/repository, RULER paper/repository/run path, and Google Research publication page opened on 2026-09-04", "article_check": "passed: article has 3158 CJK characters and includes reproduction value, formulas with variable/tensor-shape meanings, official code reading route, deterministic minimal experiment, two-dimensional evaluation protocol, real-model replacement checklist, failure analysis, research questions, summary, primary references, and two valid relative image links", "syntax_check": "passed: PYTHONPYCACHEPREFIX=/private/tmp/codex_pycache_ai python3 -m py_compile code/minimal_position_sweep.py", "runtime_check": "passed: dependency-free CPU smoke test with fixed seed 62, three document counts, five positions, unique-target/index/range/position-gap/control assertions", "image_generation": "passed: two 1672x941 project PNGs generated with Codex built-in GPT-Image-2/imagegen only, using existing series images as style references; both first attempts succeeded and were copied into images/; no CLI, API key, SVG, or code drawing used", "visual_quality_check": "passed: both final project PNGs inspected with view_image at original detail; cover has no text or watermark and clearly depicts strong endpoint versus attenuated middle evidence detection; method figure has exactly eight specified labels and correctly moves one answer document through beginning, middle, and end while preserving a shared model/scorer and separate position-gap/task-slice reporting", "limitations": "controlled protocol simulator only; no real language model, NaturalQuestions, LongBench, RULER, GPU, or paper result was reproduced; dynamic model/API behavior and leaderboards require manual verification" } }, { "index": 63, "directory": "2026-09-04_063_prompt_injection_untrusted_context_isolation", "title": "Prompt Injection 防御:隔离不可信上下文的最小复现", "quality_status": "passed", "cjk": 3118, "images": [ "images/prompt_injection_method.png", "images/cover.png" ], "python_files": 1, "metadata_sha256": "a008c9abf47874040908788a103a4b0eb2516470be2cba6bc7e820ac18d28ac0", "prior_verification": { "source_check": "passed: original indirect-injection paper and demonstrations, Microsoft Research Spotlighting page, OpenAI Instruction Hierarchy page, USENIX StruQ paper and official repository, NeurIPS AgentDojo paper and official repository, CaMeL paper and Google Research artifact, and OWASP cheat sheet opened on 2026-09-04", "article_check": "passed: article has 3118 CJK characters and includes reproduction value, threat model, formulas with variable/shape meanings, official code reading route, deterministic minimal experiment, paired security/utility protocol, failure analysis, research questions, summary, primary references, and two valid relative image links", "syntax_check": "passed: PYTHONPYCACHEPREFIX=/private/tmp/codex_pycache_ai python3 -m py_compile code/minimal_isolation_gateway.py", "runtime_check": "passed: dependency-free CPU smoke test and --check-only assertions; plain-concat and delimiter-only ASR=1.000 with benign utility=1.000, isolated-gateway ASR=0.000 with benign utility=1.000", "image_generation": "passed: two project PNGs generated with Codex built-in GPT-Image-2/imagegen only; recovery cover generation succeeded and one targeted built-in edit removed document text; method generation succeeded on the second recovery attempt after one network error; no CLI, API key, SVG, screenshot, or code-drawn substitute used", "visual_quality_check": "passed: both final 1672x941 project PNGs inspected with view_image at original detail; cover contains no text, pseudo-text, logo, watermark or crop and clearly depicts quarantine plus an authorized one-way path; method figure contains exactly eight specified labels with correct trusted-task capability minting, untrusted-data isolation, policy branching, execution and state verification", "limitations": "controlled parser and capability-gateway protocol only; no real LLM, StruQ training, AgentDojo benchmark, CaMeL experiment, GPU run, or paper result was reproduced; dynamic models, APIs, leaderboards, and repository interfaces require manual verification" } }, { "index": 64, "directory": "2026-09-05_064_data_deduplication_contamination", "title": "数据去重与污染检测:从 n-gram 到语义候选的最小复现", "quality_status": "passed", "cjk": 3203, "images": [ "images/cover.png", "images/deduplication_method.png" ], "python_files": 1, "metadata_sha256": "dda1a93afec0784e04f8494d858eac991a9b3a327c8523a7a9ccae90f1ed6ac2", "prior_verification": { "checked_at": "2026-09-05T17:12:11+08:00", "cjk_characters": 3203, "cjk_main_before_references": 3130, "cjk_count_method": "regex U+4E00-U+9FFF over article.md; main count excludes references section", "required_sections": [ "摘要", "复现价值", "核心思想与公式", "官方代码阅读路线", "最小实验", "评测协议", "失败排查", "后续科研问题", "总结", "参考资料" ], "metadata_consistency": "passed", "unique_topic_directory": "passed", "relative_image_links": "passed", "png_crc_and_decompression": "passed", "image_visual_check": "passed via view_image(original) on both final project files", "primary_sources": "All article URLs opened with web tool, including explicit arXiv versions; titles/authors/versions checked", "syntax_check": "passed: PYTHONPYCACHEPREFIX=/private/tmp/codex-ai-064-pycache python3 -m py_compile code/minimal_dedup.py", "syntax_initial_issue": "system Python cache path denied; resolved by writable temporary cache without escalation", "cpu_smoke_test": "passed: Python 3.9.6, seed 64, handwritten synonym bag [9,26], 15 training pairs, 18 cross-split pairs", "results_file": "code/results.json", "stdout_file": "code/smoke_test.txt", "image_prompt_record": "complete; cover first generation passed; method one targeted edit passed", "limitations": [ "Toy hand-written synonym vectors, not pretrained semantic embeddings", "No paper training results, MinHash/LSH/suffix-array execution or benchmark metrics reproduced", "Optional local SentenceTransformer branch syntax-checked only; torch/numpy/sentence_transformers absent; runtime and package/model versions 待人工核验", "Controlled fixtures have no independent validation/test split; metrics are protocol observations", "Repository branches not commit-pinned; rerun interfaces 待人工核验" ] } }, { "index": 65, "directory": "2026-09-06_065_synthetic_data_quality_diversity", "title": "合成数据过滤:质量分类器、多样性与有效难例的最小复现", "quality_status": "passed", "cjk": 3148, "images": [ "images/synthetic_filter_method.png", "images/cover.png" ], "python_files": 1, "metadata_sha256": "821570c2a2560fbfc80786b5b2c2c05dcef18a742e2bdbc502275a75aeb2926f", "prior_verification": { "checked_at": "2026-09-06T17:14:18+08:00", "cjk_characters": 3148, "cjk_main_before_references": 3061, "cjk_count_method": "Regex U+4E00-U+9FFF; body additionally excludes references", "required_sections": [ "摘要", "复现价值", "核心思想与公式", "官方代码阅读路线", "最小实验", "评测协议", "失败排查", "后续科研问题", "总结", "参考资料" ], "metadata_consistency": "passed", "unique_topic_directory": "passed", "relative_image_links": "passed", "png_crc_and_decompression": "passed", "image_visual_check": "passed via view_image(original) on both final project files", "primary_sources": "All 13 article URLs opened and checked with web; OpenReview challenge replaced with arXiv camera-ready", "syntax_check": "passed: PYTHONPYCACHEPREFIX=/private/tmp/codex-ai-065-pycache python3 -m py_compile code/minimal_filter.py", "cpu_smoke_test": "passed: Python 3.9.6 CPU, seed 65, train/dev/pool [24,4]/[6,4]/[14,4], 12 eligible, budget 3", "results_file": "code/results.json", "stdout_file": "code/smoke_test.txt", "image_prompt_record": "complete; two generation calls and two targeted edits via built-in default imagegen route", "limitations": [ "Controlled shared templates; dev is not independent-domain validation", "Only structured answers checked; prose correctness not verified", "Difficulty is handwritten, not measured student loss", "No LLM generation, neural scorer, embedding or downstream SFT was run", "One screening seed only; no independent-domain test or training error bars", "Repository branches not commit-pinned; model execution and rerun compatibility 待人工核验" ], "primary_source_count": 13 } }, { "index": 66, "directory": "2026-09-07_066_knowledge_distillation_logits_responses", "title": "蒸馏:从教师软概率到 response 监督的最小复现", "quality_status": "passed", "cjk": 3195, "images": [ "images/cover.png", "images/distillation_method.png" ], "python_files": 2, "metadata_sha256": "fb4495450cbd217b26ccc910cebb3063edcc9feb3b5239a772d21e6fcdbf8b46", "prior_verification": { "cjk_characters": 3195, "cjk_main_before_references": 3062, "cjk_count_method": "Regex U+4E00-U+9FFF; main excludes reference section", "required_sections": [ "摘要", "复现价值", "核心思想与公式", "官方代码阅读路线", "最小实验", "评测协议", "失败排查", "后续科研问题", "总结", "参考资料" ], "syntax_check": "passed: Python 3.9.6 py_compile both scripts using writable cache", "cpu_smoke_test": "passed: standard-library learned teacher and student, seeds 66/67/68, all four variants; results.json and smoke_test.txt", "results_file": "code/results.json", "stdout_file": "code/smoke_test.txt", "primary_sources": "12 distinct article primary-source URLs opened with web; OpenReview browser challenge replaced by arXiv v4 PDF; fixed PyTorch 2.12 docs replaced empty stable redirect", "limitations": [ "Synthetic classification experiment, not LLM response generation/training or original paper benchmarks", "Teacher polynomial softmax 18 parameters, student linear softmax 9 parameters; not neural language models", "Three initialization seeds share fixed data; accuracy SD zero is not uncertainty estimate", "Teacher-hard branch lacks gold mixture, soft branches use alpha=0.7; not isolated causal comparison of soft vs hard", "No dev-set selection; preset temperatures and final step used", "System and bundled Python lack torch; optional tensor checks syntax-only, runtime 待人工核验", "Repository main not commit-pinned; official GPU training and dynamic tutorial compatibility 待人工核验" ], "image_visual_check": "passed via view_image on both final project PNGs", "image_prompt_record": "complete; prior failures preserved; recovery used two successful generations and one framing edit after one network failure per image", "checked_at": "2026-09-07T23:45:37+08:00", "relative_image_links": "passed: two relative project PNG links exist", "png_count": 2, "metadata_consistency": "passed", "unique_topic_directory": "passed", "primary_source_count": 12, "png_crc_and_decompression": "passed", "image_dimensions_and_ratio": "passed: both 1672x941", "code_recovery_check": "All code/data/result hashes unchanged from successful CPU run; both scripts AST syntax checked again" } }, { "index": 67, "directory": "2026-09-08_067_small_model_instruction_tuning", "title": "小模型指令微调:数据混合、模板与过拟合的最小复现", "quality_status": "passed", "cjk": 3144, "images": [ "images/instruction_tuning_method.png", "images/cover.png" ], "python_files": 2, "metadata_sha256": "d5fc2cead8b56299133539cbc650ea4a0e3576bce7ff3691f4d50ceb3e708539", "prior_verification": { "checked_at": "2026-09-08T17:15:52+08:00", "cjk_characters": 3144, "cjk_main_before_references": 3049, "cjk_count_method": "Regex U+4E00-U+9FFF", "required_sections": [ "摘要", "复现价值", "核心思想与公式", "官方代码阅读路线", "最小实验", "评测协议", "失败排查", "后续科研问题", "总结", "参考资料" ], "metadata_consistency": "passed: unique 067 directory, title/header/topic/slug and paths consistent", "relative_image_links": "passed", "png_count": 2, "image_dimensions_and_ratio": "passed: two 1672x941 PNGs", "png_decode_check": "passed: Pillow verify and full load", "image_visual_check": "passed: final two files inspected via view_image(original); text-free complete cover, eight-label accurate method paths", "image_prompt_record": "complete: two original generations, one targeted cover framing edit, no service failures", "primary_sources": "10 distinct primary-source URLs actually opened via web, including both full paper PDFs and two official source files", "syntax_check": "passed: both Python scripts py_compile and AST", "cpu_smoke_test": "passed: Python 3.12.14, NumPy 2.3.5, CPU; seeds67/68/69; 9 fine-tuning runs, 3 synthetic warm-ups", "results_file": "code/results.json", "stdout_file": "code/smoke_test.txt", "gradient_max_error": 2.5992027974375276e-11, "result_audit": "passed: 9 runs, 4 checkpoint candidates each, dev-only selection, all recorded EM recomputed from predictions", "limitations": [ "3516-parameter NumPy Elman RNN; synthetic full-sequence warm-up, not pretrained natural-language Transformer or paper benchmark reproduction", "Length-4 group splits8/4/4 with two tasks each; test has8 examples and4 groups; initialization SD is not population confidence interval", "COPY-only repeats8 unique rows, balanced uses16; independent data coverage differs", "Full-loss branch supervises192 tokens/step vs80 response-only; same input/steps, not equal supervised-token budget", "Template B markers occur in fixed vocabulary but were never input during warm-up/SFT; extreme distribution shift, not ordinary paraphrase test", "No torch in system or bundled Python; optional local_hf_sft.py syntax only; SmolLM2 and dependency compatibility 待人工核验", "Repository main and dynamic docs not commit/version-pinned; official training, distributed paths, packing and general benchmarks 待人工核验" ] } }, { "index": 68, "directory": "2026-09-09_068_quantized_inference_int8_int4", "title": "量化推理:INT8/INT4 的体积、困惑度与速度为什么不同步", "quality_status": "passed", "cjk": 3257, "images": [ "images/quantized_inference_method.png", "images/cover.png" ], "python_files": 2, "metadata_sha256": "e73371bbc8dd37bcdc75920f27ae1ecef38229db3f9f7ab9f88e6efc603cf0a3", "prior_verification": { "required_sections": [ "摘要", "复现价值", "核心思想与公式", "官方代码阅读路线", "最小实验", "评测协议", "失败排查", "后续科研问题", "总结", "参考资料" ], "syntax_check": "passed: Python py_compile for both scripts", "cpu_smoke_test": "passed: Python3.12.14 NumPy2.3.5 macOS arm64 CPU; 3 training seeds and12 model variants", "result_audit": "passed: separate implementation restores disk weights, recomputes4096 token NLLs for each variant, PPL, byte counts, dev checkpoint selection and timing medians; all integer codes roundtrip", "results_file": "code/results.json", "stdout_file": "code/smoke_test.txt", "audit_file": "code/audit_test.txt", "image_visual_check": "passed: final project files inspected with view_image(original); cover text-free, method eight labels and correct float baseline / integer packing / dequantization paths", "image_prompt_record": "complete: two distinct built-in generations, no retry; exact model not returned", "primary_sources": "14 cited URLs actually opened with web; no search-result-page citation", "source_recovery": "AWQ v5 PDF returned Internal Error; used accessible primary arXiv v6 HTML full text and official source instead; version change disclosed", "limitations": [ "Synthetic2256-parameter two-context NumPy MLP; no natural-language pretrained model or Transformer benchmark reproduction", "Symmetric RTN weights only; no LLM.int8 mixed-precision algorithm, GPTQ compensation or AWQ scale search implemented", "G32 is C-order flattened grouping and can cross matrix rows; not official input-channel grouping", "INT4 has15 symmetric levels, one unused encoding; FP32 scales and bias counted in payload; manifest/filesystem/container overhead excluded", "FP32 compute after dequantization; cached arrays9024 bytes, not peak process memory; packed and baseline arrays coexist", "CPU batch1 forward timing excludes disk loading, sampling and full autoregressive loop; no integer compute kernel, no GPU peak memory measurements", "Three seeds share same synthetic data split; SD is initialization/minibatch variation, not confidence interval; shared short contexts are not new-task generalization", "Official main/dynamic docs not pinned; hardware, CUDA, natural-language PPL and full-paper benchmark reproduction 待人工核验" ], "checked_at": "2026-09-09T17:13:59+08:00", "cjk_characters": 3257, "cjk_main_before_references": 3146, "cjk_count_method": "Regex U+4E00-U+9FFF", "metadata_consistency": "passed: title/header, topic index, unique directory and slug", "relative_image_links": "passed", "png_count": 2, "png_decode_check": "passed: Pillow verify and complete pixel load", "image_dimensions_and_ratio": "passed: two1672x941 PNGs", "primary_source_count": 14, "gradient_max_error": 1.520601700111257e-11, "final_acceptance": "passed" } }, { "index": 69, "directory": "2026-09-10_069_kv_cache_speculative_decoding", "title": "KV Cache 与推测解码:先验证输出,再讨论加速", "quality_status": "passed", "cjk": 3184, "images": [ "images/cover.png", "images/kv_cache_speculative_method.png" ], "python_files": 2, "metadata_sha256": "69a9009abc9c801d6764bda788845760ab0c6c2a9f07e38e6cd7c093c6777c6b", "prior_verification": { "checked_at": "2026-09-10T17:19:06+08:00", "cjk_characters": 3184, "cjk_main_before_references": 3121, "cjk_count_method": "Regex U+4E00-U+9FFF", "required_sections": [ "摘要", "复现价值", "核心思想与公式", "官方代码阅读路线", "最小实验", "评测协议", "失败排查", "后续科研问题", "总结", "参考资料" ], "metadata_consistency": "passed", "relative_image_links": "passed", "png_count": 2, "png_decode_dimensions_ratio": "passed: full Pillow decode;1672x941", "image_visual_check": "passed: final project files viewed with view_image(original)", "image_prompt_record": "complete:2 original generations,2 targeted cover edits; no other image channel", "primary_sources": "9 distinct cited primary URLs actually opened; bibliographic identities and versions checked", "syntax_check": "passed:both scripts py_compile after final edit", "cpu_smoke_test": "passed:Python3.12.14 NumPy2.3.5 macOS arm64 CPU float64", "result_audit": "passed:independent saved-trace/workload/median/residual-law audit", "results_file": "code/results.json", "stdout_file": "code/smoke_test.txt", "audit_file": "code/audit_test.txt", "development_failures_and_fixes": "verification.md:negative-control threshold,terminal cache invariant,audit fixture dtype", "limitations": [ "Untrained NumPy decoder on fixed integer tokens; no natural-language quality or paper benchmark reproduction", "MC checks fixed one-step3-category law only; full multistep loops checked by causal/logit/cache invariants", "Sequential single-machine timing includes prefill and draft overhead;7 repeats,not cross-hardware CI", "Cache bytes are array payload,not peak RSS or GPU memory", "Official Transformers/T5X/pretrained models,CUDA,low precision,RoPE,sliding windows,EOS,batch>1 and full-paper benchmarks 待人工核验" ], "final_acceptance": "passed" } }, { "index": 70, "directory": "2026-09-11_070_moe_routing_load_collapse", "title": "MoE 路由可视化:负载均衡损失为何不能证明专家健康", "quality_status": "passed", "cjk": 3174, "images": [ "images/cover.png", "images/moe_routing_method.png" ], "python_files": 2, "metadata_sha256": "804f580179dac87ca14a3208833668f77c2616dbdd8c60ecea26c9e763a7e6f3", "prior_verification": { "checked_at": "2026-09-11T17:17:31+08:00", "cjk_characters": 3174, "cjk_main_before_references": 3082, "cjk_count_method": "Regex U+4E00–U+9FFF", "required_sections": [ "摘要", "目录", "复现价值", "核心思想与公式", "官方代码阅读路线", "最小实验", "评测协议", "失败排查", "后续科研问题", "总结", "参考资料" ], "metadata_consistency": "passed", "relative_links": "passed", "png_count": 2, "png_full_decode_dimensions_ratio": "passed:1672x941,near16:9", "image_visual_check": "passed:cover text-free;method seven correct labels and separate raw/accepted/overflow paths", "image_prompt_record": "complete:1 cover generation;method initial + 1 edit + 1 regeneration;no network failures", "primary_sources": "7 distinct primary URLs actually opened and article citations mapped", "syntax_check": "passed:both Python scripts py_compile", "cpu_smoke_test": "passed:9 training runs;32-parameter gradients and dispatch checks", "result_audit": "passed:45 capacity evaluations;11520 independent scalar predictions;3-seed statistics", "html_heatmap": "576 rows in9 expandable tables;data/probability encoding inspected programmatically", "results_file": "code/results/summary.json", "stdout_file": "code/smoke_test.txt", "audit_file": "code/audit_test.txt", "limitations": [ "Synthetic regression32 parameters;not original language-model/paper benchmark reproduction", "No padding,attention,low precision,distributed/GPU/runtime performance tests;待人工核验", "3 seeds sample SD not confidence interval;finite-window underuse not proof of permanent collapse", "Live Mesh master not commit-pinned;Transformersv4.57.1 code read but framework not executed" ], "final_acceptance": "passed" } }, { "index": 71, "directory": "2026-09-12_071_clip_multimodal_alignment", "title": "CLIP 图文对齐:先核对正样本,再相信检索分数", "quality_status": "passed", "cjk": 3148, "images": [ "images/clip_alignment_method.png", "images/cover.png" ], "python_files": 2, "metadata_sha256": "cab7f3bb3f0059d1f82c98b58cd4922962732e2c63e7fa7d8ac8ddcd6ab82eba", "prior_verification": { "checked_at": "2026-09-12T17:12:38.625078+08:00", "cjk_characters": 3148, "cjk_main_before_references": 3013, "cjk_count_method": "Regex U+4E00–U+9FFF", "required_sections": [ "摘要", "目录", "复现价值", "核心思想与公式", "官方代码阅读路线", "最小实验", "评测协议", "失败排查", "后续科研问题", "总结", "参考资料" ], "metadata_consistency": "passed", "relative_links": "passed", "png_count": 2, "png_full_decode_dimensions_ratio": "passed:1672x941,near16:9", "image_visual_check": "passed:cover text-free;method8correct labels and correct independent train/retrieval paths", "image_prompt_record": "complete:2independent first-pass built-in calls,no generation retries", "primary_sources": "7distinct cited primary URLs actually opened with web", "syntax_check": "passed:2Python scripts", "cpu_smoke_test": "passed:6training runs and3initialization baselines", "result_audit": "passed:486queries,3645scalar cosines", "results_file": "code/results/summary.json", "stdout_file": "code/smoke_test.txt", "audit_file": "code/audit_test.txt", "limitations": [ "Synthetic RGB/6-word bag-of-words/linear encoders;not original CLIP pretraining or natural-image benchmark", "No pretrained model,torch,GPU,multicard,memory,latency orChinese evaluation;待人工核验", "OpenAI main not commit-pinned;commit retrieval failed;OpenCLIPv2.32.0 fixed tag", "Same9classes in train/test;no compositional holdout;3seeds sampleSD not confidence interval" ], "final_acceptance": "passed" } }, { "index": 72, "directory": "2026-09-13_072_vlm_ocr_chart_evaluation", "title": "VLM 读图评测:先审计评分器,再判断读错还是算错", "quality_status": "passed", "cjk": 3125, "images": [ "images/vlm_ocr_chart_method.png", "images/cover.png" ], "python_files": 4, "metadata_sha256": "cc450501945f0f6707680492b1a2dd5fc1816d58ca1c39dd444f75918b2f311c", "prior_verification": { "checked_at": "2026-09-13T17:13:57.264654+08:00", "cjk_characters": 3125, "cjk_main_before_references": 3002, "cjk_count_method": "Regex U+4E00–U+9FFF", "required_sections": [ "摘要", "目录", "复现价值", "核心思想与公式", "官方代码阅读路线", "最小实验", "评测协议", "失败排查", "后续科研问题", "总结", "参考资料" ], "metadata_consistency": "passed", "relative_links": "passed", "png_count": 2, "png_full_decode_dimensions_ratio": "passed:1672x941,near16:9", "image_visual_check": "passed:cover text-free;method7correct labels,isolated gold path", "image_prompt_record": "complete:2independent built-in calls;0retries", "primary_sources": "8distinct cited primary URLs actually opened", "syntax_check": "passed:2authored scripts py_compile;2source snapshots AST parse", "cpu_smoke_test": "passed:60test questions,5synthetic prediction modes", "result_audit": "passed:80gold answers,225numeric rows,961edit-distance pairs,11upstream metric edges", "external_prediction_cli_replay": "passed:synthetic predictions only", "results_file": "code/results/summary.json", "stdout_file": "code/smoke_test.txt", "audit_file": "code/audit_test.txt", "source_manifest": "code/source_snapshots/manifest.json", "limitations": [ "No actual OCR,VLM,images,pretrained weights,real benchmark or GPU inference executed;待人工核验", "Fault labels and predictions are constructed fixtures,not model failures", "Source snapshots are hash-identified but not commit-pinned;dynamic branches待人工核验", "Trailing-whitespace ANLS fixture applies to standalone evaluator;upstream generation strip prevents direct extrapolation", "No real benchmark impact estimate,confidence interval,Chinese,multipage,resolution or contamination validation" ], "final_acceptance": "passed" } }, { "index": 73, "directory": "2026-09-14_073_embodied_agent_gridworld", "title": "Embodied Agent 小环境:用状态日志解释成功、绕路与超时", "quality_status": "passed", "cjk": 3181, "images": [ "images/cover.png", "images/embodied_agent_method.png" ], "python_files": 2, "metadata_sha256": "640e05b99d3991f1cc8c2d89a1a4eb5bc9f4418fc94365f20ad30b8250414820", "prior_verification": { "checked_at": "2026-09-14T17:13:55.643280+08:00", "cjk_characters": 3181, "cjk_main_before_references": 3089, "cjk_count_method": "regex U+4E00–U+9FFF", "syntax_check": "passed:two authored scripts py_compile", "cpu_smoke_test": "passed:Python3.9.6 standard library;360episodes;35unique maps", "independent_audit": "passed:9508transitions;19016observations;84metrics;240prefix pairs", "primary_sources": "8distinct primary URLs actually opened;sourcev3.0.0", "image_visual_check": "passed:cover text-free;method7labels and correct isolated logging flow", "png_count": 2, "image_prompt_record": "complete:2independent generations;0retries", "results_file": "code/results/summary.json", "audit_file": "code/results/audit.json", "details": "verification.md", "limitations": [ "Synthetic navigation only;no learning,LLM,real robot,official MiniGrid runtime or benchmark;待人工核验", "Protocol revised after preliminary observations;teaching diagnostic set,not untouched generalization test", "Perfect localization,known goal,static walls,3x3nonoccluded categorical observation,four absolute actions", "One fixed derived RNG seed per map;no repeated policy seeds or confidence interval", "Source tag v3.0.0 not immutable commit hash;future compatibility待人工核验" ], "final_acceptance": "passed", "required_sections": [ "摘要", "目录", "复现价值", "核心思想与公式", "官方代码阅读路线", "最小实验", "评测协议", "失败排查", "后续科研问题", "总结", "参考资料" ], "metadata_consistency": "passed", "relative_links": "passed", "png_full_decode_dimensions_ratio": "passed:CRC,decompressed scanline sizes/filter bytes;1672x941" } }, { "index": 74, "directory": "2026-09-15_074_world_model_rollout_error", "title": "World Model 最小实验:单步预测准确,为什么 rollout 仍会漂移?", "quality_status": "passed", "cjk": 3371, "images": [ "images/world_model_method.png", "images/cover.png" ], "python_files": 2, "metadata_sha256": "5132d82a129657b0b88d490ad80f4a407615a422544eec8a14af2b2c6d3bb5ba", "prior_verification": { "checked_at": "2026-09-15T17:14:43.641519+08:00", "cjk_characters": 3371, "cjk_main_before_references": 3255, "cjk_count_method": "regex U+4E00-U+9FFF", "required_sections": [ "摘要", "目录", "复现价值", "核心思想与公式", "官方代码阅读路线", "最小实验", "评测协议", "失败排查", "后续科研问题", "总结", "参考资料" ], "metadata_consistency": "passed", "relative_links": "passed", "primary_sources": "8 cited primary URLs actually opened; current main/master not pinned", "syntax_check": "passed: 2 authored scripts py_compile", "cpu_smoke_test": "passed: Python3.12.14 NumPy2.3.5 CPU float64, 9 fitted models, seeds74/75/76", "independent_audit": { "status": "passed", "true_transitions_replayed": 34560, "predicted_state_vectors_replayed": 172800, "metric_values_recomputed": 4512, "unique_initial_states": 864, "max_prediction_abs_difference": 3.9968028886505635e-15, "sine_coefficients_match_true_dynamics": true }, "png_count": 2, "png_full_decode_dimensions_ratio": "passed: signature, all chunk CRCs, decompression, scanline lengths/filter bytes; both1672x941", "image_visual_check": "passed: individually view_image checked; no cover text; method7labels accurate flow", "image_prompt_record": "complete; two independent generations, one cover edit, no service failures", "limitations": [ "Synthetic deterministic state dynamics only; official neural frameworks/control/benchmarks not run, 待人工核验", "Known sine feature supplies correct function form; near-zero error is diagnostic only", "Perfect state correction increases observation budget", "Three seeds resample data; seed SD not confidence interval", "Official main/master not immutable pinned version; compatibility待人工核验" ], "details": "verification.md", "results_file": "code/results/summary.json", "audit_file": "code/results/audit.json", "final_acceptance": "passed" } }, { "index": 75, "directory": "2026-09-16_075_gnn_llm_graph_evidence", "title": "GNN + LLM:找对答案节点,为什么仍然缺少推理证据?", "quality_status": "passed", "cjk": 3157, "images": [ "images/gnn_llm_method.png", "images/cover.png" ], "python_files": 2, "metadata_sha256": "dcdf7ffa67c946cc64564d36f8a9b5047e769c7d2b0401ad8ac486827561241b", "prior_verification": { "details": "verification.md", "syntax_check": "passed: both authored Python scripts", "cpu_smoke_test": "passed: 3 CPU NumPy training runs; see code/smoke_test.txt", "independent_audit": { "status": "passed", "unique_graphs": 1344, "ground_truth_queries_rebuilt": 1344, "logit_values_rebuilt": 13824, "evidence_records_replayed": 3072, "max_logit_abs_error": 1.0658141036401503e-14, "witness": { "seed": 75, "query_id": 0, "source": 3, "relations": [ 0, 0 ], "candidates": [ 11, 6, 2 ], "gold": [ 11 ], "induced_answer": [], "missing_target": 11, "full_graph_paths": [ [ [ 3, 0, 5 ], [ 5, 0, 11 ] ] ] } }, "results_file": "code/results/summary.json", "limitations": [ "No LLM/PCST/embeddings/official benchmark/GPU executed; 待人工核验", "Bridge condition uses more triples; not equal-cost gain", "Two-hop function family supplied by architecture", "Three seed sample SD is not confidence interval", "Main branch unpinned; compatibility 待人工核验" ], "checked_at": "2026-09-16T17:17:32.762012+08:00", "cjk_characters": 3157, "cjk_main_before_references": 3083, "cjk_count_method": "regex U+4E00-U+9FFF", "required_sections": [ "摘要", "目录", "复现价值", "核心思想与公式", "官方代码阅读路线", "最小实验", "评测协议", "失败排查", "后续科研问题", "总结", "参考资料" ], "metadata_consistency": "passed", "relative_image_links": "passed", "primary_sources": "9 cited primary URLs actually opened", "png_count": 2, "png_full_decode_dimensions_ratio": "passed: all chunk CRCs, zlib decode, Pillow full pixel decode, 1672x941 each", "image_visual_check": "passed: each project PNG viewed with view_image(original); cover text-free, seven-label method accurate", "image_prompt_record": "complete: two independent first generations, no retries/edits", "final_acceptance": "passed" } }, { "index": 76, "directory": "2026-09-17_076_activation_patching_causal_evidence", "title": "Activation Patching:恢复答案之后,能否说找到了机制?", "quality_status": "passed", "cjk": 3169, "images": [ "images/activation_patching_method.png", "images/cover.png" ], "python_files": 3, "metadata_sha256": "2ad6cf3d91e0ccc2623db4ea0076c20c9e4395011d1551f859418a046ff596e5", "prior_verification": { "details": "verification.md", "checked_at": "2026-09-17T17:18:21.561933+08:00", "cjk_characters": 3169, "cjk_main_before_references": 3118, "cjk_count_method": "regex U+4E00-U+9FFF", "required_sections": [ "摘要", "目录", "复现价值", "核心思想与公式", "官方代码阅读路线", "最小实验", "评测协议", "失败排查", "后续科研问题", "总结", "参考资料" ], "metadata_consistency": "passed", "relative_image_links": "passed", "primary_sources": "8 cited primary URLs actually opened; versions and authors checked", "syntax_check": "passed: 3 scripts", "cpu_smoke_test": "passed: 3 NumPy training runs and control assertions", "independent_audit": { "status": "passed", "records_replayed": 24960, "max_abs_error": 8.881784197001252e-15 }, "results_file": "code/results/summary.json", "png_count": 2, "png_full_decode_dimensions_ratio": "passed: CRC/zlib/Pillow full pixel decode; both1672x941", "image_visual_check": "passed: both final project files individually viewed; cover text-free and complete; method seven labels and correct separate baselines", "image_prompt_record": "complete: two independent generations and one targeted edit each", "limitations": [ "No original LM/benchmark/GPU run; 待人工核验", "local_gpt2_patch.py syntax-only: torch and transformers absent", "Main sources not commit-pinned", "Diagnostic data share all 8 core patterns with training", "Circular shifted donors are weak controls", "3-seed SD not CI" ], "final_acceptance": "passed" } }, { "index": 77, "directory": "2026-09-18_077_linear_probing_causal_limits", "title": "Probing:线性探针读到了什么,模型就用到了什么吗?", "quality_status": "passed", "cjk": 3292, "images": [ "images/linear_probing_method.png", "images/cover.png" ], "python_files": 2, "metadata_sha256": "22fc03e1d2d534623d6a471a39e9766518ab64717937ff26e2e9d80ef6d12e52", "prior_verification": { "cjk_total": 3292, "cjk_before_references": 3233, "syntax": "passed", "cpu_smoke": "passed", "independent_audit": "passed", "unverified": [ "Official ELMo/NLP corpus execution", "Real-text group generalization", "Dynamic repository version compatibility", "Re-trained probe after projection" ], "scope": "Frozen locally trained toy model protocol, not paper benchmark reproduction", "checked_at": "2026-09-18T17:14:14.557900+08:00", "checks": { "json_title_index_slug": true, "unique_topic_directory": true, "cjk_count": true, "all_sections": true, "all_local_links_exist": true, "citations_to_opened_primary_sources": true, "two_pngs_decoded_dimensions": true, "visual_inspection": true, "code_syntax": true, "cpu_smoke": true, "independent_audit": true, "prompt_record": true }, "acceptance_record": "acceptance.json", "details": "verification.md", "experiment_summary": "code/results/summary.json", "audit_record": "code/results/audit.json", "max_audit_error": 1.021405182655144e-14 } }, { "index": 78, "directory": "2026-09-20_078_calibration_ece_abstention", "title": "Calibration:ECE 降低后,拒答阈值就可靠吗?", "quality_status": "passed", "cjk": 3284, "images": [ "images/calibration_method.png", "images/cover.png" ], "python_files": 2, "metadata_sha256": "0d682c2b6425f82bb5d771c656f500c6e1557c0dca71e358fe4e46c7ee51843a", "prior_verification": { "cjk_total": 3284, "cjk_before_references": 3159, "syntax": "passed", "cpu_smoke": "passed", "independent_audit": "passed", "scope": "Synthetic frozen analytic binary scorer; not neural training or paper benchmark reproduction", "details": "verification.md", "experiment_summary": "code/results/summary.json", "audit_record": "code/results/audit.json", "unverified": [ "Official neural model/framework/SGR execution", "Real datasets, GPU and natural domain shifts", "LLM free-form answer confidence", "Statistical risk guarantees", "Group conditional calibration", "Dynamic repository compatibility; master not commit-pinned" ], "checked_at": "2026-09-20T17:16:22.540501+08:00", "checks": { "unique_topic_directory": true, "json_title_index_slug": true, "cjk_count": true, "required_sections": true, "all_local_links_exist": true, "all_citations_to_opened_primary_sources": true, "png_crc_decode_dimensions_ratio": true, "image_links": true, "visual_inspection": true, "cover_full_preview_hash": true, "complete_prompt_records": true, "code_syntax": true, "cpu_smoke": true, "independent_audit": true, "runtime_logs": true }, "acceptance_record": "acceptance.json", "independent_scalar_prediction_rows": 61440, "max_audit_error": 1.5820687768286088e-10 } }, { "index": 79, "directory": "2026-09-21_079_adversarial_robustness_attack_audit", "title": "Adversarial Robustness:攻击没成功,模型就鲁棒吗?", "quality_status": "passed", "cjk": 3381, "images": [ "images/adversarial_robustness_method.png", "images/cover.png" ], "python_files": 2, "metadata_sha256": "8af4eb7b095cbc4decd946432ffa163c5568a5681be528de9b31d431b49d4725", "prior_verification": { "cjk_total": 3381, "cjk_before_references": 3220, "syntax": "passed", "cpu_smoke": "passed", "independent_audit": "passed", "scope": "Three synthetic 21-parameter ordinary logistic models; attack/protocol replication, not original benchmark", "details": "verification.md", "experiment_summary": "code/results/summary.json", "audit_record": "code/results/audit.json", "unverified": [ "Official TensorFlow/PyTorch networks, datasets, AutoAttack execution and GPU", "LLM inference and human review of template labels", "Real image or text robustness and semantic preservation", "Repository master compatibility and commit pinning", "Adversarial training and nonlinear-model certification" ], "checked_at": "2026-09-21T17:19:30.675853+08:00", "checks": { "unique_topic_directory": true, "metadata_title_index_slug": true, "cjk_count": true, "required_sections": true, "local_links": true, "all_citations_opened_primary_sources": true, "two_formal_pngs": true, "png_full_decode_crc_dimensions": true, "view_image_visual_qa": true, "cover_no_text": true, "method_correct_seven_labels": true, "complete_prompt_records": true, "python_syntax": true, "cpu_smoke": true, "independent_scalar_audit": true, "runtime_logs": true }, "acceptance_record": "acceptance.json", "scalar_candidate_predictions": 153600, "max_audit_error": 4.440892098500626e-15 } }, { "index": 80, "directory": "2026-09-22_080_paper_results_reporting", "title": "论文结果报告模板:让每个数字都能回到实验日志", "quality_status": "passed", "cjk": 3232, "images": [ "images/paper_reporting_method.png", "images/cover.png" ], "python_files": 2, "metadata_sha256": "23b91942d1f1c6a9dc9140f433a4c92553389c2a1e696699b3525d1c41b6f989", "prior_verification": { "cjk_total": 3232, "cjk_before_references": 3109, "syntax": "passed", "cpu_smoke": "passed", "independent_audit": "passed", "python": "3.9.6", "source_rows": 60, "aggregate_groups": 20, "bootstrap_mass": 27, "max_audit_error": 2.168404344971009e-19, "scope": "Replay of saved synthetic experiment summaries, not model retraining or original-paper benchmark reproduction", "unverified": [ "Original rliable/SciPy execution and repository commit compatibility", "Individual-prediction replay and retraining of 079 models in this run", "Population coverage of nominal bootstrap interval, large-seed or multi-task generalization", "Actual new quantitative error-bar plot; method diagram is schematic only" ], "checks": { "unique_topic_directory": true, "metadata_title_index_slug": true, "cjk_count": true, "required_sections": true, "local_links": true, "opened_primary_sources": true, "two_formal_pngs": true, "png_crc_dimensions_decompression": true, "source_image_byte_identity": true, "view_image_visual_qa": true, "cover_no_text": true, "method_correct_seven_labels": true, "complete_prompt_records": true, "python_syntax": true, "cpu_smoke": true, "independent_count_fraction_audit": true, "runtime_logs": true }, "checked_at": "2026-09-22T17:17:03.143688+08:00", "details": "verification.md", "acceptance_record": "acceptance.json" } } ] }