From 8ab1c3db9964ae68db625b4689c7cd85607203c7 Mon Sep 17 00:00:00 2001 From: Cooki-e <1291852105@qq.com> Date: Tue, 22 Sep 2026 19:23:12 +0800 Subject: [PATCH] Add bingyu filtered ANN task submission --- task-submissions/bingyu/1-x-1/.gitignore | 7 + task-submissions/bingyu/1-x-1/README.md | 111 ++++++ task-submissions/bingyu/1-x-1/SOURCES.md | 54 +++ task-submissions/bingyu/1-x-1/assets.json | 104 ++++++ .../bingyu/1-x-1/environment/.dockerignore | 3 + .../bingyu/1-x-1/environment/Dockerfile | 20 + .../1-x-1/environment/docker-compose.yaml | 15 + .../environment/docs/available_resources.md | 9 + .../1-x-1/environment/docs/environment.md | 84 +++++ task-submissions/bingyu/1-x-1/instruction.md | 123 ++++++ task-submissions/bingyu/1-x-1/task.toml | 51 +++ .../bingyu/1-x-1/tests/.dockerignore | 3 + .../bingyu/1-x-1/tests/Dockerfile | 21 ++ task-submissions/bingyu/1-x-1/tests/checks.py | 49 +++ .../1-x-1/tests/configure_trajectory_judge.py | 46 +++ .../1-x-1/tests/data/ground_truth.jsonl | 10 + .../bingyu/1-x-1/tests/data/queries.jsonl | 10 + .../bingyu/1-x-1/tests/data_io.py | 64 ++++ .../bingyu/1-x-1/tests/docker-compose.yaml | 15 + .../bingyu/1-x-1/tests/finalize_reward.py | 55 +++ task-submissions/bingyu/1-x-1/tests/grader.py | 352 ++++++++++++++++++ .../1-x-1/tests/jailbreak_judge/codex.toml | 42 +++ .../bingyu/1-x-1/tests/offline_exec.py | 59 +++ task-submissions/bingyu/1-x-1/tests/test.sh | 99 +++++ .../bingyu/1-x-1/tests/test_contract.py | 292 +++++++++++++++ 25 files changed, 1698 insertions(+) create mode 100644 task-submissions/bingyu/1-x-1/.gitignore create mode 100644 task-submissions/bingyu/1-x-1/README.md create mode 100644 task-submissions/bingyu/1-x-1/SOURCES.md create mode 100644 task-submissions/bingyu/1-x-1/assets.json create mode 100644 task-submissions/bingyu/1-x-1/environment/.dockerignore create mode 100644 task-submissions/bingyu/1-x-1/environment/Dockerfile create mode 100644 task-submissions/bingyu/1-x-1/environment/docker-compose.yaml create mode 100644 task-submissions/bingyu/1-x-1/environment/docs/available_resources.md create mode 100644 task-submissions/bingyu/1-x-1/environment/docs/environment.md create mode 100644 task-submissions/bingyu/1-x-1/instruction.md create mode 100644 task-submissions/bingyu/1-x-1/task.toml create mode 100644 task-submissions/bingyu/1-x-1/tests/.dockerignore create mode 100644 task-submissions/bingyu/1-x-1/tests/Dockerfile create mode 100644 task-submissions/bingyu/1-x-1/tests/checks.py create mode 100644 task-submissions/bingyu/1-x-1/tests/configure_trajectory_judge.py create mode 100644 task-submissions/bingyu/1-x-1/tests/data/ground_truth.jsonl create mode 100644 task-submissions/bingyu/1-x-1/tests/data/queries.jsonl create mode 100644 task-submissions/bingyu/1-x-1/tests/data_io.py create mode 100644 task-submissions/bingyu/1-x-1/tests/docker-compose.yaml create mode 100644 task-submissions/bingyu/1-x-1/tests/finalize_reward.py create mode 100644 task-submissions/bingyu/1-x-1/tests/grader.py create mode 100644 task-submissions/bingyu/1-x-1/tests/jailbreak_judge/codex.toml create mode 100644 task-submissions/bingyu/1-x-1/tests/offline_exec.py create mode 100755 task-submissions/bingyu/1-x-1/tests/test.sh create mode 100644 task-submissions/bingyu/1-x-1/tests/test_contract.py diff --git a/task-submissions/bingyu/1-x-1/.gitignore b/task-submissions/bingyu/1-x-1/.gitignore new file mode 100644 index 0000000..ca574f7 --- /dev/null +++ b/task-submissions/bingyu/1-x-1/.gitignore @@ -0,0 +1,7 @@ +/data/ +/models/ +/jobs/ +/solution/ +__pycache__/ +.pytest_cache/ +*.py[cod] diff --git a/task-submissions/bingyu/1-x-1/README.md b/task-submissions/bingyu/1-x-1/README.md new file mode 100644 index 0000000..53babf2 --- /dev/null +++ b/task-submissions/bingyu/1-x-1/README.md @@ -0,0 +1,111 @@ +# Faiss 属性过滤 ANN 检索后端 + +作者:bingyu。投稿编号:`task-1-x-1`。Implementation / CPU,版本 0.8.1。 +本任务从既有任务迁移;v0.8.0 将公开和隐藏 query 分别扩为 10 条,逐条计分后取平均。 + +Agent 从空的 `/app` 开始,实现数据读取、Faiss ANN 索引、tag 过滤、持久化与 +逐条 JSON 查询服务,通过 `build.sh`、`run.sh` 接入独立验证器。 + +## 任务约定 + +- 1,000 万个 192 维 uint8 向量,200,386 个 tag,距离为原始 squared L2。 +- tag 条件按 AND 组合;每条评分查询返回精确 Top-10,边界等距邻居可互换。 +- 10 条公开验证查询、10 条隐藏评测查询;每组四条单 tag、四条双 tag、两条三 tag。 +- 资源分配见 `task.toml`:8 CPU、16 GiB RAM、64 GiB 存储、无 GPU;建库 600 秒。 +- 服务启动后等待 `ready`,加载阶段不设独立时限,由整个检索评测的 3900 秒总超时兜底。 +- `ready` 后每条查询从发送请求到响应解码最多 15 ms,包含第一条查询;启动耗时不计入查询耗时。 +- 同一组 10 条隐藏查询在索引搬迁重载后再执行一次,共 20 次调用,按 10 条计分。 +- 每条 query 两轮均正确、均在 15 ms 内且答案一致得 1,否则得 0;最终 reward 为十条分数的平均值。 +- 单条错误或超时只扣该条分数,后续继续。通过 7 条即 0.7(70 分);公开 query 不计分。 +- 建库、索引、加载及独立轨迹审查仍是整体门槛,未通过则最终得分为 0。 + +完整接口和执行时限见 [instruction.md](instruction.md)。查询数量与执行时限直接写在 +评分器中;CPU、内存、存储配置统一保留在 `task.toml`。v0.8.1 将建库时限改为 +600 秒,移除独立的限制配置文件、进程 RSS 门槛和 12 GiB 索引大小门槛。 + +v0.8.1 移除独立限制配置后的全量隐藏集复验:建库 120.74 秒(新时限 600 秒), +10 条查询及重载后的 20 次请求全部通过,最慢 5.67 ms。11 项评分回归及真实 +子进程异常/隔离检查通过,三条失败时仍正确得到 0.7 的检索分。模型轨迹审查未运行; +该离线运行检索分为 1,最终综合 reward 为 0。记录位于 +`jobs/task-submissions/bingyu/1-x-1/simplify-limits/`(fork 仓库内的本地忽略目录)。 + +## 数据与隔离 + +本地运行数据在 `data/`;`assets.json` 记录全部 9 个输入文件的大小、SHA-256、 +HF 来源和固定版本。数据共 2,865,730,488 字节,不提交到 Git。 +来源、衍生方式和逐类许可见 [SOURCES.md](SOURCES.md)。 + +公开开发数据已发布到 +[Cooki-e/search-swe-development](https://huggingface.co/datasets/Cooki-e/search-swe-development)。 +`assets.json` 已固定实际 HF 提交版本,并记录逐文件大小和 SHA-256。 +从仓库根目录恢复并核验: + +```bash +python scripts/download_assets.py --task-path task-submissions/bingyu/1-x-1 +python scripts/download_assets.py --task-path task-submissions/bingyu/1-x-1 --verify-only +``` + +Agent 只读挂载公开 corpus、validation、example 与环境说明。Verifier 只挂载 corpus +和环境说明,评测查询与标签放在其镜像的 `tests/data/`。参考实现保留在作者的本地 +验证副本中,不属于这个投稿包,也不进入 Agent 镜像或数据集。 + +只有 `/app` 和真实记录的 `/logs/agent/trajectory.json` 传递给 verifier。 +提交代码以无特权用户运行,不能创建 Internet socket,不获得任何模型凭据; +轨迹审查在提交进程停止后单独运行。 + +## 检查与运行 + +Python 3.12+,仓库宿主依赖和 Docker/Compose 的安装要求见 +[贡献指南](../../../docs/contributing.md)。从仓库根目录执行: + +```bash +python scripts/check_submission.py task-submissions/bingyu/1-x-1 +python scripts/check_release.py +git diff --check +``` + +评分器回归测试在 verifier 镜像中运行,其中已包含 Faiss、NumPy 和 SciPy: + +```bash +docker build -t search-swe-local:bingyu-1-x-1-verifier task-submissions/bingyu/1-x-1/tests +docker run --rm --network none --cpus 2 --memory 2g \ + search-swe-local:bingyu-1-x-1-verifier \ + /opt/conda/bin/python -B -m unittest discover -s /tests -p test_contract.py -v +``` + +真实 Agent 运行使用仓库启动器。模型 ID 按实际账号替换;先查看 dry-run: + +```bash +python scripts/run_task.py --task-path task-submissions/bingyu/1-x-1 \ + --agent codex --model MODEL_ID --dry-run +``` + +Agent 使用 `AGENT_OPENAI_BASE_URL` / `AGENT_OPENAI_API_KEY`;轨迹审查使用独立的 +`VERIFIER_OPENAI_BASE_URL` / `VERIFIER_OPENAI_API_KEY`,对应源任务已有的 +`gpt-5.6-sol` Codex 审查模型及兼容 Responses 的服务端点。填写变量后,真实运行 +需去掉 `--dry-run` 并指定新的 `--output jobs/...`。不要提交本机 `.env`。 + +Verifier 镜像在构建时安装固定版本 `harbor-rewardkit==0.1.7`,评测期间不安装依赖。 +缺少凭据、真实轨迹或审查失败时,最终得分为零。离线检索通过不等于含轨迹审查的 +最终评测通过。迁移的设计、执行记录和未运行检查保存在本地忽略目录 +`jobs/task-submissions/bingyu/1-x-1/migration/`。 + +旧版 v0.7.1 的本地迁移验证中,两套镜像构建、9 项评分器回归测试和实际提交进程 +隔离/异常检查通过。作者参考解在完整 1,000 万向量上通过 5 条评测查询及重载后的 +10 次请求,最慢 5.58 ms,建库 136.36 秒。该离线运行没有真实 Agent 轨迹和模型 +审查凭据,因此检索部分得分 1、最终综合得分 0;真实 Agent 和轨迹审查仍待验证。 + +v0.8.0 已在完整 1,000 万向量、8 CPU / 16 GiB 的容器中独立复验公开与隐藏两组: +每组 10 条 query、两轮共 20 次请求全部通过。公开最慢 6.02 ms, +隐藏最慢 7.15 ms;两组检索分均为 1。10 项评分回归测试通过, +真实子进程的超时、坏 JSON 和异常退出注入得到预期的 7/10 = 0.7。 +本轮未调用模型审查 API,没有生成真实 Agent 轨迹;因此离线最终综合 reward 为 0, +不代表已完成含轨迹审查的端到端评测。 + +新版十条查询的构造与评分验证记录在本地忽略目录 +`jobs/task-submissions/bingyu/1-x-1/ten-query-scoring/`。 + +## 审核后发布 + +保持当前临时编号,正式编号由维护者分配。官方数据迁移、正式编号替换和最终 +GitHub 合并按同一个 PR 的流程完成;个人开发数据在官方固定版本可下载之前保留。 diff --git a/task-submissions/bingyu/1-x-1/SOURCES.md b/task-submissions/bingyu/1-x-1/SOURCES.md new file mode 100644 index 0000000..04afd03 --- /dev/null +++ b/task-submissions/bingyu/1-x-1/SOURCES.md @@ -0,0 +1,54 @@ +# Input provenance and publication + +The corpus is the YFCC-10M filtered-search dataset released for the NeurIPS 2023 +Big-ANN competition. It contains precomputed vectors and tag metadata, not the +original photographs. The competition's [dataset table](https://big-ann-benchmarks.com/neurips23.html) +and [upstream README](https://github.com/harsha-simhadri/big-ann-benchmarks/blob/main/neurips23/README.md) +list this release under [CC BY 4.0](https://creativecommons.org/licenses/by/4.0/). +Credit belongs to the dataset creators and NeurIPS 2023 Big-ANN Filter Track +organizers; upstream references and original filenames are preserved below. + +| Local file | Original source | Transformation | +| --- | --- | --- | +| `data/corpus/vectors.u8bin` | `https://dl.fbaipublicfiles.com/billion-scale-ann-benchmarks/yfcc100M/base.10M.u8bin` | Byte-identical copy under a task-local name | +| `data/corpus/metadata.spmat` | `https://dl.fbaipublicfiles.com/billion-scale-ann-benchmarks/yfcc100M/base.metadata.10M.spmat` | Byte-identical copy under a task-local name | +| `data/corpus/config.json` | Task-authored description of the upstream file headers | Records dimensions, dtypes, distance and tag counts | +| `data/validation/queries.jsonl` | Task-selected vectors and tag predicates derived from YFCC-10M inputs | Ten frozen public cases; query IDs are task-local | +| `data/validation/ground_truth.jsonl` | Task-author computation from the corpus and public queries | Exact filtered Top-10 with original-vector squared L2 | +| `data/example/*` | Task-authored small synthetic example | 64 documents and 17 functional queries, independent of the scored corpus | + +The original query sources are `query.public.100K.u8bin` and +`query.metadata.public.100K.spmat` under the same upstream URL prefix. Query +selection originated in the original task's authoring workspace. Version 0.8.0 +retains the original five cases per split and adds five from that workspace's +existing development/holdout candidate pool, using tag diversity before new +latency measurements. All twenty source rows, vectors and predicates are distinct; +each split has four single-tag, four double-tag and two triple-tag queries. Query +vectors are unchanged upstream rows. Exact labels for all cases were independently +recomputed by exhaustive float64 squared L2 over every eligible original vector. +The author-only provenance record retains source rows and selection details. + +The original author recorded these full-corpus SHA-256 checksums; the migrated +copies have been checked against them: + +- Vectors: `589030afcbcc44a50ff798cec935f0980815a02eb7075b224d4f7c12febe96bf` +- Metadata: `2f9c9a533fde31cb062edfdf5c9410f088bb5d7bf4a721ab98c47e3d2954287d` + +`assets.json` records actual byte sizes and SHA-256 for all nine runtime input +files. The development dataset is `Cooki-e/search-swe-development`, under +`development/bingyu/1-x-1/`; use only the immutable revision recorded in +`assets.json` after publication. + +Upstream CC BY 4.0 terms and attribution apply to the original dataset and its +derived inputs. The synthetic example and task-authored metadata are published +with the task author's authorization; no additional standalone license for +those original files was supplied. Their downstream license should be settled +with the author before official redistribution. The development dataset card +therefore points to these per-file terms instead of claiming one license for +every file. This document does not change the repository's code license. + +Verifier-only queries and labels remain in `tests/data/` and are excluded from +the public input dataset and the agent image/mounts. Files under `tests/data/` +will still be visible to readers if this task code is published on GitHub; +runtime isolation is not a promise of private repository storage. Reference +solutions, private credentials and job outputs are excluded from publication. diff --git a/task-submissions/bingyu/1-x-1/assets.json b/task-submissions/bingyu/1-x-1/assets.json new file mode 100644 index 0000000..9933b42 --- /dev/null +++ b/task-submissions/bingyu/1-x-1/assets.json @@ -0,0 +1,104 @@ +{ + "schema_version": 1, + "files": [ + { + "path": "data/corpus/config.json", + "size_bytes": 194, + "sha256": "6ea9278f9b46ff0cc2dc2b0d61741224aa9a91c1e147268aaec04c2714b5ef57", + "source": { + "repo_id": "Cooki-e/search-swe-development", + "repo_type": "dataset", + "revision": "d82ef21a21cecab959c77a7133e0ab00146a8318", + "filename": "development/bingyu/1-x-1/corpus/config.json" + } + }, + { + "path": "data/corpus/metadata.spmat", + "size_bytes": 945683840, + "sha256": "2f9c9a533fde31cb062edfdf5c9410f088bb5d7bf4a721ab98c47e3d2954287d", + "source": { + "repo_id": "Cooki-e/search-swe-development", + "repo_type": "dataset", + "revision": "d82ef21a21cecab959c77a7133e0ab00146a8318", + "filename": "development/bingyu/1-x-1/corpus/metadata.spmat" + } + }, + { + "path": "data/corpus/vectors.u8bin", + "size_bytes": 1920000008, + "sha256": "589030afcbcc44a50ff798cec935f0980815a02eb7075b224d4f7c12febe96bf", + "source": { + "repo_id": "Cooki-e/search-swe-development", + "repo_type": "dataset", + "revision": "d82ef21a21cecab959c77a7133e0ab00146a8318", + "filename": "development/bingyu/1-x-1/corpus/vectors.u8bin" + } + }, + { + "path": "data/example/ground_truth.jsonl", + "size_bytes": 3197, + "sha256": "faa8d8576a226952824699125796d3807c6fbf0d94404d8df0d033eaf7e3da30", + "source": { + "repo_id": "Cooki-e/search-swe-development", + "repo_type": "dataset", + "revision": "d82ef21a21cecab959c77a7133e0ab00146a8318", + "filename": "development/bingyu/1-x-1/example/ground_truth.jsonl" + } + }, + { + "path": "data/example/metadata.spmat", + "size_bytes": 1680, + "sha256": "b5a27083b6f29606130de3a2f5fdc85101acb6003aded8d177163b15d1770351", + "source": { + "repo_id": "Cooki-e/search-swe-development", + "repo_type": "dataset", + "revision": "d82ef21a21cecab959c77a7133e0ab00146a8318", + "filename": "development/bingyu/1-x-1/example/metadata.spmat" + } + }, + { + "path": "data/example/queries.jsonl", + "size_bytes": 16221, + "sha256": "5628db7064338187f250ae4ee6fbbefa80d313d164d2ef6cf222089ad3a9df30", + "source": { + "repo_id": "Cooki-e/search-swe-development", + "repo_type": "dataset", + "revision": "d82ef21a21cecab959c77a7133e0ab00146a8318", + "filename": "development/bingyu/1-x-1/example/queries.jsonl" + } + }, + { + "path": "data/example/vectors.u8bin", + "size_bytes": 12296, + "sha256": "252eaaa1883e3f06248e6ac725fa5199b3e7343b99e886fa299e75a7042b22c0", + "source": { + "repo_id": "Cooki-e/search-swe-development", + "repo_type": "dataset", + "revision": "d82ef21a21cecab959c77a7133e0ab00146a8318", + "filename": "development/bingyu/1-x-1/example/vectors.u8bin" + } + }, + { + "path": "data/validation/ground_truth.jsonl", + "size_bytes": 2840, + "sha256": "cdedf6d38d12193103afa62695242dde237dace90e6524fc7f338bed975e517f", + "source": { + "repo_id": "Cooki-e/search-swe-development", + "repo_type": "dataset", + "revision": "d82ef21a21cecab959c77a7133e0ab00146a8318", + "filename": "development/bingyu/1-x-1/validation/ground_truth.jsonl" + } + }, + { + "path": "data/validation/queries.jsonl", + "size_bytes": 10212, + "sha256": "3d088f2e857f94282d78e623f87d94333da8bace1a5454139b6adf0f5ef8fe60", + "source": { + "repo_id": "Cooki-e/search-swe-development", + "repo_type": "dataset", + "revision": "d82ef21a21cecab959c77a7133e0ab00146a8318", + "filename": "development/bingyu/1-x-1/validation/queries.jsonl" + } + } + ] +} diff --git a/task-submissions/bingyu/1-x-1/environment/.dockerignore b/task-submissions/bingyu/1-x-1/environment/.dockerignore new file mode 100644 index 0000000..97176c4 --- /dev/null +++ b/task-submissions/bingyu/1-x-1/environment/.dockerignore @@ -0,0 +1,3 @@ +**/__pycache__ +**/*.pyc +**/.pytest_cache diff --git a/task-submissions/bingyu/1-x-1/environment/Dockerfile b/task-submissions/bingyu/1-x-1/environment/Dockerfile new file mode 100644 index 0000000..fec3009 --- /dev/null +++ b/task-submissions/bingyu/1-x-1/environment/Dockerfile @@ -0,0 +1,20 @@ +FROM docker.io/hanhainebula/search-swe-base:cpu-py3.12-1.0.0 + +ARG SEARCH_SWE_CODEX_VERSION=0.147.0 +ARG SEARCH_SWE_CLAUDE_CODE_VERSION=2.1.273 +ARG SEARCH_SWE_PI_VERSION=0.85.1 +RUN npm install --global --ignore-scripts \ + --registry=https://registry.npmmirror.com \ + "@openai/codex@${SEARCH_SWE_CODEX_VERSION}" \ + "@earendil-works/pi-coding-agent@${SEARCH_SWE_PI_VERSION}" \ + && npm install --global \ + --registry=https://registry.npmmirror.com \ + "@anthropic-ai/claude-code@${SEARCH_SWE_CLAUDE_CODE_VERSION}" \ + && codex --version | grep -Fx "codex-cli ${SEARCH_SWE_CODEX_VERSION}" \ + && test "$(pi --version)" = "${SEARCH_SWE_PI_VERSION}" \ + && test "$(claude --version)" = "${SEARCH_SWE_CLAUDE_CODE_VERSION} (Claude Code)" +RUN claude --help | grep -F "(low, medium, high, xhigh, max)" >/dev/null + +RUN mkdir -p /app /task/data /task/docs /logs + +WORKDIR /app diff --git a/task-submissions/bingyu/1-x-1/environment/docker-compose.yaml b/task-submissions/bingyu/1-x-1/environment/docker-compose.yaml new file mode 100644 index 0000000..829e7e6 --- /dev/null +++ b/task-submissions/bingyu/1-x-1/environment/docker-compose.yaml @@ -0,0 +1,15 @@ +services: + main: + volumes: + - type: bind + source: ../data + target: /task/data + read_only: true + bind: + create_host_path: false + - type: bind + source: ./docs + target: /task/docs + read_only: true + bind: + create_host_path: false diff --git a/task-submissions/bingyu/1-x-1/environment/docs/available_resources.md b/task-submissions/bingyu/1-x-1/environment/docs/available_resources.md new file mode 100644 index 0000000..18523f6 --- /dev/null +++ b/task-submissions/bingyu/1-x-1/environment/docs/available_resources.md @@ -0,0 +1,9 @@ +# Available Resources + +The task runtime has no public network access. Use only the supplied vectors, +task data, installed Python packages, and system tools. Do not download models, +datasets, labels, result mappings, packages, or remote retrieval results. + +The coding agent's own model connection is a Harbor phase-scoped exception and +does not grant the submitted search system access to that model or to the +internet. The final system must run entirely from local task inputs. diff --git a/task-submissions/bingyu/1-x-1/environment/docs/environment.md b/task-submissions/bingyu/1-x-1/environment/docs/environment.md new file mode 100644 index 0000000..3195386 --- /dev/null +++ b/task-submissions/bingyu/1-x-1/environment/docs/environment.md @@ -0,0 +1,84 @@ +# CPU Docker Environment + +This image provides a Conda-managed Python 3.12 environment with CPU-only PyTorch and the common search, embedding, indexing, document-processing, media, HTTP, and service packages used by the tasks. Installed Python packages include `torch`, `torchvision`, `torchcodec`, `numpy`, `transformers`, `sentence-transformers`, `FlagEmbedding`, `deepspeed`, `faiss-cpu`, `bm25s`, `rank-bm25`, `pyserini`, `hnswlib`, `qdrant-client`, `docling`, `marker-pdf`, `pdf2image`, `pypdfium2`, `CairoSVG`, `av`, `imageio`, `imageio-ffmpeg`, `tiktoken`, `fastapi`, `uvicorn`, `python-multipart`, `requests`, `aiohttp`, `openai`, and `pydantic-settings`. + +The task Python interpreter and its installed packages are available at `/opt/conda/bin/python`. Use `/opt/conda/bin/python` and `/opt/conda/bin/pip` when invoking Python or installing packages. + +The image also includes JDK 21, FFmpeg, Poppler utilities, Cairo, Git, curl, `jq`, `build-essential`, `ca-certificates`, `libffi`, `libgomp`, `netbase`, `netcat`, `procps`, `tzdata`, `unzip`, and the related system runtime libraries. + +## Dataset file formats + +The full corpus is under `/task/data/corpus/`. The small public example under +`/task/data/example/` uses the same binary formats. All multibyte fields are +**little-endian**, and fields and arrays are stored consecutively with no padding. +Read dimensions from the file headers so the same reader works on both corpora. + +### `vectors.u8bin` + +| Byte offset | Type and count | Meaning | +| --- | --- | --- | +| 0 | `uint32`, 1 value | Number of document vectors, `N` | +| 4 | `uint32`, 1 value | Vector dimension, `d` | +| 8 | `uint8`, `N * d` values | Vectors in row-major order, shape `(N, d)` | + +Document ID `i` is the zero-based row `i`. The file size is `8 + N * d` bytes. +The full corpus has `N = 10000000`, `d = 192`, and a file size of +`1920000008` bytes. Cast vector values to `float32` or `float64` before +subtraction and squaring to avoid uint8 overflow when computing squared L2 distances. + +### `metadata.spmat` + +This file stores a document-by-tag matrix in compressed sparse row (CSR) format. +`N` is the number of documents, `T` the number of possible tags, and `nnz` the +number of stored entries. + +| Byte offset | Type and count | Meaning | +| --- | --- | --- | +| 0 | `int64`, 1 value | Number of document rows, `N` | +| 8 | `int64`, 1 value | Number of tag columns, `T` | +| 16 | `int64`, 1 value | Number of stored entries, `nnz` | +| 24 | `int64`, `N + 1` values | CSR row pointers, `indptr` | +| `24 + 8 * (N + 1)` | `int32`, `nnz` values | Zero-based tag IDs, `indices` | +| `24 + 8 * (N + 1) + 4 * nnz` | `float32`, `nnz` values | Entry values, `data` | + +For document `i`, its entries occupy the half-open interval +`[indptr[i], indptr[i + 1])`. A tag in `indices` belongs to that document only +when the corresponding value in `data` is nonzero. Missing or zero-valued entries +mean the tag is absent. `indptr[0] = 0` and `indptr[N] = nnz`; document IDs match +the vector rows above. + +The file size is `24 + 8 * (N + 1) + 8 * nnz` bytes. The full corpus has +`N = 10000000`, `T = 200386`, `nnz = 108210476`, and a file size of +`945683840` bytes. + +### Minimal read-only loading example + +This example maps the arrays without loading the entire corpus into RAM. +In the submitted build entry point, use the paths supplied through `--vectors` +and `--metadata` instead of assuming fixed paths. + +```python +from pathlib import Path +import numpy as np + +root = Path("/task/data/corpus") # Use /task/data/example for the small corpus. +vector_path = root / "vectors.u8bin" +metadata_path = root / "metadata.spmat" + +n, d = map(int, np.fromfile(vector_path, dtype="