diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml index 4fa36d05a..a51d69a8a 100644 --- a/.github/workflows/ci.yml +++ b/.github/workflows/ci.yml @@ -102,26 +102,23 @@ jobs: cache-suffix: smoke - name: Set up Python ${{ env.PYTHON_VERSION }} run: uv python install ${{ env.PYTHON_VERSION }} - - name: Free up disk space + - name: Free up disk space when needed run: | - # Clear some space (https://github.com/actions/runner-images/issues/2840) - echo "Disk usage before cleanup:" - df -h - - # Remove unnecessary pre-installed software - sudo rm -rf /usr/share/dotnet - sudo rm -rf /opt/ghc - sudo rm -rf /usr/local/share/boost - sudo rm -rf /usr/local/lib/android - sudo rm -rf /opt/hostedtoolcache/CodeQL - sudo rm -rf /usr/local/.ghcup - sudo rm -rf /usr/share/swift - - # Clean up docker to ensure we start fresh + # Large runners normally have hundreds of GiB free. Keep the fallback for + # smaller runners, but avoid spending minutes deleting unused toolchains. + # Docker images and the host toolchains/venv may live on different + # filesystems, so both must have room before cleanup is skipped. + docker_root=$(docker info --format '{{.DockerRootDir}}') + available_kib=$(df -Pk / "$docker_root" | awk 'NR > 1 && (min == "" || $4 < min) {min = $4} END {print min}') + if (( available_kib >= 30 * 1024 * 1024 )); then + echo "Skipping cleanup: $((available_kib / 1024 / 1024)) GiB free on / and Docker storage" + exit 0 + fi + df -h / "$docker_root" + sudo rm -rf /usr/share/dotnet /opt/ghc /usr/local/share/boost \ + /usr/local/lib/android /opt/hostedtoolcache/CodeQL /usr/local/.ghcup /usr/share/swift docker system prune -af --volumes - - echo "Disk usage after cleanup:" - df -h + df -h / "$docker_root" - name: Install smoke test dependencies run: uv sync --only-group smoke --locked - name: Build Dockerfile diff --git a/sample-docs/layout-parser-paper.pdf.gz b/sample-docs/layout-parser-paper.pdf.gz deleted file mode 100644 index 7f05c02da..000000000 Binary files a/sample-docs/layout-parser-paper.pdf.gz and /dev/null differ diff --git a/scripts/smoketest.py b/scripts/smoketest.py index b89a62694..659e0179d 100644 --- a/scripts/smoketest.py +++ b/scripts/smoketest.py @@ -67,7 +67,7 @@ def send_document( (".md", "README.md", "text/markdown"), (".msg", "fake-email.msg", "application/x-ole-storage"), (".odt", "fake.odt", "application/vnd.oasis.opendocument.text"), - (".pdf", "layout-parser-paper.pdf", "application/pdf"), + (".pdf", "layout-parser-paper-fast.pdf", "application/pdf"), (".png", "english-and-korean.png", "image/png"), (".ppt", "fake-power-point.ppt", "application/vnd.ms-powerpoint"), ( @@ -89,12 +89,12 @@ def send_document( (".json", "spring-weather.html.json", "application/json"), ( ".gz", - "layout-parser-paper.pdf.gz", + "layout-parser-paper-fast.pdf", "application/gzip", ), ], ) -def test_happy_path_all_types(extension, example_filename: str, content_type: str): +def test_happy_path_all_types(extension, example_filename: str, content_type: str, tmp_path: Path): """ For the files in sample-docs, verify that we get a 200 and some structured response @@ -113,6 +113,12 @@ def test_happy_path_all_types(extension, example_filename: str, content_type: st pytest.skip("emulated hardware") test_file = str(Path("sample-docs") / example_filename) + if extension == ".gz": + # Gzip the one-page PDF at test time; full-document inference is covered + # by the table and strategy tests. + gzipped_file = tmp_path / f"{example_filename}.gz" + gzip_file(test_file, str(gzipped_file)) + test_file = str(gzipped_file) # Verify we can send with explicit content type response = send_document(filenames=[test_file], content_type=content_type)