Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
66 changes: 66 additions & 0 deletions .github/workflows/code_interpreter_tests.yml
Original file line number Diff line number Diff line change
@@ -0,0 +1,66 @@
name: Test Code Interpreter

on:
workflow_call:
inputs:
E2B_DOMAIN:
required: false
type: string
default: ''
E2B_TESTS_TEMPLATE:
required: true
type: string
run_recovery_tests:
required: false
type: boolean
default: true
secrets:
E2B_API_KEY:
required: true

permissions:
contents: read

jobs:
test:
defaults:
run:
working-directory: ./tests
shell: bash
name: Code Interpreter - HTTP tests
runs-on: ubuntu-22.04
steps:
- name: Checkout repository
uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 # v4.4.0

- name: Parse .tool-versions
uses: wistia/parse-tool-versions@32f568a4ffd4bfa7720ebf93f171597d1ebc979a # v2.1.1
with:
filename: '.tool-versions'
uppercase: 'true'
prefix: 'tool_version_'

- name: Install uv
uses: astral-sh/setup-uv@d0cc045d04ccac9d8b7881df0226f9e82c39688e # v6.8.0
with:
version: '${{ env.TOOL_VERSION_UV }}'
python-version: '${{ env.TOOL_VERSION_PYTHON }}'
enable-cache: true

- name: Install dependencies
run: uv sync --locked

- name: Run tests
run: |
if [[ "${{ inputs.run_recovery_tests }}" == "true" ]]; then
uv run pytest --verbose --numprocesses=1
else
uv run pytest --verbose --numprocesses=1 \
--ignore=test_interrupt.py \
--ignore=test_systemd.py \
--deselect=test_async.py::test_async_interrupt
fi
env:
E2B_API_KEY: ${{ secrets.E2B_API_KEY }}
E2B_DOMAIN: ${{ inputs.E2B_DOMAIN }}
E2B_TESTS_TEMPLATE: ${{ inputs.E2B_TESTS_TEMPLATE }}
4 changes: 2 additions & 2 deletions .github/workflows/lint.yml
Original file line number Diff line number Diff line change
Expand Up @@ -48,9 +48,9 @@ jobs:
enable-cache: true

- name: Install Python dependencies
working-directory: chart_data_extractor
run: |
uv sync --locked
(cd chart_data_extractor && uv sync --locked)
(cd tests && uv sync --locked)
uv tool install ruff==0.11.12

- name: Run linting
Expand Down
10 changes: 9 additions & 1 deletion .github/workflows/pull_request.yml
Original file line number Diff line number Diff line change
Expand Up @@ -20,9 +20,17 @@ jobs:
E2B_API_KEY: ${{ secrets.E2B_API_KEY }}
with:
E2B_DOMAIN: ${{ vars.E2B_DOMAIN }}
code-interpreter-tests:
uses: ./.github/workflows/code_interpreter_tests.yml
needs: [build-template]
secrets:
E2B_API_KEY: ${{ secrets.E2B_API_KEY }}
with:
E2B_DOMAIN: ${{ vars.E2B_DOMAIN }}
E2B_TESTS_TEMPLATE: ${{ needs.build-template.outputs.template_id }}
cleanup-build-template:
uses: ./.github/workflows/cleanup_build_template.yml
needs: [build-template]
needs: [build-template, code-interpreter-tests]
if: always() && !contains(needs.build-template.result, 'failure') && !contains(needs.build-template.result, 'cancelled')
secrets:
E2B_API_KEY: ${{ secrets.E2B_API_KEY }}
Expand Down
33 changes: 0 additions & 33 deletions .github/workflows/server_tests.yml

This file was deleted.

2 changes: 2 additions & 0 deletions pnpm-lock.yaml

Some generated files are not rendered by default. Learn more about how customized files appear on GitHub.

1 change: 1 addition & 0 deletions pnpm-workspace.yaml
Original file line number Diff line number Diff line change
@@ -1,6 +1,7 @@
packages:
- chart_data_extractor
- template
- tests

# Only install package versions that have been on the registry for at least
# 3 days, so malicious releases have time to be caught and unpublished.
Expand Down
4 changes: 0 additions & 4 deletions template/pytest.ini

This file was deleted.

5 changes: 0 additions & 5 deletions template/requirements-test.txt

This file was deleted.

96 changes: 0 additions & 96 deletions template/tests/test_jupyter_socket.py

This file was deleted.

7 changes: 7 additions & 0 deletions tests/.env.example
Original file line number Diff line number Diff line change
@@ -0,0 +1,7 @@
E2B_API_KEY=
# Optional, defaults to e2b.app
E2B_DOMAIN=
# Template alias/ID to create sandboxes from, defaults to code-interpreter-v1
E2B_TESTS_TEMPLATE=
# Set to true to run against a local server on http://localhost:49999 (`make start-template-server`)
E2B_DEBUG=
20 changes: 20 additions & 0 deletions tests/README.md
Original file line number Diff line number Diff line change
@@ -0,0 +1,20 @@
# Code Interpreter HTTP tests

End-to-end tests for the Code Interpreter template. Sandboxes are created with
the `e2b` SDK from `E2B_TESTS_TEMPLATE` (defaults to `code-interpreter-v1`);
everything else talks to the server's HTTP API on port `49999` directly with
`httpx`, so the server protocol is what's under test, not the SDK. See
`.env.example` for configuration.

```bash
uv sync --locked
E2B_API_KEY=... uv run pytest
```

Set `E2B_DEBUG=true` to run against a local server started with
`make start-template-server`; tests marked `skip_debug` (sandbox provisioning,
optional kernels, recovery) are skipped in that mode.

A few checks need a shell inside the sandbox (systemd restarts, the private
Jupyter Unix socket in `test_jupyter_socket.py`); those use the SDK's
`sandbox.commands.run` and are otherwise asserted the same way.
Empty file added tests/charts/__init__.py
Empty file.
15 changes: 15 additions & 0 deletions tests/charts/conftest.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,15 @@
import pytest

from harness import CodeInterpreter


@pytest.fixture()
def run_chart(client: CodeInterpreter):
def run(code: str) -> dict:
execution = client.run_code(code)
assert execution.error is None, execution.error
chart = execution.results[0].chart
assert chart
return chart

return run
43 changes: 43 additions & 0 deletions tests/charts/test_bar.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,43 @@
code = """
import matplotlib.pyplot as plt

# Prepare data
authors = ['Author A', 'Author B', 'Author C', 'Author D']
sales = [100, 200, 300, 400]

# Create and customize the bar chart
plt.figure(figsize=(10, 6))
plt.bar(authors, sales, label='Books Sold', color='blue')
plt.xlabel('Authors')
plt.ylabel('Number of Books Sold')
plt.title('Book Sales by Authors')

# Display the chart
plt.tight_layout()
plt.show()
"""


def test_chart_bar(run_chart):
chart = run_chart(code)

assert chart["type"] == "bar"
assert chart["title"] == "Book Sales by Authors"

assert chart["x_label"] == "Authors"
assert chart["y_label"] == "Number of Books Sold"

assert chart["x_unit"] is None
assert chart["y_unit"] is None

bars = chart["elements"]
assert len(bars) == 4

assert [bar["value"] for bar in bars] == [100, 200, 300, 400]
assert [bar["label"] for bar in bars] == [
"Author A",
"Author B",
"Author C",
"Author D",
]
assert [bar["group"] for bar in bars] == ["Books Sold"] * 4
53 changes: 53 additions & 0 deletions tests/charts/test_box_and_whiskers.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,53 @@
code = """
import matplotlib.pyplot as plt
import numpy as np

# Sample data
data = {
'Class A': [85, 90, 78, 92, 88],
'Class B': [95, 89, 76, 91, 84, 87],
'Class C': [75, 82, 88, 79, 86]
}

# Create figure and axis
fig, ax = plt.subplots(figsize=(10, 6))

# Customize plot
ax.set_title('Exam Scores Distribution')
ax.set_xlabel('Class')
ax.set_ylabel('Score')

# Set custom colors
ax.boxplot(data.values(), labels=data.keys(), patch_artist=True)

# Add legend
ax.legend()

# Adjust layout and show plot
plt.tight_layout()
plt.show()
"""


def test_box_and_whiskers(run_chart):
chart = run_chart(code)

assert chart["type"] == "box_and_whisker"
assert chart["title"] == "Exam Scores Distribution"

assert chart["x_label"] == "Class"
assert chart["y_label"] == "Score"

assert chart["x_unit"] is None
assert chart["y_unit"] is None

bars = chart["elements"]
assert len(bars) == 3

assert [bar["outliers"] for bar in bars] == [[], [76], []]
assert [bar["min"] for bar in bars] == [78, 84, 75]
assert [bar["first_quartile"] for bar in bars] == [85, 84.75, 79]
assert [bar["median"] for bar in bars] == [88, 88, 82]
assert [bar["third_quartile"] for bar in bars] == [90, 90.5, 86]
assert [bar["max"] for bar in bars] == [92, 95, 88]
assert [bar["label"] for bar in bars] == ["Class A", "Class B", "Class C"]
Loading
Loading