Overworld: RuntimeError: PassManager::run failed
App: Overworld (overworld.pinokio.git)
Repo: https://github.com/cocktailpeanut/overworld.pinokio.git
Generated: 2026-08-26T17:10:01.093Z
Pinokio: 8.0.40
Platform: win32 x64
Node: v22.21.1
Summary
System
{
"pinokio": {
"version": "8.0.40",
"node": "v22.21.1",
"platform": "win32",
"arch": "x64"
},
"hardware": {
"gpu": "nvidia",
"gpu_model": "nvidia geforce rtx 2080",
"ram_gb": 32,
"vram_gb": 8
},
"os": {
"platform": "Windows",
"distro": "Microsoft Windows 11 Home",
"release": "10.0.26200",
"codename": "25H2",
"kernel": "10.0.26200",
"arch": "x64",
"build": "26200",
"servicepack": "0.0",
"uefi": true
},
"system": {
"manufacturer": "ASUS",
"model": "System Product Name",
"version": "System Version",
"virtual": false
},
"cpu": {
"manufacturer": "Intel",
"brand": "Core™ i9-9900K",
"vendor": "GenuineIntel",
"family": "6",
"model": "158",
"stepping": "12",
"speed": 3.6,
"speedMin": 3.6,
"speedMax": 3.6,
"cores": 16,
"physicalCores": 8,
"processors": 1,
"performanceCores": 16,
"efficiencyCores": 0,
"virtualization": true,
"cache": {
"l1d": 256,
"l1i": 256,
"l2": 2097152,
"l3": 16777216
}
},
"memory": {
"total": 34266476544,
"free": 23887851520,
"used": 10378629120,
"active": 10378633216,
"available": 23887839232,
"buffers": 0,
"cached": 0,
"slab": 0,
"buffcache": 0,
"swaptotal": 3491758080,
"swapused": 257949696,
"swapfree": 3233808384
},
"gpus": [
{
"model": "nvidia geforce rtx 2080"
}
],
"graphics": {
"controllers": [
{
"vendor": "NVIDIA",
"model": "NVIDIA GeForce RTX 2080",
"bus": "PCI",
"vram": 8192,
"vramDynamic": false,
"driverVersion": "610.47"
}
],
"displays": [
{
"model": "VG59QM",
"main": true,
"builtin": false,
"connection": "DP",
"currentResX": 1920,
"currentResY": 1080,
"resolutionX": 1920,
"resolutionY": 1080,
"pixelDepth": 32,
"currentRefreshRate": 279
},
{
"model": "VA719-K",
"main": false,
"builtin": false,
"connection": "HDMI",
"currentResX": 1920,
"currentResY": 1080,
"resolutionX": 1920,
"resolutionY": 1080,
"pixelDepth": 32,
"currentRefreshRate": 279
}
]
}
}
Logs
logs/api/install.js/1787760478216
Source: api / install.js
Lines: 14 total, last 14 included
[api shell.run]
Microsoft Windows [Version 10.0.26200.9168]
(c) Microsoft Corporation. Tüm hakları saklıdır.
C:\pinokio\api\overworld.pinokio.git\app>conda_hook & conda deactivate & conda deactivate & conda deactivate & conda activate base && git clone --filter=blob:none --no-checkout https://github.com/Overworldai/Biome.git biome
Cloning into 'biome'...
remote: Enumerating objects: 4837, done.
remote: Counting objects: 100% (905/905), done.
remote: Compressing objects: 100% (167/167), done.
remote: Total 4837 (delta 763), reused 744 (delta 737), pack-reused 3932 (from 2)
Receiving objects: 100% (4837/4837), 684.81 KiB | 1.95 MiB/s, done.
Resolving deltas: 100% (3171/3171), done.
(base) C:\pinokio\api\overworld.pinokio.git\app>
logs/api/install.js/1787760486729
Source: api / install.js
Lines: 9 total, last 9 included
[api shell.run]
Microsoft Windows [Version 10.0.26200.9168]
(c) Microsoft Corporation. Tüm hakları saklıdır.
C:\pinokio\api\overworld.pinokio.git\app\biome>conda_hook & conda deactivate & conda deactivate & conda deactivate & conda activate base && git fetch origin main
From https://github.com/Overworldai/Biome
* branch main -> FETCH_HEAD
(base) C:\pinokio\api\overworld.pinokio.git\app\biome>
logs/api/install.js/1787760495173
Source: api / install.js
Lines: 7 total, last 7 included
[api shell.run]
Microsoft Windows [Version 10.0.26200.9168]
(c) Microsoft Corporation. Tüm hakları saklıdır.
C:\pinokio\api\overworld.pinokio.git\app\biome>conda_hook & conda deactivate & conda deactivate & conda deactivate & conda activate base && git sparse-checkout init --cone
(base) C:\pinokio\api\overworld.pinokio.git\app\biome>
logs/api/install.js/1787760502477
Source: api / install.js
Lines: 7 total, last 7 included
[api shell.run]
Microsoft Windows [Version 10.0.26200.9168]
(c) Microsoft Corporation. Tüm hakları saklıdır.
C:\pinokio\api\overworld.pinokio.git\app\biome>conda_hook & conda deactivate & conda deactivate & conda deactivate & conda activate base && git sparse-checkout set server-components seeds
(base) C:\pinokio\api\overworld.pinokio.git\app\biome>
logs/api/install.js/1787760509597
Source: api / install.js
Lines: 32 total, last 32 included
[api shell.run]
Microsoft Windows [Version 10.0.26200.9168]
(c) Microsoft Corporation. Tüm hakları saklıdır.
C:\pinokio\api\overworld.pinokio.git\app\biome>conda_hook & conda deactivate & conda deactivate & conda deactivate & conda activate base && git checkout a1ae46a
remote: Enumerating objects: 49, done.
remote: Counting objects: 100% (22/22), done.
remote: Compressing objects: 100% (21/21), done.
remote: Total 49 (delta 3), reused 1 (delta 1), pack-reused 27 (from 1)
Receiving objects: 100% (49/49), 9.05 MiB | 8.15 MiB/s, done.
Resolving deltas: 100% (3/3), done.
Updating files: 100% (51/51), done.
Note: switching to 'a1ae46a'.
You are in 'detached HEAD' state. You can look around, make experimental
changes and commit them, and you can discard any commits you make in this
state without impacting any branches by switching back to a branch.
If you want to create a new branch to retain commits you create, you may
do so (now or later) by using -c with the switch command. Example:
git switch -c <new-branch-name>
Or undo this operation with:
git switch -
Turn off this advice by setting config variable advice.detachedHead to false
HEAD is now at a1ae46a Merge pull request #104 from Overworldai/torch-2.11
(base) C:\pinokio\api\overworld.pinokio.git\app\biome>
logs/api/install.js/1787760519067
Source: api / install.js
Lines: 91 total, last 91 included
[api shell.run]
Microsoft Windows [Version 10.0.26200.9168]
(c) Microsoft Corporation. Tüm hakları saklıdır.
C:\pinokio\api\overworld.pinokio.git\app\biome\server-components>conda_hook & conda deactivate & conda deactivate & conda deactivate & conda activate base & python -m venv C:\pinokio\api\overworld.pinokio.git\app\biome\server-components\.venv & C:\pinokio\api\overworld.pinokio.git\app\biome\server-components\.venv\Scripts\activate C:\pinokio\api\overworld.pinokio.git\app\biome\server-components\.venv & C:\pinokio\api\overworld.pinokio.git\app\biome\server-components\.venv\Scripts\deactivate & C:\pinokio\api\overworld.pinokio.git\app\biome\server-components\.venv\Scripts\activate C:\pinokio\api\overworld.pinokio.git\app\biome\server-components\.venv && uv sync
Using CPython 3.12.13
Removed virtual environment at: .venv
Creating virtual environment at: .venv
Resolved 96 packages in 2m 28s
Built taehv @ https://github.com/madebyollin/taehv/archive/7dc60ec6601af2e668e31bc70acc4cb3665e4c22.zip
Built gemlite @ https://github.com/dropbox/gemlite/archive/ebfcc197d694acd4ca8dfaadf5fceb3570401df3.zip
Built world-engine @ https://github.com/Overworldai/world_engine/archive/refs/tags/1.5.5.zip
Prepared 52 packages in 4m 47s
Installed 75 packages in 8.12s
+ accelerate==1.12.0
+ annotated-doc==0.0.5
+ annotated-types==0.8.0
+ antlr4-python3-runtime==4.9.3
+ anyio==4.14.2
+ bitsandbytes==0.49.2
+ certifi==2026.7.22
+ charset-normalizer==3.5.1
+ click==8.5.0
+ cloudpickle==3.1.2
+ colorama==0.4.6
+ diffusers==0.39.0
+ diskcache==5.6.3
+ einops==0.8.2
+ fastapi==0.135.3
+ filelock==3.32.4
+ fsspec==2026.7.0
+ ftfy==6.3.1
+ gemlite==0.5.1.post1 (from https://github.com/dropbox/gemlite/archive/ebfcc197d694acd4ca8dfaadf5fceb3570401df3.zip)
+ gguf==0.19.0
+ h11==0.16.0
+ hf-xet==1.4.3
+ httpcore==1.0.9
+ httpx==0.28.1
+ huggingface-hub==1.18.0
+ idna==3.19
+ imageio-ffmpeg==0.6.0
+ importlib-metadata==9.0.0
+ jinja2==3.1.6
+ llama-cpp-python==0.3.36+cu128.basic (from https://github.com/JamePeng/llama-cpp-python/releases/download/v0.3.36-cu128-Basic-win-20260417/llama_cpp_python-0.3.36%2Bcu128.basic-cp312-cp312-win_amd64.whl)
+ markdown-it-py==4.2.0
+ markupsafe==3.0.3
+ mdurl==0.1.2
+ mpmath==1.3.0
+ networkx==3.6.1
+ numpy==2.3.2
+ nvidia-ml-py==13.595.45
+ omegaconf==2.3.1
+ orjson==3.12.0
+ packaging==25.0
+ pillow==12.2.0
+ psutil==7.1.3
+ py-cpuinfo==9.0.0
+ pydantic==2.13.4
+ pydantic-core==2.46.4
+ pygments==2.21.0
+ pyvers==0.1.0
+ pyyaml==6.0.3
+ regex==2026.7.19
+ requests==2.34.2
+ rich==15.0.0
+ safetensors==0.8.0
+ setuptools==81.0.0
+ shellingham==1.5.4
+ simplejpeg==1.9.0
+ starlette==1.6.0
+ sympy==1.14.0
+ taehv==0.1.0 (from https://github.com/madebyollin/taehv/archive/7dc60ec6601af2e668e31bc70acc4cb3665e4c22.zip)
+ tensordict==0.10.0
+ timm==1.0.26
+ tokenizers==0.23.1
+ torch==2.11.0+cu128
+ torchvision==0.26.0+cu128
+ tqdm==4.70.0
+ transformers==5.16.1
+ triton-windows==3.6.0.post26
+ typer==0.25.1
+ typing-extensions==4.16.0
+typing-inspection==0.4.4
+ urllib3==2.7.0
+ uvicorn==0.44.0
+ wcwidth==0.8.2
+ websockets==16.0
+ world-engine==1.5.0 (from https://github.com/Overworldai/world_engine/archive/refs/tags/1.5.5.zip)
+ zipp==4.1.0
(.venv) (base) C:\pinokio\api\overworld.pinokio.git\app\biome\server-components>
logs/api/install.js/1787760995199
Source: api / install.js
Lines: 582 total, last 582 included
[api shell.run]
Microsoft Windows [Version 10.0.26200.9168]
(c) Microsoft Corporation. Tüm hakları saklıdır.
C:\pinokio\api\overworld.pinokio.git\app\biome\server-components>conda_hook & conda deactivate & conda deactivate & conda deactivate & conda activate base & C:\pinokio\api\overworld.pinokio.git\app\biome\server-components\.venv\Scripts\activate C:\pinokio\api\overworld.pinokio.git\app\biome\server-components\.venv && uv run python ../../prepare_profile.py --profile low
Uninstalled 1 package in 91ms
Installed 1 package in 70ms
[prepare] Profile: Low VRAM 360p INT8
[prepare] Model: Overworld/Waypoint-1.5-1B-360P quant=intw8a8
[prepare] Seed: C:\pinokio\api\overworld.pinokio.git\app\biome\seeds\default.jpg
[prepare] Downloading/loading model files
C:\pinokio\api\overworld.pinokio.git\app\biome\server-components\.venv\Lib\site-packages\torch\cuda\__init__.py:209: UserWarning: expandable_segments not supported on this platform (Triggered internally at C:\actions-runner\_work\pytorch\pytorch\pytorch\c10/cuda/CUDAAllocatorConfig.h:39.)
torch.tensor([1.0], dtype=torch.bfloat16, device=device)
0%| | 0/1 [00:00<?, ?it/s][prepare] Still working: Downloading/loading model files (30s in this step, 30s total). Torch/Triton compile can take several minutes.
[prepare] Still working: Downloading/loading model files (60s in this step, 60s total). Torch/Triton compile can take several minutes.
[prepare] Still working: Downloading/loading model files (90s in this step, 90s total). Torch/Triton compile can take several minutes.
[prepare] Still working: Downloading/loading model files (120s in this step, 120s total). Torch/Triton compile can take several minutes.
100%|███████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████| 1/1 [02:00<00:00, 120.16s/it]
0%| | 0/1 [00:00<?, ?it/s][prepare] Still working: Downloading/loading model files (150s in this step, 150s total). Torch/Triton compile can take several minutes.
100%|████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████| 1/1 [00:26<00:00, 26.16s/it]
0%| | 0/1 [00:00<?, ?it/s][prepare] Still working: Downloading/loading model files (180s in this step, 180s total). Torch/Triton compile can take several minutes.
[prepare] Still working: Downloading/loading model files (210s in this step, 210s total). Torch/Triton compile can take several minutes.
[prepare] Still working: Downloading/loading model files (240s in this step, 240s total). Torch/Triton compile can take several minutes.
[prepare] Still working: Downloading/loading model files (270s in this step, 270s total). Torch/Triton compile can take several minutes.
100%|███████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████| 1/1 [01:57<00:00, 117.23s/it]
0%| | 0/1 [00:00<?, ?it/s][prepare] Still working: Downloading/loading model files (300s in this step, 300s total). Torch/Triton compile can take several minutes.
100%|████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████| 1/1 [00:26<00:00, 26.19s/it]
0%| | 0/1 [00:00<?, ?it/s][prepare] Still working: Downloading/loading model files (330s in this step, 330s total). Torch/Triton compile can take several minutes.
100%|████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████| 1/1 [00:26<00:00, 26.50s/it]
[prepare] Loading WorldEngine code
[prepare] Loading model
config.yaml: 100%|████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████| 807/807 [00:00<?, ?B/s]
Downloading (incomplete total...): 0.00B [00:00, ?B/s]Warning: You are sending unauthenticated requests to the HF Hub. Please set a HF_TOKEN to enable higher rate limits and faster downloads. | 0/3 [00:00<?, ?it/s]
Fetching 3 files: 100%|██████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████| 3/3 [00:14<00:00, 4.90s/it]
Download complete: 100%|██████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████| 22.8M/22.8M [00:14<00:00, 2.02MB/s][prepare] Still working: Loading model (19s in this step, 360s total). Torch/Triton compile can take several minutes. | 0.00/3.72G [00:01<?, ?B/s]
Download complete: 100%|██████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████| 22.8M/22.8M [00:25<00:00, 2.02MB/s][prepare] Still working: Loading model (49s in this step, 390s total). Torch/Triton compile can take several minutes. | 0.00/3.72G [00:31<?, ?B/s]
[prepare] Still working: Loading model (79s in this step, 420s total). Torch/Triton compile can take several minutes. | 0.00/3.72G [01:01<?, ?B/s]
[prepare] Still working: Loading model (109s in this step, 450s total). Torch/Triton compile can take several minutes. | 0.00/3.72G [01:31<?, ?B/s]
[prepare] Still working: Loading model (139s in this step, 480s total). Torch/Triton compile can take several minutes. | 268M/3.72G [01:56<00:04, 731MB/s]
[prepare] Still working: Loading model (169s in this step, 510s total). Torch/Triton compile can take several minutes. | 268M/3.72G [02:10<00:04, 731MB/s]
[prepare] Still working: Loading model (199s in this step, 540s total). Torch/Triton compile can take several minutes.██████████████████████████████████████████▌ | 2.54G/3.72G [02:57<00:05, 211MB/s]
[prepare] Still working: Loading model (229s in this step, 570s total). Torch/Triton compile can take several minutes.██████████████████████████████████████████████████████████████▍ | 2.98G/3.72G [03:22<00:23, 31.7MB/s]
Fetching 2 files: 100%|█████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████| 2/2 [03:55<00:00, 117.96s/it]
Download complete: 100%|██████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████| 22.8M/22.8M [04:10<00:00, 90.8kB/s]
Download complete: 100%|██████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████| 3.72G/3.72G [03:55<00:00, 15.8MB/s]
C:\pinokio\api\overworld.pinokio.git\app\biome\server-components\.venv\Lib\site-packages\torch\nn\modules\module.py:934: UserWarning: NoiseConditioner: requested dtype cast ignored; keeping torch.float32.
module._apply(fn)
C:\pinokio\api\overworld.pinokio.git\app\biome\server-components\.venv\Lib\site-packages\world_engine\model\nn.py:24: UserWarning: NoiseConditioner: requested dtype cast ignored; keeping torch.float32.
return super()._apply(keep_dtype)
C:\pinokio\api\overworld.pinokio.git\app\biome\server-components\.venv\Lib\site-packages\world_engine\model\nn.py:24: UserWarning: OrthoRoPEAngles: requested dtype cast ignored; keeping torch.float32.
return super()._apply(keep_dtype)
[prepare] Model loaded in 599.00s
[prepare] Loading seed image
[prepare] Instantiating model weights
[prepare] Model load complete
[prepare] Starting warmup compile
[prepare] Warmup 1/4: reset engine state
[prepare] Warmup 2/4: append seed frame and compile seed path
[prepare] Still working: Warmup 2/4: append seed frame and compile seed path (1s in this step, 600s total). Torch/Triton compile can take several minutes.
C:\pinokio\api\overworld.pinokio.git\app\biome\server-components\.venv\Lib\site-packages\torch\_inductor\compile_fx.py:2941: UserWarning: NVIDIA GeForce RTX 2080 does not support bfloat16 compilation natively, skipping
warnings.warn(
W0826 19:28:15.210000 12284 .venv\Lib\site-packages\torch\_inductor\utils.py:1731] [1/0] Not enough SMs to use max_autotune_gemm mode
[prepare] Still working: Warmup 2/4: append seed frame and compile seed path (31s in this step, 630s total). Torch/Triton compile can take several minutes.
E0826 19:28:28.921000 12284 .venv\Lib\site-packages\torch\_inductor\select_algorithm.py:3541] [1/0] Exception No valid triton configs. OutOfMemoryError: out of resource: triton_flex_attention Required: 65536 Hardware limit:65536 Reducing block sizes or `num_stages` may help. for benchmark choice TritonTemplateCaller(C:\pinokio\cache\overworld.pinokio.git\torchinductor\sd\csdgfpw5iv5loan7teimgpydzx725drw22b3z3hcln6rrkuivedo.py, ALLOW_TF32='False', BLOCKS_ARE_CONTIGUOUS=False, BLOCK_M=128, BLOCK_N=64, FLOAT32_PRECISION="'tf32'", GQA_SHARED_HEADS=2, HAS_FULL_BLOCKS=True, IS_DIVISIBLE=True, OUTPUT_LOGSUMEXP=False, OUTPUT_MAX=False, PRESCALE_QK=False, QK_HEAD_DIM=64, QK_HEAD_DIM_ROUNDED=64, ROWS_GUARANTEED_SAFE=False, SAFE_HEAD_DIM=True, SM_SCALE=0.125, SPARSE_KV_BLOCK_SIZE=128, SPARSE_Q_BLOCK_SIZE=128, USE_TMA=False, V_HEAD_DIM=64, V_HEAD_DIM_ROUNDED=64, WRITE_DQ=True, num_stages=3, num_warps=4)
E0826 19:28:28.921000 12284 .venv\Lib\site-packages\torch\_inductor\select_algorithm.py:3541] [1/0] Traceback (most recent call last):
E0826 19:28:28.921000 12284 .venv\Lib\site-packages\torch\_inductor\select_algorithm.py:3541] [1/0] File "C:\Users\mesut\AppData\Roaming\uv\python\cpython-3.12-windows-x86_64-none\Lib\concurrent\futures\thread.py", line 59, in run
E0826 19:28:28.921000 12284 .venv\Lib\site-packages\torch\_inductor\select_algorithm.py:3541] [1/0] result = self.fn(*self.args, **self.kwargs)
E0826 19:28:28.921000 12284 .venv\Lib\site-packages\torch\_inductor\select_algorithm.py:3541] [1/0] ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
E0826 19:28:28.921000 12284 .venv\Lib\site-packages\torch\_inductor\select_algorithm.py:3541] [1/0] File "C:\pinokio\api\overworld.pinokio.git\app\biome\server-components\.venv\Lib\site-packages\torch\_inductor\select_algorithm.py", line 3464, in precompile_with_captured_stdout
E0826 19:28:28.921000 12284 .venv\Lib\site-packages\torch\_inductor\select_algorithm.py:3541] [1/0] choice.precompile()
E0826 19:28:28.921000 12284 .venv\Lib\site-packages\torch\_inductor\select_algorithm.py:3541] [1/0] File "C:\pinokio\api\overworld.pinokio.git\app\biome\server-components\.venv\Lib\site-packages\torch\_inductor\select_algorithm.py", line 2388, in precompile
E0826 19:28:28.921000 12284 .venv\Lib\site-packages\torch\_inductor\select_algorithm.py:3541] [1/0] self.bmreq.precompile()
E0826 19:28:28.921000 12284 .venv\Lib\site-packages\torch\_inductor\select_algorithm.py:3541] [1/0] File "C:\pinokio\api\overworld.pinokio.git\app\biome\server-components\.venv\Lib\site-packages\torch\_inductor\autotune_process.py", line 714, in precompile
E0826 19:28:28.921000 12284 .venv\Lib\site-packages\torch\_inductor\select_algorithm.py:3541] [1/0] getattr(mod, self.kernel_name).precompile()
E0826 19:28:28.921000 12284 .venv\Lib\site-packages\torch\_inductor\select_algorithm.py:3541] [1/0] File "C:\pinokio\api\overworld.pinokio.git\app\biome\server-components\.venv\Lib\site-packages\torch\_inductor\runtime\triton_heuristics.py", line 503, in precompile
E0826 19:28:28.921000 12284 .venv\Lib\site-packages\torch\_inductor\select_algorithm.py:3541] [1/0] self._make_launchers()
E0826 19:28:28.921000 12284 .venv\Lib\site-packages\torch\_inductor\select_algorithm.py:3541] [1/0] File "C:\pinokio\api\overworld.pinokio.git\app\biome\server-components\.venv\Lib\site-packages\torch\_inductor\runtime\triton_heuristics.py", line 664, in _make_launchers
E0826 19:28:28.921000 12284 .venv\Lib\site-packages\torch\_inductor\select_algorithm.py:3541] [1/0] raise RuntimeError(f"No valid triton configs. {type(exc).__name__}: {exc}")
E0826 19:28:28.921000 12284 .venv\Lib\site-packages\torch\_inductor\select_algorithm.py:3541] [1/0] RuntimeError: No valid triton configs. OutOfMemoryError: out of resource: triton_flex_attention Required: 65536 Hardware limit:65536 Reducing block sizes or `num_stages` may help.
E0826 19:28:29.422000 12284 .venv\Lib\site-packages\torch\_inductor\select_algorithm.py:3541] [1/0] Exception No valid triton configs. OutOfMemoryError: out of resource: triton_flex_attention Required: 98304 Hardware limit:65536 Reducing block sizes or `num_stages` may help. for benchmark choice TritonTemplateCaller(C:\pinokio\cache\overworld.pinokio.git\torchinductor\yg\cygr2cw4swt2q6r7pgjrfgdi762firmxje5czyvssdu5rzcpqkua.py, ALLOW_TF32='False', BLOCKS_ARE_CONTIGUOUS=False, BLOCK_M=128, BLOCK_N=128, FLOAT32_PRECISION="'tf32'", GQA_SHARED_HEADS=2, HAS_FULL_BLOCKS=True, IS_DIVISIBLE=True, OUTPUT_LOGSUMEXP=False, OUTPUT_MAX=False, PRESCALE_QK=False, QK_HEAD_DIM=64, QK_HEAD_DIM_ROUNDED=64, ROWS_GUARANTEED_SAFE=False, SAFE_HEAD_DIM=True, SM_SCALE=0.125, SPARSE_KV_BLOCK_SIZE=128, SPARSE_Q_BLOCK_SIZE=128, USE_TMA=False, V_HEAD_DIM=64, V_HEAD_DIM_ROUNDED=64, WRITE_DQ=True, num_stages=2, num_warps=8)
E0826 19:28:29.422000 12284 .venv\Lib\site-packages\torch\_inductor\select_algorithm.py:3541] [1/0] Traceback (most recent call last):
E0826 19:28:29.422000 12284 .venv\Lib\site-packages\torch\_inductor\select_algorithm.py:3541] [1/0] File "C:\Users\mesut\AppData\Roaming\uv\python\cpython-3.12-windows-x86_64-none\Lib\concurrent\futures\thread.py", line 59, in run
E0826 19:28:29.422000 12284 .venv\Lib\site-packages\torch\_inductor\select_algorithm.py:3541] [1/0] result = self.fn(*self.args, **self.kwargs)
E0826 19:28:29.422000 12284 .venv\Lib\site-packages\torch\_inductor\select_algorithm.py:3541] [1/0] ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
E0826 19:28:29.422000 12284 .venv\Lib\site-packages\torch\_inductor\select_algorithm.py:3541] [1/0] File "C:\pinokio\api\overworld.pinokio.git\app\biome\server-components\.venv\Lib\site-packages\torch\_inductor\select_algorithm.py", line 3464, in precompile_with_captured_stdout
E0826 19:28:29.422000 12284 .venv\Lib\site-packages\torch\_inductor\select_algorithm.py:3541] [1/0] choice.precompile()
E0826 19:28:29.422000 12284 .venv\Lib\site-packages\torch\_inductor\select_algorithm.py:3541] [1/0] File "C:\pinokio\api\overworld.pinokio.git\app\biome\server-components\.venv\Lib\site-packages\torch\_inductor\select_algorithm.py", line 2388, in precompile
E0826 19:28:29.422000 12284 .venv\Lib\site-packages\torch\_inductor\select_algorithm.py:3541] [1/0] self.bmreq.precompile()
E0826 19:28:29.422000 12284 .venv\Lib\site-packages\torch\_inductor\select_algorithm.py:3541] [1/0] File "C:\pinokio\api\overworld.pinokio.git\app\biome\server-components\.venv\Lib\site-packages\torch\_inductor\autotune_process.py", line 714, in precompile
E0826 19:28:29.422000 12284 .venv\Lib\site-packages\torch\_inductor\select_algorithm.py:3541] [1/0] getattr(mod, self.kernel_name).precompile()
E0826 19:28:29.422000 12284 .venv\Lib\site-packages\torch\_inductor\select_algorithm.py:3541] [1/0] File "C:\pinokio\api\overworld.pinokio.git\app\biome\server-components\.venv\Lib\site-packages\torch\_inductor\runtime\triton_heuristics.py", line 503, in precompile
E0826 19:28:29.422000 12284 .venv\Lib\site-packages\torch\_inductor\select_algorithm.py:3541] [1/0] self._make_launchers()
E0826 19:28:29.422000 12284 .venv\Lib\site-packages\torch\_inductor\select_algorithm.py:3541] [1/0] File "C:\pinokio\api\overworld.pinokio.git\app\biome\server-components\.venv\Lib\site-packages\torch\_inductor\runtime\triton_heuristics.py", line 664, in _make_launchers
E0826 19:28:29.422000 12284 .venv\Lib\site-packages\torch\_inductor\select_algorithm.py:3541] [1/0] raise RuntimeError(f"No valid triton configs. {type(exc).__name__}: {exc}")
E0826 19:28:29.422000 12284 .venv\Lib\site-packages\torch\_inductor\select_algorithm.py:3541] [1/0] RuntimeError: No valid triton configs. OutOfMemoryError: out of resource: triton_flex_attention Required: 98304 Hardware limit:65536 Reducing block sizes or `num_stages` may help.
E0826 19:28:29.556000 12284 .venv\Lib\site-packages\torch\_inductor\select_algorithm.py:3541] [1/0] Exception No valid triton configs. OutOfMemoryError: out of resource: triton_flex_attention Required: 98304 Hardware limit:65536 Reducing block sizes or `num_stages` may help. for benchmark choice TritonTemplateCaller(C:\pinokio\cache\overworld.pinokio.git\torchinductor\gt\cgtsvyz3vsnyloyddvqdhsp5v7wqndjaiyzm6lmlkrgdpv64ojoj.py, ALLOW_TF32='False', BLOCKS_ARE_CONTIGUOUS=False, BLOCK_M=128, BLOCK_N=128, FLOAT32_PRECISION="'tf32'", GQA_SHARED_HEADS=2, HAS_FULL_BLOCKS=True, IS_DIVISIBLE=True, OUTPUT_LOGSUMEXP=False, OUTPUT_MAX=False, PRESCALE_QK=False, QK_HEAD_DIM=64, QK_HEAD_DIM_ROUNDED=64, ROWS_GUARANTEED_SAFE=False, SAFE_HEAD_DIM=True, SM_SCALE=0.125, SPARSE_KV_BLOCK_SIZE=128, SPARSE_Q_BLOCK_SIZE=128, USE_TMA=False, V_HEAD_DIM=64, V_HEAD_DIM_ROUNDED=64, WRITE_DQ=True, num_stages=1, num_warps=8)
E0826 19:28:29.556000 12284 .venv\Lib\site-packages\torch\_inductor\select_algorithm.py:3541] [1/0] Traceback (most recent call last):
E0826 19:28:29.556000 12284 .venv\Lib\site-packages\torch\_inductor\select_algorithm.py:3541] [1/0] File "C:\Users\mesut\AppData\Roaming\uv\python\cpython-3.12-windows-x86_64-none\Lib\concurrent\futures\thread.py", line 59, in run
E0826 19:28:29.556000 12284 .venv\Lib\site-packages\torch\_inductor\select_algorithm.py:3541] [1/0] result = self.fn(*self.args, **self.kwargs)
E0826 19:28:29.556000 12284 .venv\Lib\site-packages\torch\_inductor\select_algorithm.py:3541] [1/0] ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
E0826 19:28:29.556000 12284 .venv\Lib\site-packages\torch\_inductor\select_algorithm.py:3541] [1/0] File "C:\pinokio\api\overworld.pinokio.git\app\biome\server-components\.venv\Lib\site-packages\torch\_inductor\select_algorithm.py", line 3464, in precompile_with_captured_stdout
E0826 19:28:29.556000 12284 .venv\Lib\site-packages\torch\_inductor\select_algorithm.py:3541] [1/0] choice.precompile()
E0826 19:28:29.556000 12284 .venv\Lib\site-packages\torch\_inductor\select_algorithm.py:3541] [1/0] File "C:\pinokio\api\overworld.pinokio.git\app\biome\server-components\.venv\Lib\site-packages\torch\_inductor\select_algorithm.py", line 2388, in precompile
E0826 19:28:29.556000 12284 .venv\Lib\site-packages\torch\_inductor\select_algorithm.py:3541] [1/0] self.bmreq.precompile()
E0826 19:28:29.556000 12284 .venv\Lib\site-packages\torch\_inductor\select_algorithm.py:3541] [1/0] File "C:\pinokio\api\overworld.pinokio.git\app\biome\server-components\.venv\Lib\site-packages\torch\_inductor\autotune_process.py", line 714, in precompile
E0826 19:28:29.556000 12284 .venv\Lib\site-packages\torch\_inductor\select_algorithm.py:3541] [1/0] getattr(mod, self.kernel_name).precompile()
E0826 19:28:29.556000 12284 .venv\Lib\site-packages\torch\_inductor\select_algorithm.py:3541] [1/0] File "C:\pinokio\api\overworld.pinokio.git\app\biome\server-components\.venv\Lib\site-packages\torch\_inductor\runtime\triton_heuristics.py", line 503, in precompile
E0826 19:28:29.556000 12284 .venv\Lib\site-packages\torch\_inductor\select_algorithm.py:3541] [1/0] self._make_launchers()
E0826 19:28:29.556000 12284 .venv\Lib\site-packages\torch\_inductor\select_algorithm.py:3541] [1/0] File "C:\pinokio\api\overworld.pinokio.git\app\biome\server-components\.venv\Lib\site-packages\torch\_inductor\runtime\triton_heuristics.py", line 664, in _make_launchers
E0826 19:28:29.556000 12284 .venv\Lib\site-packages\torch\_inductor\select_algorithm.py:3541] [1/0] raise RuntimeError(f"No valid triton configs. {type(exc).__name__}: {exc}")
E0826 19:28:29.556000 12284 .venv\Lib\site-packages\torch\_inductor\select_algorithm.py:3541] [1/0] RuntimeError: No valid triton configs. OutOfMemoryError: out of resource: triton_flex_attention Required: 98304 Hardware limit:65536 Reducing block sizes or `num_stages` may help.
E0826 19:28:50.021000 12284 .venv\Lib\site-packages\torch\_inductor\select_algorithm.py:3541] [1/0] Exception No valid triton configs. OutOfMemoryError: out of resource: triton_flex_attention Required: 98304 Hardware limit:65536 Reducing block sizes or `num_stages` may help. for benchmark choice TritonTemplateCaller(C:\pinokio\cache\overworld.pinokio.git\torchinductor\f7\cf7g6hoytsg2zf3b53xs5skt3fok4673jyubfvzvjpmm7xvektjt.py, ALLOW_TF32='False', BLOCKS_ARE_CONTIGUOUS=False, BLOCK_M=128, BLOCK_N=128, FLOAT32_PRECISION="'tf32'", GQA_SHARED_HEADS=2, HAS_FULL_BLOCKS=True, IS_DIVISIBLE=True, OUTPUT_LOGSUMEXP=False, OUTPUT_MAX=False, PRESCALE_QK=False, QK_HEAD_DIM=64, QK_HEAD_DIM_ROUNDED=64, ROWS_GUARANTEED_SAFE=False, SAFE_HEAD_DIM=True, SM_SCALE=0.125, SPARSE_KV_BLOCK_SIZE=128, SPARSE_Q_BLOCK_SIZE=128, USE_TMA=False, V_HEAD_DIM=64, V_HEAD_DIM_ROUNDED=64, WRITE_DQ=True, num_stages=3, num_warps=4)
E0826 19:28:50.021000 12284 .venv\Lib\site-packages\torch\_inductor\select_algorithm.py:3541] [1/0] Traceback (most recent call last):
E0826 19:28:50.021000 12284 .venv\Lib\site-packages\torch\_inductor\select_algorithm.py:3541] [1/0] File "C:\Users\mesut\AppData\Roaming\uv\python\cpython-3.12-windows-x86_64-none\Lib\concurrent\futures\thread.py", line 59, in run
E0826 19:28:50.021000 12284 .venv\Lib\site-packages\torch\_inductor\select_algorithm.py:3541] [1/0] result = self.fn(*self.args, **self.kwargs)
E0826 19:28:50.021000 12284 .venv\Lib\site-packages\torch\_inductor\select_algorithm.py:3541] [1/0] ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
E0826 19:28:50.021000 12284 .venv\Lib\site-packages\torch\_inductor\select_algorithm.py:3541] [1/0] File "C:\pinokio\api\overworld.pinokio.git\app\biome\server-components\.venv\Lib\site-packages\torch\_inductor\select_algorithm.py", line 3464, in precompile_with_captured_stdout
E0826 19:28:50.021000 12284 .venv\Lib\site-packages\torch\_inductor\select_algorithm.py:3541] [1/0] choice.precompile()
E0826 19:28:50.021000 12284 .venv\Lib\site-packages\torch\_inductor\select_algorithm.py:3541] [1/0] File "C:\pinokio\api\overworld.pinokio.git\app\biome\server-components\.venv\Lib\site-packages\torch\_inductor\select_algorithm.py", line 2388, in precompile
E0826 19:28:50.021000 12284 .venv\Lib\site-packages\torch\_inductor\select_algorithm.py:3541] [1/0] self.bmreq.precompile()
E0826 19:28:50.021000 12284 .venv\Lib\site-packages\torch\_inductor\select_algorithm.py:3541] [1/0] File "C:\pinokio\api\overworld.pinokio.git\app\biome\server-components\.venv\Lib\site-packages\torch\_inductor\autotune_process.py", line 714, in precompile
E0826 19:28:50.021000 12284 .venv\Lib\site-packages\torch\_inductor\select_algorithm.py:3541] [1/0] getattr(mod, self.kernel_name).precompile()
E0826 19:28:50.021000 12284 .venv\Lib\site-packages\torch\_inductor\select_algorithm.py:3541] [1/0] File "C:\pinokio\api\overworld.pinokio.git\app\biome\server-components\.venv\Lib\site-packages\torch\_inductor\runtime\triton_heuristics.py", line 503, in precompile
E0826 19:28:50.021000 12284 .venv\Lib\site-packages\torch\_inductor\select_algorithm.py:3541] [1/0] self._make_launchers()
E0826 19:28:50.021000 12284 .venv\Lib\site-packages\torch\_inductor\select_algorithm.py:3541] [1/0] File "C:\pinokio\api\overworld.pinokio.git\app\biome\server-components\.venv\Lib\site-packages\torch\_inductor\runtime\triton_heuristics.py", line 664, in _make_launchers
E0826 19:28:50.021000 12284 .venv\Lib\site-packages\torch\_inductor\select_algorithm.py:3541] [1/0] raise RuntimeError(f"No valid triton configs. {type(exc).__name__}: {exc}")
E0826 19:28:50.021000 12284 .venv\Lib\site-packages\torch\_inductor\select_algorithm.py:3541] [1/0] RuntimeError: No valid triton configs. OutOfMemoryError: out of resource: triton_flex_attention Required: 98304 Hardware limit:65536 Reducing block sizes or `num_stages` may help.
C:\pinokio\api\overworld.pinokio.git\app\biome\server-components\.venv\Lib\site-packages\torch\_inductor\select_algorithm.py:3686: UserWarning: TypedStorage is deprecated. It will be removed in the future and UntypedStorage will be the only storage class. This should only matter to you if you are using storages directly. To access UntypedStorage directly, use tensor.untyped_storage() instead of tensor.storage()
current_out_size = out_base.storage().size()
[prepare] Still working: Warmup 2/4: append seed frame and compile seed path (61s in this step, 660s total). Torch/Triton compile can take several minutes.
Autotune Choices Stats:
{"num_choices": 2, "num_triton_choices": 2, "best_kernel": "triton_flex_attention_5", "best_kernel_desc": "ALLOW_TF32='False', BLOCKS_ARE_CONTIGUOUS=False, BLOCK_M=64, BLOCK_N=64, FLOAT32_PRECISION=\"'tf32'\", GQA_SHARED_HEADS=2, HAS_FULL_BLOCKS=True, IS_DIVISIBLE=True, OUTPUT_LOGSUMEXP=False, OUTPUT_MAX=False, PRESCALE_QK=False, QK_HEAD_DIM=64, QK_HEAD_DIM_ROUNDED=64, ROWS_GUARANTEED_SAFE=False, SAFE_HEAD_DIM=True, SM_SCALE=0.125, SPARSE_KV_BLOCK_SIZE=128, SPARSE_Q_BLOCK_SIZE=128, USE_TMA=False, V_HEAD_DIM=64, V_HEAD_DIM_ROUNDED=64, WRITE_DQ=True, num_stages=3, num_warps=4", "best_time": 5.761023998260498, "best_triton_pos": 0}
AUTOTUNE flex_attention(1x32x128x64, 1x16x2176x64, 1x16x2176x64, 1x32x128, 1x32x128, 1x1x1, 1x1x1x17, 1x1x1, 1x1x1x17)
strides: [262144, 8192, 64, 1], [2228224, 139264, 64, 1], [2228224, 139264, 64, 1], [4096, 128, 1], [4096, 128, 1], [1, 1, 1], [17, 17, 17, 1], [1, 1, 1], [0, 0, 0, 1]
dtypes: torch.bfloat16, torch.bfloat16, torch.bfloat16, torch.float32, torch.float32, torch.int32, torch.int32, torch.int32, torch.int32
triton_flex_attention_5 5.7610 ms 100.0% ALLOW_TF32='False', BLOCKS_ARE_CONTIGUOUS=False, BLOCK_M=64, BLOCK_N=64, FLOAT32_PRECISION="'tf32'", GQA_SHARED_HEADS=2, HAS_FULL_BLOCKS=True, IS_DIVISIBLE=True, OUTPUT_LOGSUMEXP=False, OUTPUT_MAX=False, PRESCALE_QK=False, QK_HEAD_DIM=64, QK_HEAD_DIM_ROUNDED=64, ROWS_GUARANTEED_SAFE=False, SAFE_HEAD_DIM=True, SM_SCALE=0.125, SPARSE_KV_BLOCK_SIZE=128, SPARSE_Q_BLOCK_SIZE=128, USE_TMA=False, V_HEAD_DIM=64, V_HEAD_DIM_ROUNDED=64, WRITE_DQ=True, num_stages=3, num_warps=4
triton_flex_attention_4 19.9439 ms 28.9% ALLOW_TF32='False', BLOCKS_ARE_CONTIGUOUS=False, BLOCK_M=64, BLOCK_N=128, FLOAT32_PRECISION="'tf32'", GQA_SHARED_HEADS=2, HAS_FULL_BLOCKS=True, IS_DIVISIBLE=True, OUTPUT_LOGSUMEXP=False, OUTPUT_MAX=False, PRESCALE_QK=False, QK_HEAD_DIM=64, QK_HEAD_DIM_ROUNDED=64, ROWS_GUARANTEED_SAFE=False, SAFE_HEAD_DIM=True, SM_SCALE=0.125, SPARSE_KV_BLOCK_SIZE=128, SPARSE_Q_BLOCK_SIZE=128, USE_TMA=False, V_HEAD_DIM=64, V_HEAD_DIM_ROUNDED=64, WRITE_DQ=True, num_stages=3, num_warps=4
SingleProcess AUTOTUNE benchmarking takes 0.4679 seconds and 34.2951 seconds precompiling for 2 choices
E0826 19:28:50.809000 12284 .venv\Lib\site-packages\torch\_inductor\select_algorithm.py:3924] [1/0] Runtime error during autotuning:
E0826 19:28:50.809000 12284 .venv\Lib\site-packages\torch\_inductor\select_algorithm.py:3924] [1/0] No valid triton configs. OutOfMemoryError: out of resource: triton_flex_attention Required: 65536 Hardware limit:65536 Reducing block sizes or `num_stages` may help..
E0826 19:28:50.809000 12284 .venv\Lib\site-packages\torch\_inductor\select_algorithm.py:3924] [1/0] Ignoring this choice.
E0826 19:28:50.815000 12284 .venv\Lib\site-packages\torch\_inductor\select_algorithm.py:3924] [1/0] Runtime error during autotuning:
E0826 19:28:50.815000 12284 .venv\Lib\site-packages\torch\_inductor\select_algorithm.py:3924] [1/0] No valid triton configs. OutOfMemoryError: out of resource: triton_flex_attention Required: 98304 Hardware limit:65536 Reducing block sizes or `num_stages` may help..
E0826 19:28:50.815000 12284 .venv\Lib\site-packages\torch\_inductor\select_algorithm.py:3924] [1/0] Ignoring this choice.
E0826 19:28:50.819000 12284 .venv\Lib\site-packages\torch\_inductor\select_algorithm.py:3924] [1/0] Runtime error during autotuning:
E0826 19:28:50.819000 12284 .venv\Lib\site-packages\torch\_inductor\select_algorithm.py:3924] [1/0] No valid triton configs. OutOfMemoryError: out of resource: triton_flex_attention Required: 98304 Hardware limit:65536 Reducing block sizes or `num_stages` may help..
E0826 19:28:50.819000 12284 .venv\Lib\site-packages\torch\_inductor\select_algorithm.py:3924] [1/0] Ignoring this choice.
E0826 19:28:50.822000 12284 .venv\Lib\site-packages\torch\_inductor\select_algorithm.py:3924] [1/0] Runtime error during autotuning:
E0826 19:28:50.822000 12284 .venv\Lib\site-packages\torch\_inductor\select_algorithm.py:3924] [1/0] No valid triton configs. OutOfMemoryError: out of resource: triton_flex_attention Required: 98304 Hardware limit:65536 Reducing block sizes or `num_stages` may help..
E0826 19:28:50.822000 12284 .venv\Lib\site-packages\torch\_inductor\select_algorithm.py:3924] [1/0] Ignoring this choice.
Autotune Choices Stats:
{"num_choices": 6, "num_triton_choices": 6, "best_kernel": "triton_flex_attention_11", "best_kernel_desc": "ALLOW_TF32='False', BLOCKS_ARE_CONTIGUOUS=False, BLOCK_M=64, BLOCK_N=64, FLOAT32_PRECISION=\"'tf32'\", GQA_SHARED_HEADS=2, HAS_FULL_BLOCKS=True, IS_DIVISIBLE=True, OUTPUT_LOGSUMEXP=False, OUTPUT_MAX=False, PRESCALE_QK=False, QK_HEAD_DIM=64, QK_HEAD_DIM_ROUNDED=64, ROWS_GUARANTEED_SAFE=False, SAFE_HEAD_DIM=True, SM_SCALE=0.125, SPARSE_KV_BLOCK_SIZE=128, SPARSE_Q_BLOCK_SIZE=128, USE_TMA=False, V_HEAD_DIM=64, V_HEAD_DIM_ROUNDED=64, WRITE_DQ=True, num_stages=3, num_warps=4", "best_time": 5.312064170837402, "best_triton_pos": 0}
AUTOTUNE flex_attention(1x32x128x64, 1x16x2176x64, 1x16x2176x64, 1x32x128, 1x32x128, 1x1x1, 1x1x1x17, 1x1x1, 1x1x1x17)
strides: [262144, 8192, 64, 1], [2228224, 139264, 64, 1], [2228224, 139264, 64, 1], [4096, 128, 1], [4096, 128, 1], [1, 1, 1], [17, 17, 17, 1], [1, 1, 1], [0, 0, 0, 1]
dtypes: torch.bfloat16, torch.bfloat16, torch.bfloat16, torch.float32, torch.float32, torch.int32, torch.int32, torch.int32, torch.int32
triton_flex_attention_11 5.3121 ms 100.0% ALLOW_TF32='False', BLOCKS_ARE_CONTIGUOUS=False, BLOCK_M=64, BLOCK_N=64, FLOAT32_PRECISION="'tf32'", GQA_SHARED_HEADS=2, HAS_FULL_BLOCKS=True, IS_DIVISIBLE=True, OUTPUT_LOGSUMEXP=False, OUTPUT_MAX=False, PRESCALE_QK=False, QK_HEAD_DIM=64, QK_HEAD_DIM_ROUNDED=64, ROWS_GUARANTEED_SAFE=False, SAFE_HEAD_DIM=True, SM_SCALE=0.125, SPARSE_KV_BLOCK_SIZE=128, SPARSE_Q_BLOCK_SIZE=128, USE_TMA=False, V_HEAD_DIM=64, V_HEAD_DIM_ROUNDED=64, WRITE_DQ=True, num_stages=3, num_warps=4
triton_flex_attention_10 18.6336 ms 28.5% ALLOW_TF32='False', BLOCKS_ARE_CONTIGUOUS=False, BLOCK_M=64, BLOCK_N=128, FLOAT32_PRECISION="'tf32'", GQA_SHARED_HEADS=2, HAS_FULL_BLOCKS=True, IS_DIVISIBLE=True, OUTPUT_LOGSUMEXP=False, OUTPUT_MAX=False, PRESCALE_QK=False, QK_HEAD_DIM=64, QK_HEAD_DIM_ROUNDED=64, ROWS_GUARANTEED_SAFE=False, SAFE_HEAD_DIM=True, SM_SCALE=0.125, SPARSE_KV_BLOCK_SIZE=128, SPARSE_Q_BLOCK_SIZE=128, USE_TMA=False, V_HEAD_DIM=64, V_HEAD_DIM_ROUNDED=64, WRITE_DQ=True, num_stages=3, num_warps=4
triton_flex_attention_6 inf ms 0.0% ALLOW_TF32='False', BLOCKS_ARE_CONTIGUOUS=False, BLOCK_M=128, BLOCK_N=64, FLOAT32_PRECISION="'tf32'", GQA_SHARED_HEADS=2, HAS_FULL_BLOCKS=True, IS_DIVISIBLE=True, OUTPUT_LOGSUMEXP=False, OUTPUT_MAX=False, PRESCALE_QK=False, QK_HEAD_DIM=64, QK_HEAD_DIM_ROUNDED=64, ROWS_GUARANTEED_SAFE=False, SAFE_HEAD_DIM=True, SM_SCALE=0.125, SPARSE_KV_BLOCK_SIZE=128, SPARSE_Q_BLOCK_SIZE=128, USE_TMA=False, V_HEAD_DIM=64, V_HEAD_DIM_ROUNDED=64, WRITE_DQ=True, num_stages=3, num_warps=4
triton_flex_attention_7 inf ms 0.0% ALLOW_TF32='False', BLOCKS_ARE_CONTIGUOUS=False, BLOCK_M=128, BLOCK_N=128, FLOAT32_PRECISION="'tf32'", GQA_SHARED_HEADS=2, HAS_FULL_BLOCKS=True, IS_DIVISIBLE=True, OUTPUT_LOGSUMEXP=False, OUTPUT_MAX=False, PRESCALE_QK=False, QK_HEAD_DIM=64, QK_HEAD_DIM_ROUNDED=64, ROWS_GUARANTEED_SAFE=False, SAFE_HEAD_DIM=True, SM_SCALE=0.125, SPARSE_KV_BLOCK_SIZE=128, SPARSE_Q_BLOCK_SIZE=128, USE_TMA=False, V_HEAD_DIM=64, V_HEAD_DIM_ROUNDED=64, WRITE_DQ=True, num_stages=3, num_warps=4
triton_flex_attention_8 inf ms 0.0% ALLOW_TF32='False', BLOCKS_ARE_CONTIGUOUS=False, BLOCK_M=128, BLOCK_N=128, FLOAT32_PRECISION="'tf32'", GQA_SHARED_HEADS=2, HAS_FULL_BLOCKS=True, IS_DIVISIBLE=True, OUTPUT_LOGSUMEXP=False, OUTPUT_MAX=False, PRESCALE_QK=False, QK_HEAD_DIM=64, QK_HEAD_DIM_ROUNDED=64, ROWS_GUARANTEED_SAFE=False, SAFE_HEAD_DIM=True, SM_SCALE=0.125, SPARSE_KV_BLOCK_SIZE=128, SPARSE_Q_BLOCK_SIZE=128, USE_TMA=False, V_HEAD_DIM=64, V_HEAD_DIM_ROUNDED=64, WRITE_DQ=True, num_stages=2, num_warps=8
triton_flex_attention_9 inf ms 0.0% ALLOW_TF32='False', BLOCKS_ARE_CONTIGUOUS=False, BLOCK_M=128, BLOCK_N=128, FLOAT32_PRECISION="'tf32'", GQA_SHARED_HEADS=2, HAS_FULL_BLOCKS=True, IS_DIVISIBLE=True, OUTPUT_LOGSUMEXP=False, OUTPUT_MAX=False, PRESCALE_QK=False, QK_HEAD_DIM=64, QK_HEAD_DIM_ROUNDED=64, ROWS_GUARANTEED_SAFE=False, SAFE_HEAD_DIM=True, SM_SCALE=0.125, SPARSE_KV_BLOCK_SIZE=128, SPARSE_Q_BLOCK_SIZE=128, USE_TMA=False, V_HEAD_DIM=64, V_HEAD_DIM_ROUNDED=64, WRITE_DQ=True, num_stages=1, num_warps=8
SingleProcess AUTOTUNE benchmarking takes 0.2248 seconds and 0.0000 seconds precompiling for 6 choices
E0826 19:29:04.823000 12284 .venv\Lib\site-packages\torch\_inductor\select_algorithm.py:3541] [1/0] Exception No valid triton configs. OutOfMemoryError: out of resource: triton_flex_attention Required: 98304 Hardware limit:65536 Reducing block sizes or `num_stages` may help. for benchmark choice TritonTemplateCaller(C:\pinokio\cache\overworld.pinokio.git\torchinductor\r3\cr37xoaf5zqdtsvvymkp4eka6z3qh7447n5l66xgwqy6w2qfnrc7.py, ALLOW_TF32='False', BLOCKS_ARE_CONTIGUOUS=False, BLOCK_M=128, BLOCK_N=128, FLOAT32_PRECISION="'tf32'", GQA_SHARED_HEADS=2, HAS_FULL_BLOCKS=True, IS_DIVISIBLE=True, OUTPUT_LOGSUMEXP=False, OUTPUT_MAX=False, PRESCALE_QK=False, QK_HEAD_DIM=64, QK_HEAD_DIM_ROUNDED=64, ROWS_GUARANTEED_SAFE=False, SAFE_HEAD_DIM=True, SM_SCALE=0.125, SPARSE_KV_BLOCK_SIZE=128, SPARSE_Q_BLOCK_SIZE=128, USE_TMA=False, V_HEAD_DIM=64, V_HEAD_DIM_ROUNDED=64, WRITE_DQ=True, num_stages=2, num_warps=8)
E0826 19:29:04.823000 12284 .venv\Lib\site-packages\torch\_inductor\select_algorithm.py:3541] [1/0] Traceback (most recent call last):
E0826 19:29:04.823000 12284 .venv\Lib\site-packages\torch\_inductor\select_algorithm.py:3541] [1/0] File "C:\Users\mesut\AppData\Roaming\uv\python\cpython-3.12-windows-x86_64-none\Lib\concurrent\futures\thread.py", line 59, in run
E0826 19:29:04.823000 12284 .venv\Lib\site-packages\torch\_inductor\select_algorithm.py:3541] [1/0] result = self.fn(*self.args, **self.kwargs)
E0826 19:29:04.823000 12284 .venv\Lib\site-packages\torch\_inductor\select_algorithm.py:3541] [1/0] ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
E0826 19:29:04.823000 12284 .venv\Lib\site-packages\torch\_inductor\select_algorithm.py:3541] [1/0] File "C:\pinokio\api\overworld.pinokio.git\app\biome\server-components\.venv\Lib\site-packages\torch\_inductor\select_algorithm.py", line 3464, in precompile_with_captured_stdout
E0826 19:29:04.823000 12284 .venv\Lib\site-packages\torch\_inductor\select_algorithm.py:3541] [1/0] choice.precompile()
E0826 19:29:04.823000 12284 .venv\Lib\site-packages\torch\_inductor\select_algorithm.py:3541] [1/0] File "C:\pinokio\api\overworld.pinokio.git\app\biome\server-components\.venv\Lib\site-packages\torch\_inductor\select_algorithm.py", line 2388, in precompile
E0826 19:29:04.823000 12284 .venv\Lib\site-packages\torch\_inductor\select_algorithm.py:3541] [1/0] self.bmreq.precompile()
E0826 19:29:04.823000 12284 .venv\Lib\site-packages\torch\_inductor\select_algorithm.py:3541] [1/0] File "C:\pinokio\api\overworld.pinokio.git\app\biome\server-components\.venv\Lib\site-packages\torch\_inductor\autotune_process.py", line 714, in precompile
E0826 19:29:04.823000 12284 .venv\Lib\site-packages\torch\_inductor\select_algorithm.py:3541] [1/0] getattr(mod, self.kernel_name).precompile()
E0826 19:29:04.823000 12284 .venv\Lib\site-packages\torch\_inductor\select_algorithm.py:3541] [1/0] File "C:\pinokio\api\overworld.pinokio.git\app\biome\server-components\.venv\Lib\site-packages\torch\_inductor\runtime\triton_heuristics.py", line 503, in precompile
E0826 19:29:04.823000 12284 .venv\Lib\site-packages\torch\_inductor\select_algorithm.py:3541] [1/0] self._make_launchers()
E0826 19:29:04.823000 12284 .venv\Lib\site-packages\torch\_inductor\select_algorithm.py:3541] [1/0] File "C:\pinokio\api\overworld.pinokio.git\app\biome\server-components\.venv\Lib\site-packages\torch\_inductor\runtime\triton_heuristics.py", line 664, in _make_launchers
E0826 19:29:04.823000 12284 .venv\Lib\site-packages\torch\_inductor\select_algorithm.py:3541] [1/0] raise RuntimeError(f"No valid triton configs. {type(exc).__name__}: {exc}")
E0826 19:29:04.823000 12284 .venv\Lib\site-packages\torch\_inductor\select_algorithm.py:3541] [1/0] RuntimeError: No valid triton configs. OutOfMemoryError: out of resource: triton_flex_attention Required: 98304 Hardware limit:65536 Reducing block sizes or `num_stages` may help.
E0826 19:29:05.013000 12284 .venv\Lib\site-packages\torch\_inductor\select_algorithm.py:3541] [1/0] Exception No valid triton configs. OutOfMemoryError: out of resource: triton_flex_attention Required: 98304 Hardware limit:65536 Reducing block sizes or `num_stages` may help. for benchmark choice TritonTemplateCaller(C:\pinokio\cache\overworld.pinokio.git\torchinductor\7m\c7mbqcud3lzg6spkuyzlpprt5qj4jliz2zjccogfowjvlpccyojp.py, ALLOW_TF32='False', BLOCKS_ARE_CONTIGUOUS=False, BLOCK_M=128, BLOCK_N=128, FLOAT32_PRECISION="'tf32'", GQA_SHARED_HEADS=2, HAS_FULL_BLOCKS=True, IS_DIVISIBLE=True, OUTPUT_LOGSUMEXP=False, OUTPUT_MAX=False, PRESCALE_QK=False, QK_HEAD_DIM=64, QK_HEAD_DIM_ROUNDED=64, ROWS_GUARANTEED_SAFE=False, SAFE_HEAD_DIM=True, SM_SCALE=0.125, SPARSE_KV_BLOCK_SIZE=128, SPARSE_Q_BLOCK_SIZE=128, USE_TMA=False, V_HEAD_DIM=64, V_HEAD_DIM_ROUNDED=64, WRITE_DQ=True, num_stages=1, num_warps=8)
E0826 19:29:05.013000 12284 .venv\Lib\site-packages\torch\_inductor\select_algorithm.py:3541] [1/0] Traceback (most recent call last):
E0826 19:29:05.013000 12284 .venv\Lib\site-packages\torch\_inductor\select_algorithm.py:3541] [1/0] File "C:\Users\mesut\AppData\Roaming\uv\python\cpython-3.12-windows-x86_64-none\Lib\concurrent\futures\thread.py", line 59, in run
E0826 19:29:05.013000 12284 .venv\Lib\site-packages\torch\_inductor\select_algorithm.py:3541] [1/0] result = self.fn(*self.args, **self.kwargs)
E0826 19:29:05.013000 12284 .venv\Lib\site-packages\torch\_inductor\select_algorithm.py:3541] [1/0] ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
E0826 19:29:05.013000 12284 .venv\Lib\site-packages\torch\_inductor\select_algorithm.py:3541] [1/0] File "C:\pinokio\api\overworld.pinokio.git\app\biome\server-components\.venv\Lib\site-packages\torch\_inductor\select_algorithm.py", line 3464, in precompile_with_captured_stdout
E0826 19:29:05.013000 12284 .venv\Lib\site-packages\torch\_inductor\select_algorithm.py:3541] [1/0] choice.precompile()
E0826 19:29:05.013000 12284 .venv\Lib\site-packages\torch\_inductor\select_algorithm.py:3541] [1/0] File "C:\pinokio\api\overworld.pinokio.git\app\biome\server-components\.venv\Lib\site-packages\torch\_inductor\select_algorithm.py", line 2388, in precompile
E0826 19:29:05.013000 12284 .venv\Lib\site-packages\torch\_inductor\select_algorithm.py:3541] [1/0] self.bmreq.precompile()
E0826 19:29:05.013000 12284 .venv\Lib\site-packages\torch\_inductor\select_algorithm.py:3541] [1/0] File "C:\pinokio\api\overworld.pinokio.git\app\biome\server-components\.venv\Lib\site-packages\torch\_inductor\autotune_process.py", line 714, in precompile
E0826 19:29:05.013000 12284 .venv\Lib\site-packages\torch\_inductor\select_algorithm.py:3541] [1/0] getattr(mod, self.kernel_name).precompile()
E0826 19:29:05.013000 12284 .venv\Lib\site-packages\torch\_inductor\select_algorithm.py:3541] [1/0] File "C:\pinokio\api\overworld.pinokio.git\app\biome\server-components\.venv\Lib\site-packages\torch\_inductor\runtime\triton_heuristics.py", line 503, in precompile
E0826 19:29:05.013000 12284 .venv\Lib\site-packages\torch\_inductor\select_algorithm.py:3541] [1/0] self._make_launchers()
E0826 19:29:05.013000 12284 .venv\Lib\site-packages\torch\_inductor\select_algorithm.py:3541] [1/0] File "C:\pinokio\api\overworld.pinokio.git\app\biome\server-components\.venv\Lib\site-packages\torch\_inductor\runtime\triton_heuristics.py", line 664, in _make_launchers
E0826 19:29:05.013000 12284 .venv\Lib\site-packages\torch\_inductor\select_algorithm.py:3541] [1/0] raise RuntimeError(f"No valid triton configs. {type(exc).__name__}: {exc}")
E0826 19:29:05.013000 12284 .venv\Lib\site-packages\torch\_inductor\select_algorithm.py:3541] [1/0] RuntimeError: No valid triton configs. OutOfMemoryError: out of resource: triton_flex_attention Required: 98304 Hardware limit:65536 Reducing block sizes or `num_stages` may help.
E0826 19:29:05.141000 12284 .venv\Lib\site-packages\torch\_inductor\select_algorithm.py:3541] [1/0] Exception No valid triton configs. OutOfMemoryError: out of resource: triton_flex_attention Required: 65536 Hardware limit:65536 Reducing block sizes or `num_stages` may help. for benchmark choice TritonTemplateCaller(C:\pinokio\cache\overworld.pinokio.git\torchinductor\u2\cu2zasq3zej33atgohmyzzqalyjru446reuqmjdm6r3zkcs7oexo.py, ALLOW_TF32='False', BLOCKS_ARE_CONTIGUOUS=False, BLOCK_M=128, BLOCK_N=64, FLOAT32_PRECISION="'tf32'", GQA_SHARED_HEADS=2, HAS_FULL_BLOCKS=True, IS_DIVISIBLE=True, OUTPUT_LOGSUMEXP=False, OUTPUT_MAX=False, PRESCALE_QK=False, QK_HEAD_DIM=64, QK_HEAD_DIM_ROUNDED=64, ROWS_GUARANTEED_SAFE=False, SAFE_HEAD_DIM=True, SM_SCALE=0.125, SPARSE_KV_BLOCK_SIZE=128, SPARSE_Q_BLOCK_SIZE=128, USE_TMA=False, V_HEAD_DIM=64, V_HEAD_DIM_ROUNDED=64, WRITE_DQ=True, num_stages=3, num_warps=4)
E0826 19:29:05.141000 12284 .venv\Lib\site-packages\torch\_inductor\select_algorithm.py:3541] [1/0] Traceback (most recent call last):
E0826 19:29:05.141000 12284 .venv\Lib\site-packages\torch\_inductor\select_algorithm.py:3541] [1/0] File "C:\Users\mesut\AppData\Roaming\uv\python\cpython-3.12-windows-x86_64-none\Lib\concurrent\futures\thread.py", line 59, in run
E0826 19:29:05.141000 12284 .venv\Lib\site-packages\torch\_inductor\select_algorithm.py:3541] [1/0] result = self.fn(*self.args, **self.kwargs)
E0826 19:29:05.141000 12284 .venv\Lib\site-packages\torch\_inductor\select_algorithm.py:3541] [1/0] ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
E0826 19:29:05.141000 12284 .venv\Lib\site-packages\torch\_inductor\select_algorithm.py:3541] [1/0] File "C:\pinokio\api\overworld.pinokio.git\app\biome\server-components\.venv\Lib\site-packages\torch\_inductor\select_algorithm.py", line 3464, in precompile_with_captured_stdout
E0826 19:29:05.141000 12284 .venv\Lib\site-packages\torch\_inductor\select_algorithm.py:3541] [1/0] choice.precompile()
E0826 19:29:05.141000 12284 .venv\Lib\site-packages\torch\_inductor\select_algorithm.py:3541] [1/0] File "C:\pinokio\api\overworld.pinokio.git\app\biome\server-components\.venv\Lib\site-packages\torch\_inductor\select_algorithm.py", line 2388, in precompile
E0826 19:29:05.141000 12284 .venv\Lib\site-packages\torch\_inductor\select_algorithm.py:3541] [1/0] self.bmreq.precompile()
E0826 19:29:05.141000 12284 .venv\Lib\site-packages\torch\_inductor\select_algorithm.py:3541] [1/0] File "C:\pinokio\api\overworld.pinokio.git\app\biome\server-components\.venv\Lib\site-packages\torch\_inductor\autotune_process.py", line 714, in precompile
E0826 19:29:05.141000 12284 .venv\Lib\site-packages\torch\_inductor\select_algorithm.py:3541] [1/0] getattr(mod, self.kernel_name).precompile()
E0826 19:29:05.141000 12284 .venv\Lib\site-packages\torch\_inductor\select_algorithm.py:3541] [1/0] File "C:\pinokio\api\overworld.pinokio.git\app\biome\server-components\.venv\Lib\site-packages\torch\_inductor\runtime\triton_heuristics.py", line 503, in precompile
E0826 19:29:05.141000 12284 .venv\Lib\site-packages\torch\_inductor\select_algorithm.py:3541] [1/0] self._make_launchers()
E0826 19:29:05.141000 12284 .venv\Lib\site-packages\torch\_inductor\select_algorithm.py:3541] [1/0] File "C:\pinokio\api\overworld.pinokio.git\app\biome\server-components\.venv\Lib\site-packages\torch\_inductor\runtime\triton_heuristics.py", line 664, in _make_launchers
E0826 19:29:05.141000 12284 .venv\Lib\site-packages\torch\_inductor\select_algorithm.py:3541] [1/0] raise RuntimeError(f"No valid triton configs. {type(exc).__name__}: {exc}")
E0826 19:29:05.141000 12284 .venv\Lib\site-packages\torch\_inductor\select_algorithm.py:3541] [1/0] RuntimeError: No valid triton configs. OutOfMemoryError: out of resource: triton_flex_attention Required: 65536 Hardware limit:65536 Reducing block sizes or `num_stages` may help.
[prepare] Still working: Warmup 2/4: append seed frame and compile seed path (91s in this step, 690s total). Torch/Triton compile can take several minutes.
E0826 19:29:24.353000 12284 .venv\Lib\site-packages\torch\_inductor\select_algorithm.py:3541] [1/0] Exception No valid triton configs. OutOfMemoryError: out of resource: triton_flex_attention Required: 98304 Hardware limit:65536 Reducing block sizes or `num_stages` may help. for benchmark choice TritonTemplateCaller(C:\pinokio\cache\overworld.pinokio.git\torchinductor\sz\cszcumk7oufd7jva3p434n6z2abwhji34zqdjxorgrzwwei7gd3p.py, ALLOW_TF32='False', BLOCKS_ARE_CONTIGUOUS=False, BLOCK_M=128, BLOCK_N=128, FLOAT32_PRECISION="'tf32'", GQA_SHARED_HEADS=2, HAS_FULL_BLOCKS=True, IS_DIVISIBLE=True, OUTPUT_LOGSUMEXP=False, OUTPUT_MAX=False, PRESCALE_QK=False, QK_HEAD_DIM=64, QK_HEAD_DIM_ROUNDED=64, ROWS_GUARANTEED_SAFE=False, SAFE_HEAD_DIM=True, SM_SCALE=0.125, SPARSE_KV_BLOCK_SIZE=128, SPARSE_Q_BLOCK_SIZE=128, USE_TMA=False, V_HEAD_DIM=64, V_HEAD_DIM_ROUNDED=64, WRITE_DQ=True, num_stages=3, num_warps=4)
E0826 19:29:24.353000 12284 .venv\Lib\site-packages\torch\_inductor\select_algorithm.py:3541] [1/0] Traceback (most recent call last):
E0826 19:29:24.353000 12284 .venv\Lib\site-packages\torch\_inductor\select_algorithm.py:3541] [1/0] File "C:\Users\mesut\AppData\Roaming\uv\python\cpython-3.12-windows-x86_64-none\Lib\concurrent\futures\thread.py", line 59, in run
E0826 19:29:24.353000 12284 .venv\Lib\site-packages\torch\_inductor\select_algorithm.py:3541] [1/0] result = self.fn(*self.args, **self.kwargs)
E0826 19:29:24.353000 12284 .venv\Lib\site-packages\torch\_inductor\select_algorithm.py:3541] [1/0] ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
E0826 19:29:24.353000 12284 .venv\Lib\site-packages\torch\_inductor\select_algorithm.py:3541] [1/0] File "C:\pinokio\api\overworld.pinokio.git\app\biome\server-components\.venv\Lib\site-packages\torch\_inductor\select_algorithm.py", line 3464, in precompile_with_captured_stdout
E0826 19:29:24.353000 12284 .venv\Lib\site-packages\torch\_inductor\select_algorithm.py:3541] [1/0] choice.precompile()
E0826 19:29:24.353000 12284 .venv\Lib\site-packages\torch\_inductor\select_algorithm.py:3541] [1/0] File "C:\pinokio\api\overworld.pinokio.git\app\biome\server-components\.venv\Lib\site-packages\torch\_inductor\select_algorithm.py", line 2388, in precompile
E0826 19:29:24.353000 12284 .venv\Lib\site-packages\torch\_inductor\select_algorithm.py:3541] [1/0] self.bmreq.precompile()
E0826 19:29:24.353000 12284 .venv\Lib\site-packages\torch\_inductor\select_algorithm.py:3541] [1/0] File "C:\pinokio\api\overworld.pinokio.git\app\biome\server-components\.venv\Lib\site-packages\torch\_inductor\autotune_process.py", line 714, in precompile
E0826 19:29:24.353000 12284 .venv\Lib\site-packages\torch\_inductor\select_algorithm.py:3541] [1/0] getattr(mod, self.kernel_name).precompile()
E0826 19:29:24.353000 12284 .venv\Lib\site-packages\torch\_inductor\select_algorithm.py:3541] [1/0] File "C:\pinokio\api\overworld.pinokio.git\app\biome\server-components\.venv\Lib\site-packages\torch\_inductor\runtime\triton_heuristics.py", line 503, in precompile
E0826 19:29:24.353000 12284 .venv\Lib\site-packages\torch\_inductor\select_algorithm.py:3541] [1/0] self._make_launchers()
E0826 19:29:24.353000 12284 .venv\Lib\site-packages\torch\_inductor\select_algorithm.py:3541] [1/0] File "C:\pinokio\api\overworld.pinokio.git\app\biome\server-components\.venv\Lib\site-packages\torch\_inductor\runtime\triton_heuristics.py", line 664, in _make_launchers
E0826 19:29:24.353000 12284 .venv\Lib\site-packages\torch\_inductor\select_algorithm.py:3541] [1/0] raise RuntimeError(f"No valid triton configs. {type(exc).__name__}: {exc}")
E0826 19:29:24.353000 12284 .venv\Lib\site-packages\torch\_inductor\select_algorithm.py:3541] [1/0] RuntimeError: No valid triton configs. OutOfMemoryError: out of resource: triton_flex_attention Required: 98304 Hardware limit:65536 Reducing block sizes or `num_stages` may help.
Autotune Choices Stats:
{"num_choices": 2, "num_triton_choices": 2, "best_kernel": "triton_flex_attention_23", "best_kernel_desc": "ALLOW_TF32='False', BLOCKS_ARE_CONTIGUOUS=False, BLOCK_M=64, BLOCK_N=64, FLOAT32_PRECISION=\"'tf32'\", GQA_SHARED_HEADS=2, HAS_FULL_BLOCKS=True, IS_DIVISIBLE=True, OUTPUT_LOGSUMEXP=False, OUTPUT_MAX=False, PRESCALE_QK=False, QK_HEAD_DIM=64, QK_HEAD_DIM_ROUNDED=64, ROWS_GUARANTEED_SAFE=False, SAFE_HEAD_DIM=True, SM_SCALE=0.125, SPARSE_KV_BLOCK_SIZE=128, SPARSE_Q_BLOCK_SIZE=128, USE_TMA=False, V_HEAD_DIM=64, V_HEAD_DIM_ROUNDED=64, WRITE_DQ=True, num_stages=3, num_warps=4", "best_time": 41.233985900878906, "best_triton_pos": 0}
AUTOTUNE flex_attention(1x32x128x64, 1x16x16512x64, 1x16x16512x64, 1x32x128, 1x32x128, 1x1x1, 1x1x1x129, 1x1x1, 1x1x1x129)
strides: [262144, 8192, 64, 1], [16908288, 1056768, 64, 1], [16908288, 1056768, 64, 1], [4096, 128, 1], [4096, 128, 1], [1, 1, 1], [129, 129, 129, 1], [1, 1, 1], [0, 0, 0, 1]
dtypes: torch.bfloat16, torch.bfloat16, torch.bfloat16, torch.float32, torch.float32, torch.int32, torch.int32, torch.int32, torch.int32
triton_flex_attention_23 41.2340 ms 100.0% ALLOW_TF32='False', BLOCKS_ARE_CONTIGUOUS=False, BLOCK_M=64, BLOCK_N=64, FLOAT32_PRECISION="'tf32'", GQA_SHARED_HEADS=2, HAS_FULL_BLOCKS=True, IS_DIVISIBLE=True, OUTPUT_LOGSUMEXP=False, OUTPUT_MAX=False, PRESCALE_QK=False, QK_HEAD_DIM=64, QK_HEAD_DIM_ROUNDED=64, ROWS_GUARANTEED_SAFE=False, SAFE_HEAD_DIM=True, SM_SCALE=0.125, SPARSE_KV_BLOCK_SIZE=128, SPARSE_Q_BLOCK_SIZE=128, USE_TMA=False, V_HEAD_DIM=64, V_HEAD_DIM_ROUNDED=64, WRITE_DQ=True, num_stages=3, num_warps=4
triton_flex_attention_22 141.0580 ms 29.2% ALLOW_TF32='False', BLOCKS_ARE_CONTIGUOUS=False, BLOCK_M=64, BLOCK_N=128, FLOAT32_PRECISION="'tf32'", GQA_SHARED_HEADS=2, HAS_FULL_BLOCKS=True, IS_DIVISIBLE=True, OUTPUT_LOGSUMEXP=False, OUTPUT_MAX=False, PRESCALE_QK=False, QK_HEAD_DIM=64, QK_HEAD_DIM_ROUNDED=64, ROWS_GUARANTEED_SAFE=False, SAFE_HEAD_DIM=True, SM_SCALE=0.125, SPARSE_KV_BLOCK_SIZE=128, SPARSE_Q_BLOCK_SIZE=128, USE_TMA=False, V_HEAD_DIM=64, V_HEAD_DIM_ROUNDED=64, WRITE_DQ=True, num_stages=3, num_warps=4
SingleProcess AUTOTUNE benchmarking takes 1.6052 seconds and 32.6211 seconds precompiling for 2 choices
E0826 19:29:26.996000 12284 .venv\Lib\site-packages\torch\_inductor\select_algorithm.py:3924] [1/0] Runtime error during autotuning:
E0826 19:29:26.996000 12284 .venv\Lib\site-packages\torch\_inductor\select_algorithm.py:3924] [1/0] No valid triton configs. OutOfMemoryError: out of resource: triton_flex_attention Required: 65536 Hardware limit:65536 Reducing block sizes or `num_stages` may help..
E0826 19:29:26.996000 12284 .venv\Lib\site-packages\torch\_inductor\select_algorithm.py:3924] [1/0] Ignoring this choice.
E0826 19:29:27.001000 12284 .venv\Lib\site-packages\torch\_inductor\select_algorithm.py:3924] [1/0] Runtime error during autotuning:
E0826 19:29:27.001000 12284 .venv\Lib\site-packages\torch\_inductor\select_algorithm.py:3924] [1/0] No valid triton configs. OutOfMemoryError: out of resource: triton_flex_attention Required: 98304 Hardware limit:65536 Reducing block sizes or `num_stages` may help..
E0826 19:29:27.001000 12284 .venv\Lib\site-packages\torch\_inductor\select_algorithm.py:3924] [1/0] Ignoring this choice.
E0826 19:29:27.004000 12284 .venv\Lib\site-packages\torch\_inductor\select_algorithm.py:3924] [1/0] Runtime error during autotuning:
E0826 19:29:27.004000 12284 .venv\Lib\site-packages\torch\_inductor\select_algorithm.py:3924] [1/0] No valid triton configs. OutOfMemoryError: out of resource: triton_flex_attention Required: 98304 Hardware limit:65536 Reducing block sizes or `num_stages` may help..
E0826 19:29:27.004000 12284 .venv\Lib\site-packages\torch\_inductor\select_algorithm.py:3924] [1/0] Ignoring this choice.
E0826 19:29:27.007000 12284 .venv\Lib\site-packages\torch\_inductor\select_algorithm.py:3924] [1/0] Runtime error during autotuning:
E0826 19:29:27.007000 12284 .venv\Lib\site-packages\torch\_inductor\select_algorithm.py:3924] [1/0] No valid triton configs. OutOfMemoryError: out of resource: triton_flex_attention Required: 98304 Hardware limit:65536 Reducing block sizes or `num_stages` may help..
E0826 19:29:27.007000 12284 .venv\Lib\site-packages\torch\_inductor\select_algorithm.py:3924] [1/0] Ignoring this choice.
Autotune Choices Stats:
{"num_choices": 6, "num_triton_choices": 6, "best_kernel": "triton_flex_attention_47", "best_kernel_desc": "ALLOW_TF32='False', BLOCKS_ARE_CONTIGUOUS=False, BLOCK_M=64, BLOCK_N=64, FLOAT32_PRECISION=\"'tf32'\", GQA_SHARED_HEADS=2, HAS_FULL_BLOCKS=True, IS_DIVISIBLE=True, OUTPUT_LOGSUMEXP=False, OUTPUT_MAX=False, PRESCALE_QK=False, QK_HEAD_DIM=64, QK_HEAD_DIM_ROUNDED=64, ROWS_GUARANTEED_SAFE=False, SAFE_HEAD_DIM=True, SM_SCALE=0.125, SPARSE_KV_BLOCK_SIZE=128, SPARSE_Q_BLOCK_SIZE=128, USE_TMA=False, V_HEAD_DIM=64, V_HEAD_DIM_ROUNDED=64, WRITE_DQ=True, num_stages=3, num_warps=4", "best_time": 41.24262237548828, "best_triton_pos": 0}
AUTOTUNE flex_attention(1x32x128x64, 1x16x16512x64, 1x16x16512x64, 1x32x128, 1x32x128, 1x1x1, 1x1x1x129, 1x1x1, 1x1x1x129)
strides: [262144, 8192, 64, 1], [16908288, 1056768, 64, 1], [16908288, 1056768, 64, 1], [4096, 128, 1], [4096, 128, 1], [1, 1, 1], [129, 129, 129, 1], [1, 1, 1], [0, 0, 0, 1]
dtypes: torch.bfloat16, torch.bfloat16, torch.bfloat16, torch.float32, torch.float32, torch.int32, torch.int32, torch.int32, torch.int32
triton_flex_attention_47 41.2426 ms 100.0% ALLOW_TF32='False', BLOCKS_ARE_CONTIGUOUS=False, BLOCK_M=64, BLOCK_N=64, FLOAT32_PRECISION="'tf32'", GQA_SHARED_HEADS=2, HAS_FULL_BLOCKS=True, IS_DIVISIBLE=True, OUTPUT_LOGSUMEXP=False, OUTPUT_MAX=False, PRESCALE_QK=False, QK_HEAD_DIM=64, QK_HEAD_DIM_ROUNDED=64, ROWS_GUARANTEED_SAFE=False, SAFE_HEAD_DIM=True, SM_SCALE=0.125, SPARSE_KV_BLOCK_SIZE=128, SPARSE_Q_BLOCK_SIZE=128, USE_TMA=False, V_HEAD_DIM=64, V_HEAD_DIM_ROUNDED=64, WRITE_DQ=True, num_stages=3, num_warps=4
triton_flex_attention_46 141.5188 ms 29.1% ALLOW_TF32='False', BLOCKS_ARE_CONTIGUOUS=False, BLOCK_M=64, BLOCK_N=128, FLOAT32_PRECISION="'tf32'", GQA_SHARED_HEADS=2, HAS_FULL_BLOCKS=True, IS_DIVISIBLE=True, OUTPUT_LOGSUMEXP=False, OUTPUT_MAX=False, PRESCALE_QK=False, QK_HEAD_DIM=64, QK_HEAD_DIM_ROUNDED=64, ROWS_GUARANTEED_SAFE=False, SAFE_HEAD_DIM=True, SM_SCALE=0.125, SPARSE_KV_BLOCK_SIZE=128, SPARSE_Q_BLOCK_SIZE=128, USE_TMA=False, V_HEAD_DIM=64, V_HEAD_DIM_ROUNDED=64, WRITE_DQ=True, num_stages=3, num_warps=4
triton_flex_attention_42 inf ms 0.0% ALLOW_TF32='False', BLOCKS_ARE_CONTIGUOUS=False, BLOCK_M=128, BLOCK_N=64, FLOAT32_PRECISION="'tf32'", GQA_SHARED_HEADS=2, HAS_FULL_BLOCKS=True, IS_DIVISIBLE=True, OUTPUT_LOGSUMEXP=False, OUTPUT_MAX=False, PRESCALE_QK=False, QK_HEAD_DIM=64, QK_HEAD_DIM_ROUNDED=64, ROWS_GUARANTEED_SAFE=False, SAFE_HEAD_DIM=True, SM_SCALE=0.125, SPARSE_KV_BLOCK_SIZE=128, SPARSE_Q_BLOCK_SIZE=128, USE_TMA=False, V_HEAD_DIM=64, V_HEAD_DIM_ROUNDED=64, WRITE_DQ=True, num_stages=3, num_warps=4
triton_flex_attention_43 inf ms 0.0% ALLOW_TF32='False', BLOCKS_ARE_CONTIGUOUS=False, BLOCK_M=128, BLOCK_N=128, FLOAT32_PRECISION="'tf32'", GQA_SHARED_HEADS=2, HAS_FULL_BLOCKS=True, IS_DIVISIBLE=True, OUTPUT_LOGSUMEXP=False, OUTPUT_MAX=False, PRESCALE_QK=False, QK_HEAD_DIM=64, QK_HEAD_DIM_ROUNDED=64, ROWS_GUARANTEED_SAFE=False, SAFE_HEAD_DIM=True, SM_SCALE=0.125, SPARSE_KV_BLOCK_SIZE=128, SPARSE_Q_BLOCK_SIZE=128, USE_TMA=False, V_HEAD_DIM=64, V_HEAD_DIM_ROUNDED=64, WRITE_DQ=True, num_stages=3, num_warps=4
triton_flex_attention_44 inf ms 0.0% ALLOW_TF32='False', BLOCKS_ARE_CONTIGUOUS=False, BLOCK_M=128, BLOCK_N=128, FLOAT32_PRECISION="'tf32'", GQA_SHARED_HEADS=2, HAS_FULL_BLOCKS=True, IS_DIVISIBLE=True, OUTPUT_LOGSUMEXP=False, OUTPUT_MAX=False, PRESCALE_QK=False, QK_HEAD_DIM=64, QK_HEAD_DIM_ROUNDED=64, ROWS_GUARANTEED_SAFE=False, SAFE_HEAD_DIM=True, SM_SCALE=0.125, SPARSE_KV_BLOCK_SIZE=128, SPARSE_Q_BLOCK_SIZE=128, USE_TMA=False, V_HEAD_DIM=64, V_HEAD_DIM_ROUNDED=64, WRITE_DQ=True, num_stages=2, num_warps=8
triton_flex_attention_45 inf ms 0.0% ALLOW_TF32='False', BLOCKS_ARE_CONTIGUOUS=False, BLOCK_M=128, BLOCK_N=128, FLOAT32_PRECISION="'tf32'", GQA_SHARED_HEADS=2, HAS_FULL_BLOCKS=True, IS_DIVISIBLE=True, OUTPUT_LOGSUMEXP=False, OUTPUT_MAX=False, PRESCALE_QK=False, QK_HEAD_DIM=64, QK_HEAD_DIM_ROUNDED=64, ROWS_GUARANTEED_SAFE=False, SAFE_HEAD_DIM=True, SM_SCALE=0.125, SPARSE_KV_BLOCK_SIZE=128, SPARSE_Q_BLOCK_SIZE=128, USE_TMA=False, V_HEAD_DIM=64, V_HEAD_DIM_ROUNDED=64, WRITE_DQ=True, num_stages=1, num_warps=8
SingleProcess AUTOTUNE benchmarking takes 1.3140 seconds and 0.0000 seconds precompiling for 6 choices
C:\pinokio\api\overworld.pinokio.git\app\biome\server-components\.venv\Lib\site-packages\torch\_inductor\compile_fx.py:2941: UserWarning: NVIDIA GeForce RTX 2080 does not support bfloat16 compilation natively, skipping
warnings.warn(
[prepare] Still working: Warmup 2/4: append seed frame and compile seed path (121s in this step, 720s total). Torch/Triton compile can take several minutes.
C:\pinokio\api\overworld.pinokio.git\app\biome\server-components\.venv\Lib\site-packages\torch\_inductor\compile_fx.py:2941: UserWarning: NVIDIA GeForce RTX 2080 does not support bfloat16 compilation natively, skipping
warnings.warn(
C:\pinokio\api\overworld.pinokio.git\app\biome\server-components\.venv\Lib\site-packages\torch\_inductor\compile_fx.py:2941: UserWarning: NVIDIA GeForce RTX 2080 does not support bfloat16 compilation natively, skipping
warnings.warn(
C:\pinokio\api\overworld.pinokio.git\app\biome\server-components\.venv\Lib\site-packages\torch\_inductor\compile_fx.py:2941: UserWarning: NVIDIA GeForce RTX 2080 does not support bfloat16 compilation natively, skipping
warnings.warn(
C:\pinokio\api\overworld.pinokio.git\app\biome\server-components\.venv\Lib\site-packages\torch\_inductor\compile_fx.py:2941: UserWarning: NVIDIA GeForce RTX 2080 does not support bfloat16 compilation natively, skipping
warnings.warn(
C:\pinokio\api\overworld.pinokio.git\app\biome\server-components\.venv\Lib\site-packages\torch\_inductor\compile_fx.py:2941: UserWarning: NVIDIA GeForce RTX 2080 does not support bfloat16 compilation natively, skipping
warnings.warn(
C:\pinokio\api\overworld.pinokio.git\app\biome\server-components\.venv\Lib\site-packages\torch\_inductor\compile_fx.py:2941: UserWarning: NVIDIA GeForce RTX 2080 does not support bfloat16 compilation natively, skipping
warnings.warn(
C:\pinokio\api\overworld.pinokio.git\app\biome\server-components\.venv\Lib\site-packages\torch\_dynamo\variables\functions.py:2202: UserWarning: Dynamo detected a call to a `functools.lru_cache`-wrapped function at 'einops.py:539'. Dynamo ignores the cache wrapper and directly traces the wrapped function. Silent incorrectness is only a *potential* risk, not something we have observed. Enable TORCH_LOGS=+dynamo for a DEBUG stack trace.
This call originates from:
File "C:\pinokio\api\overworld.pinokio.git\app\biome\server-components\.venv\Lib\site-packages\einops\einops.py", line 539, in reduce
recipe = _prepare_transformation_recipe(pattern, reduction, axes_names=tuple(axes_lengths), ndim=len(shape))
torch._dynamo.utils.warn_once(msg)
C:\pinokio\api\overworld.pinokio.git\app\biome\server-components\.venv\Lib\site-packages\torch\_dynamo\variables\functions.py:2202: UserWarning: Dynamo detected a call to a `functools.lru_cache`-wrapped function at 'einops.py:235'. Dynamo ignores the cache wrapper and directly traces the wrapped function. Silent incorrectness is only a *potential* risk, not something we have observed. Enable TORCH_LOGS=+dynamo for a DEBUG stack trace.
This call originates from:
File "C:\pinokio\api\overworld.pinokio.git\app\biome\server-components\.venv\Lib\site-packages\einops\einops.py", line 235, in _apply_recipe
init_shapes, axes_reordering, reduced_axes, added_axes, final_shapes, n_axes_w_added = _reconstruct_from_shape(
torch._dynamo.utils.warn_once(msg)
C:\pinokio\api\overworld.pinokio.git\app\biome\server-components\.venv\Lib\site-packages\torch\_inductor\compile_fx.py:2941: UserWarning: NVIDIA GeForce RTX 2080 does not support bfloat16 compilation natively, skipping
warnings.warn(
C:\pinokio\api\overworld.pinokio.git\app\biome\server-components\.venv\Lib\site-packages\torch\_inductor\compile_fx.py:2941: UserWarning: NVIDIA GeForce RTX 2080 does not support bfloat16 compilation natively, skipping
warnings.warn(
C:\pinokio\api\overworld.pinokio.git\app\biome\server-components\.venv\Lib\site-packages\torch\_inductor\compile_fx.py:2941: UserWarning: NVIDIA GeForce RTX 2080 does not support bfloat16 compilation natively, skipping
warnings.warn(
C:\pinokio\api\overworld.pinokio.git\app\biome\server-components\.venv\Lib\site-packages\torch\_inductor\compile_fx.py:2941: UserWarning: NVIDIA GeForce RTX 2080 does not support bfloat16 compilation natively, skipping
warnings.warn(
[prepare] Still working: Warmup 2/4: append seed frame and compile seed path (151s in this step, 750s total). Torch/Triton compile can take several minutes.
C:\pinokio\api\overworld.pinokio.git\app\biome\server-components\.venv\Lib\site-packages\torch\_inductor\compile_fx.py:2941: UserWarning: NVIDIA GeForce RTX 2080 does not support bfloat16 compilation natively, skipping
warnings.warn(
C:\pinokio\api\overworld.pinokio.git\app\biome\server-components\.venv\Lib\site-packages\torch\_inductor\compile_fx.py:2941: UserWarning: NVIDIA GeForce RTX 2080 does not support bfloat16 compilation natively, skipping
warnings.warn(
C:\pinokio\api\overworld.pinokio.git\app\biome\server-components\.venv\Lib\site-packages\torch\_inductor\compile_fx.py:2941: UserWarning: NVIDIA GeForce RTX 2080 does not support bfloat16 compilation natively, skipping
warnings.warn(
C:\pinokio\api\overworld.pinokio.git\app\biome\server-components\.venv\Lib\site-packages\torch\_inductor\compile_fx.py:2941: UserWarning: NVIDIA GeForce RTX 2080 does not support bfloat16 compilation natively, skipping
warnings.warn(
C:\pinokio\api\overworld.pinokio.git\app\biome\server-components\.venv\Lib\site-packages\torch\_inductor\compile_fx.py:2941: UserWarning: NVIDIA GeForce RTX 2080 does not support bfloat16 compilation natively, skipping
warnings.warn(
C:\pinokio\api\overworld.pinokio.git\app\biome\server-components\.venv\Lib\site-packages\torch\_inductor\compile_fx.py:2941: UserWarning: NVIDIA GeForce RTX 2080 does not support bfloat16 compilation natively, skipping
warnings.warn(
C:\pinokio\api\overworld.pinokio.git\app\biome\server-components\.venv\Lib\site-packages\torch\_inductor\compile_fx.py:2941: UserWarning: NVIDIA GeForce RTX 2080 does not support bfloat16 compilation natively, skipping
warnings.warn(
C:\pinokio\api\overworld.pinokio.git\app\biome\server-components\.venv\Lib\site-packages\torch\_inductor\compile_fx.py:2941: UserWarning: NVIDIA GeForce RTX 2080 does not support bfloat16 compilation natively, skipping
warnings.warn(
C:\pinokio\api\overworld.pinokio.git\app\biome\server-components\.venv\Lib\site-packages\gemlite\triton_kernels\gemm_kernels.py:384:24: error: 'arith.extf' op operand #0 must be floating-point-like, but got 'tensor<64x32xi8, #ttg.dot_op<{opIdx = 0, parent = #ttg.blocked<{sizePerThread = [4, 4], threadsPerWarp = [4, 8], warpsPerCTA = [4, 1], order = [1, 0]}>}>>'
acc = tl.dot(a, b.to(input_dtype), acc=acc, out_dtype=acc_dtype)
^
module {
tt.func public @gemm_INT_kernel(%arg0: !tt.ptr<i8> {tt.divisibility = 16 : i32}, %arg1: !tt.ptr<i8> {tt.divisibility = 16 : i32}, %arg2: !tt.ptr<bf16> {tt.divisibility = 16 : i32}, %arg3: !tt.ptr<f32> {tt.divisibility = 16 : i32}, %arg4: !tt.ptr<i32> {tt.divisibility = 16 : i32}, %arg5: !tt.ptr<f32> {tt.divisibility = 16 : i32}, %arg6: i32 {tt.divisibility = 16 : i32}, %arg7: i32 {tt.divisibility = 16 : i32}, %arg8: i32 {tt.divisibility = 16 : i32}, %arg9: i32 {tt.divisibility = 16 : i32}, %arg10: i32 {tt.divisibility = 16 : i32}, %arg11: i32 {tt.divisibility = 16 : i32}, %arg12: i32 {tt.divisibility = 16 : i32}, %arg13: i1) attributes {noinline = false} {
%cst = arith.constant dense<0> : tensor<64x32xi32>
%c63_i32 = arith.constant 63 : i32
%c8_i32 = arith.constant 8 : i32
%c31_i32 = arith.constant 31 : i32
%cst_0 = arith.constant dense<1.000000e+00> : tensor<32xf32>
%cst_1 = arith.constant dense<1.000000e+00> : tensor<64xf32>
%cst_2 = arith.constant dense<0> : tensor<64x32xi8>
%c1_i32 = arith.constant 1 : i32
%c0_i32 = arith.constant 0 : i32
%cst_3 = arith.constant dense<32> : tensor<32x32xi32>
%cst_4 = arith.constant dense<32> : tensor<64x32xi32>
%c32_i32 = arith.constant 32 : i32
%c64_i32 = arith.constant 64 : i32
%0 = tt.get_program_id x : i32
%1 = arith.addi %arg6, %c63_i32 : i32
%2 = arith.divsi %1, %c64_i32 : i32
%3 = arith.addi %arg7, %c31_i32 : i32
%4 = arith.divsi %3, %c32_i32 : i32
%5 = arith.muli %4, %c8_i32 : i32
%6 = arith.divsi %0, %5 : i32
%7 = arith.muli %6, %c8_i32 : i32
%8 = arith.subi %2, %7 : i32
%9 = arith.minsi %8, %c8_i32 : i32
%10 = arith.remsi %0, %9 : i32
%11 = arith.addi %7, %10 : i32
%12 = arith.remsi %0, %5 : i32
%13 = arith.divsi %12, %9 : i32
%14 = arith.addi %arg8, %c31_i32 : i32
%15 = arith.divsi %14, %c32_i32 : i32
%16 = arith.muli %11, %c64_i32 : i32
%17 = tt.make_range {end = 64 : i32, start = 0 : i32} : tensor<64xi32>
%18 = tt.splat %16 : i32 -> tensor<64xi32>
%19 = arith.addi %18, %17 {tt.contiguity = dense<64> : tensor<1xi32>, tt.divisibility = dense<64> : tensor<1xi32>} : tensor<64xi32>
%20 = arith.muli %13, %c32_i32 : i32
%21 = tt.make_range {end = 32 : i32, start = 0 : i32} : tensor<32xi32>
%22 = tt.splat %20 : i32 -> tensor<32xi32>
%23 = arith.addi %22, %21 : tensor<32xi32>
%24 = tt.expand_dims %21 {axis = 1 : i32} : tensor<32xi32> -> tensor<32x1xi32>
%25 = tt.expand_dims %23 {axis = 0 : i32} : tensor<32xi32> -> tensor<1x32xi32>
%26 = tt.splat %arg11 : i32 -> tensor<1x32xi32>
%27 = arith.muli %25, %26 : tensor<1x32xi32>
%28 = tt.broadcast %24 : tensor<32x1xi32> -> tensor<32x32xi32>
%29 = tt.broadcast %27 : tensor<1x32xi32> -> tensor<32x32xi32>
%30 = arith.addi %28, %29 : tensor<32x32xi32>
%31 = tt.splat %arg1 : !tt.ptr<i8> -> tensor<32x32x!tt.ptr<i8>>
%32 = tt.addptr %31, %30 : tensor<32x32x!tt.ptr<i8>>, tensor<32x32xi32>
%33 = tt.expand_dims %19 {axis = 1 : i32} : tensor<64xi32> -> tensor<64x1xi32>
%34 = tt.splat %arg10 : i32 -> tensor<64x1xi32>
%35 = arith.muli %33, %34 : tensor<64x1xi32>
%36 = tt.expand_dims %21 {axis = 0 : i32} : tensor<32xi32> -> tensor<1x32xi32>
%37 = tt.broadcast %35 : tensor<64x1xi32> -> tensor<64x32xi32>
%38 = tt.broadcast %36 : tensor<1x32xi32> -> tensor<64x32xi32>
%39 = arith.addi %37, %38 : tensor<64x32xi32>
%40 = tt.splat %arg0 : !tt.ptr<i8> -> tensor<64x32x!tt.ptr<i8>>
%41 = tt.addptr %40, %39 : tensor<64x32x!tt.ptr<i8>>, tensor<64x32xi32>
%42 = tt.splat %arg6 : i32 -> tensor<64x1xi32>
%43 = arith.cmpi slt, %33, %42 : tensor<64x1xi32>
%44 = tt.splat %arg8 : i32 -> tensor<1x32xi32>
%45 = arith.cmpi slt, %36, %44 : tensor<1x32xi32>
%46 = tt.broadcast %43 : tensor<64x1xi1> -> tensor<64x32xi1>
%47 = tt.broadcast %45 : tensor<1x32xi1> -> tensor<64x32xi1>
%48 = arith.andi %46, %47 : tensor<64x32xi1>
%49:3 = scf.for %arg14 = %c0_i32 to %15 step %c1_i32 iter_args(%arg15 = %32, %arg16 = %41, %arg17 = %cst) -> (tensor<32x32x!tt.ptr<i8>>, tensor<64x32x!tt.ptr<i8>>, tensor<64x32xi32>) : i32 {
%85 = tt.load %arg16, %48, %cst_2 : tensor<64x32x!tt.ptr<i8>>
%86 = tt.load %arg15 : tensor<32x32x!tt.ptr<i8>>
%87 = tt.dot %85, %86, %arg17, inputPrecision = tf32 : tensor<64x32xi8> * tensor<32x32xi8> -> tensor<64x32xi32>
%88 = tt.addptr %arg16, %cst_4 : tensor<64x32x!tt.ptr<i8>>, tensor<64x32xi32>
%89 = tt.addptr %arg15, %cst_3 : tensor<32x32x!tt.ptr<i8>>, tensor<32x32xi32>
scf.yield %89, %88, %87 : tensor<32x32x!tt.ptr<i8>>, tensor<64x32x!tt.ptr<i8>>, tensor<64x32xi32>
}
%50 = tt.splat %arg6 : i32 -> tensor<64xi32>
%51 = arith.cmpi slt, %19, %50 : tensor<64xi32>
%52 = tt.splat %arg5 : !tt.ptr<f32> -> tensor<64x!tt.ptr<f32>>
%53 = tt.addptr %52, %19 : tensor<64x!tt.ptr<f32>>, tensor<64xi32>
%54 = tt.load %53, %51, %cst_1 : tensor<64x!tt.ptr<f32>>
%55 = tt.splat %arg7 : i32 -> tensor<32xi32>
%56 = arith.cmpi slt, %23, %55 : tensor<32xi32>
%57 = tt.splat %arg3 : !tt.ptr<f32> -> tensor<32x!tt.ptr<f32>>
%58 = tt.addptr %57, %23 : tensor<32x!tt.ptr<f32>>, tensor<32xi32>
%59 = tt.load %58, %56, %cst_0 : tensor<32x!tt.ptr<f32>>
%60 = arith.sitofp %49#2 : tensor<64x32xi32> to tensor<64x32xf32>
%61 = tt.expand_dims %54 {axis = 1 : i32} : tensor<64xf32> -> tensor<64x1xf32>
%62 = tt.expand_dims %59 {axis = 0 : i32} : tensor<32xf32> -> tensor<1x32xf32>
%63 = tt.broadcast %61 : tensor<64x1xf32> -> tensor<64x32xf32>
%64 = tt.broadcast %62 : tensor<1x32xf32> -> tensor<64x32xf32>
%65 = arith.mulf %63, %64 : tensor<64x32xf32>
%66 = arith.mulf %60, %65 : tensor<64x32xf32>
%67 = arith.truncf %66 : tensor<64x32xf32> to tensor<64x32xbf16>
%68 = arith.addi %18, %17 : tensor<64xi32>
%69 = arith.addi %22, %21 {tt.contiguity = dense<32> : tensor<1xi32>, tt.divisibility = dense<32> : tensor<1xi32>} : tensor<32xi32>
%70 = tt.expand_dims %68 {axis = 1 : i32} : tensor<64xi32> -> tensor<64x1xi32>
%71 = tt.splat %arg12 : i32 -> tensor<64x1xi32>
%72 = arith.muli %70, %71 : tensor<64x1xi32>
%73 = tt.expand_dims %69 {axis = 0 : i32} : tensor<32xi32> -> tensor<1x32xi32>
%74 = tt.broadcast %72 : tensor<64x1xi32> -> tensor<64x32xi32>
%75 = tt.broadcast %73 : tensor<1x32xi32> -> tensor<64x32xi32>
%76 = arith.addi %74, %75 : tensor<64x32xi32>
%77 = tt.splat %arg2 : !tt.ptr<bf16> -> tensor<64x32x!tt.ptr<bf16>>
%78 = tt.addptr %77, %76 : tensor<64x32x!tt.ptr<bf16>>, tensor<64x32xi32>
%79 = arith.cmpi slt, %70, %42 : tensor<64x1xi32>
%80 = tt.splat %arg7 :i32 -> tensor<1x32xi32>
%81 = arith.cmpi slt, %73, %80 : tensor<1x32xi32>
%82 = tt.broadcast %79 : tensor<64x1xi1> -> tensor<64x32xi1>
%83 = tt.broadcast %81 : tensor<1x32xi1> -> tensor<64x32xi1>
%84 = arith.andi %82, %83 : tensor<64x32xi1>
tt.store %78, %67, %84 : tensor<64x32x!tt.ptr<bf16>>
tt.return
}
}
{-#
external_resources: {
mlir_reproducer: {
pipeline: "builtin.module(convert-triton-to-tritongpu{enable-source-remat=false num-ctas=1 num-warps=4 target=cuda:75 threads-per-warp=32}, tritongpu-coalesce, tritongpu-F32DotTC{emu-tf32=false}, triton-nvidia-gpu-plan-cta, tritongpu-remove-layout-conversions, tritongpu-optimize-thread-locality, tritongpu-accelerate-matmul, tritongpu-remove-layout-conversions, tritongpu-optimize-dot-operands{hoist-layout-conversion=false}, triton-nvidia-optimize-descriptor-encoding, triton-loop-aware-cse, triton-licm, canonicalize{ max-iterations=10 max-num-rewrites=-1 region-simplify=normal test-convergence=false top-down=true}, triton-loop-aware-cse, tritongpu-prefetch, tritongpu-optimize-dot-operands{hoist-layout-conversion=false}, tritongpu-coalesce-async-copy, triton-nvidia-optimize-tmem-layouts, tritongpu-remove-layout-conversions, triton-nvidia-interleave-tmem, tritongpu-reduce-data-duplication, tritongpu-reorder-instructions, triton-loop-aware-cse, symbol-dce, triton-nvidia-gpu-fence-insertion{compute-capability=75}, triton-nvidia-mma-lowering, sccp, cse, canonicalize{ max-iterations=10 max-num-rewrites=-1 region-simplify=normal test-convergence=false top-down=true})",
disable_threading: true,
verify_each: true
}
}
#-}
C:\pinokio\api\overworld.pinokio.git\app\biome\server-components\.venv\Lib\site-packages\gemlite\triton_kernels\gemm_kernels.py:249:0: error: Failures have been detected while processing an MLIR pass pipeline
C:\pinokio\api\overworld.pinokio.git\app\biome\server-components\.venv\Lib\site-packages\gemlite\triton_kernels\gemm_kernels.py:249:0: note: Pipeline failed while executing [`TritonGPUAccelerateMatmul` on 'builtin.module' operation]: reproducer generated at `std::errs, please share the reproducer above with Triton project.`
Traceback (most recent call last):
File "C:\pinokio\api\overworld.pinokio.git\app\prepare_profile.py", line 208, in <module>
main()
File "C:\pinokio\api\overworld.pinokio.git\app\prepare_profile.py", line 204, in main
asyncio.run(prepare(args.profile, args.seed))
File "C:\Users\mesut\AppData\Roaming\uv\python\cpython-3.12-windows-x86_64-none\Lib\asyncio\runners.py", line 195, in run
return runner.run(main)
^^^^^^^^^^^^^^^^
File "C:\Users\mesut\AppData\Roaming\uv\python\cpython-3.12-windows-x86_64-none\Lib\asyncio\runners.py", line 118, in run
return self._loop.run_until_complete(task)
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
File "C:\Users\mesut\AppData\Roaming\uv\python\cpython-3.12-windows-x86_64-none\Lib\asyncio\base_events.py", line 691, in run_until_complete
return future.result()
^^^^^^^^^^^^^^^
File "C:\pinokio\api\overworld.pinokio.git\app\prepare_profile.py", line 152, in prepare
await manager.warmup()
File "C:\pinokio\api\overworld.pinokio.git\app\biome\server-components\engine_manager.py", line 656, in warmup
warmup_time = await self._run_on_cuda_thread(do_warmup)
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
File "C:\pinokio\api\overworld.pinokio.git\app\biome\server-components\engine_manager.py", line 214, in _run_on_cuda_thread
return await loop.run_in_executor(self.cuda_executor, fn)
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
File "C:\Users\mesut\AppData\Roaming\uv\python\cpython-3.12-windows-x86_64-none\Lib\concurrent\futures\thread.py", line 59, in run
result = self.fn(*self.args, **self.kwargs)
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
File "C:\pinokio\api\overworld.pinokio.git\app\biome\server-components\engine_manager.py", line 619, in do_warmup
self.engine.append_frame(self.seed_frame)
File "C:\pinokio\api\overworld.pinokio.git\app\biome\server-components\.venv\Lib\site-packages\torch\utils\_contextlib.py", line 124, in decorate_context
return func(*args, **kwargs)
^^^^^^^^^^^^^^^^^^^^^
File "C:\pinokio\api\overworld.pinokio.git\app\biome\server-components\.venv\Lib\site-packages\world_engine\world_engine.py", line 143, in append_frame
self._cache_pass(x0, inputs, self.kv_cache)
File "C:\pinokio\api\overworld.pinokio.git\app\biome\server-components\.venv\Lib\site-packages\torch\_dynamo\eval_frame.py", line 1024, in compile_wrapper
return fn(*args, **kwargs)
^^^^^^^^^^^^^^^^^^^
File "C:\pinokio\api\overworld.pinokio.git\app\biome\server-components\.venv\Lib\site-packages\world_engine\world_engine.py", line 198, in _cache_pass
self.model(x, x.new_zeros((x.size(0), x.size(1))), **ctx, kv_cache=kv_cache)
File "C:\pinokio\api\overworld.pinokio.git\app\biome\server-components\.venv\Lib\site-packages\torch\nn\modules\module.py", line 1779, in _wrapped_call_impl
return self._call_impl(*args, **kwargs)
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
File "C:\pinokio\api\overworld.pinokio.git\app\biome\server-components\.venv\Lib\site-packages\torch\nn\modules\module.py", line 1790, in _call_impl
return forward_call(*args, **kwargs)
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
File "C:\pinokio\api\overworld.pinokio.git\app\biome\server-components\.venv\Lib\site-packages\world_engine\model\world_model.py", line 347, in forward
x = self.transformer(x, pos_ids, cond, ctx, kv_cache)
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
File "C:\pinokio\api\overworld.pinokio.git\app\biome\server-components\.venv\Lib\site-packages\torch\nn\modules\module.py", line 1779, in _wrapped_call_impl
return self._call_impl(*args, **kwargs)
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
File "C:\pinokio\api\overworld.pinokio.git\app\biome\server-components\.venv\Lib\site-packages\torch\nn\modules\module.py", line 1790, in _call_impl
return forward_call(*args, **kwargs)
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
File "C:\pinokio\api\overworld.pinokio.git\app\biome\server-components\.venv\Lib\site-packages\world_engine\model\world_model.py", line 251, in forward
x, v = block(x, pos_ids, rope_angles, cond, ctx, v, kv_cache=kv_cache)
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
File "C:\pinokio\api\overworld.pinokio.git\app\biome\server-components\.venv\Lib\site-packages\torch\nn\modules\module.py", line 1779, in _wrapped_call_impl
return self._call_impl(*args, **kwargs)
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
File "C:\pinokio\api\overworld.pinokio.git\app\biome\server-components\.venv\Lib\site-packages\torch\nn\modules\module.py", line 1790, in _call_impl
return forward_call(*args, **kwargs)
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
File "C:\pinokio\api\overworld.pinokio.git\app\biome\server-components\.venv\Lib\site-packages\world_engine\model\world_model.py", line 213, in forward
x, v = self.attn(x, pos_ids, rope_angles, v, kv_cache=kv_cache)
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
File "C:\pinokio\api\overworld.pinokio.git\app\biome\server-components\.venv\Lib\site-packages\torch\nn\modules\module.py", line 1779, in _wrapped_call_impl
return self._call_impl(*args, **kwargs)
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
File "C:\pinokio\api\overworld.pinokio.git\app\biome\server-components\.venv\Lib\site-packages\torch\nn\modules\module.py", line 1790, in _call_impl
return forward_call(*args, **kwargs)
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
File "C:\pinokio\api\overworld.pinokio.git\app\biome\server-components\.venv\Lib\site-packages\world_engine\patch_model.py", line 117, in forward
q, k, v = self.qkv_proj(x).split((self.q_out, self.kv_out, self.kv_out), dim=-1)
^^^^^^^^^^^^^^^^
File "C:\pinokio\api\overworld.pinokio.git\app\biome\server-components\.venv\Lib\site-packages\torch\nn\modules\module.py", line 1779, in _wrapped_call_impl
return self._call_impl(*args, **kwargs)
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
File "C:\pinokio\api\overworld.pinokio.git\app\biome\server-components\.venv\Lib\site-packages\torch\nn\modules\module.py", line 1790, in _call_impl
return forward_call(*args, **kwargs)
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
File "C:\pinokio\api\overworld.pinokio.git\app\biome\server-components\.venv\Lib\site-packages\world_engine\quantize.py", line 270, in forward
return self.impl(x).type_as(x)
^^^^^^^^^^^^
File "C:\pinokio\api\overworld.pinokio.git\app\biome\server-components\.venv\Lib\site-packages\torch\nn\modules\module.py", line 1779, in _wrapped_call_impl
return self._call_impl(*args, **kwargs)
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
File "C:\pinokio\api\overworld.pinokio.git\app\biome\server-components\.venv\Lib\site-packages\torch\nn\modules\module.py", line 1790, in _call_impl
return forward_call(*args, **kwargs)
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
File "C:\pinokio\api\overworld.pinokio.git\app\biome\server-components\.venv\Lib\site-packages\gemlite\core.py", line 552, in forward_auto_no_warmup
return forward_functional(
^^^^^^^^^^^^^^^^^^^
File "C:\pinokio\api\overworld.pinokio.git\app\biome\server-components\.venv\Lib\site-packages\torch\_library\custom_ops.py", line 698, in __call__
return self._opoverload(*args, **kwargs)
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
File "C:\pinokio\api\overworld.pinokio.git\app\biome\server-components\.venv\Lib\site-packages\torch\_ops.py", line 865, in __call__
return self._op(*args, **kwargs)
^^^^^^^^^^^^^^^^^^^^^^^^^
File "C:\pinokio\api\overworld.pinokio.git\app\biome\server-components\.venv\Lib\site-packages\torch\_library\custom_ops.py", line 347, in backend_impl
result = self._backend_fns[device_type](*args, **kwargs)
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
File "C:\pinokio\api\overworld.pinokio.git\app\biome\server-components\.venv\Lib\site-packages\torch\_compile.py", line 54, in inner
return disable_fn(*args, **kwargs)
^^^^^^^^^^^^^^^^^^^^^^^^^^^
File "C:\pinokio\api\overworld.pinokio.git\app\biome\server-components\.venv\Lib\site-packages\torch\_dynamo\eval_frame.py", line 1263, in _fn
return fn(*args, **kwargs)
^^^^^^^^^^^^^^^^^^^
File "C:\pinokio\api\overworld.pinokio.git\app\biome\server-components\.venv\Lib\site-packages\torch\_library\custom_ops.py", line 382, in wrapped_fn
return fn(*args, **kwargs)
^^^^^^^^^^^^^^^^^^^
File "C:\pinokio\api\overworld.pinokio.git\app\biome\server-components\.venv\Lib\site-packages\gemlite\core.py", line 185, in forward_functional
GEMLITE_TRITON_MAPPING[matmul_type_str]
File "C:\pinokio\api\overworld.pinokio.git\app\biome\server-components\.venv\Lib\site-packages\gemlite\triton_kernels\gemm_kernels.py", line 575, in gemm_forward
gemm_kernel[grid](
File "C:\pinokio\api\overworld.pinokio.git\app\biome\server-components\.venv\Lib\site-packages\triton\runtime\jit.py", line 370, in <lambda>
return lambda *args, **kwargs: self.run(grid=grid, warmup=False, *args, **kwargs)
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
File "C:\pinokio\api\overworld.pinokio.git\app\biome\server-components\.venv\Lib\site-packages\triton\runtime\autotuner.py", line 240, in run
benchmark()
File "C:\pinokio\api\overworld.pinokio.git\app\biome\server-components\.venv\Lib\site-packages\triton\runtime\autotuner.py", line 229, in benchmark
timings = {config: self._bench(*args, config=config, **kwargs) for config in pruned_configs}
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
File "C:\pinokio\api\overworld.pinokio.git\app\biome\server-components\.venv\Lib\site-packages\triton\runtime\autotuner.py", line 164, in _bench
return self.do_bench(kernel_call, quantiles=(0.5, 0.2, 0.8))
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
File "C:\pinokio\api\overworld.pinokio.git\app\biome\server-components\.venv\Lib\site-packages\triton\testing.py", line 149, in do_bench
fn()
File "C:\pinokio\api\overworld.pinokio.git\app\biome\server-components\.venv\Lib\site-packages\triton\runtime\autotuner.py", line 150, in kernel_call
self.fn.run(
File "C:\pinokio\api\overworld.pinokio.git\app\biome\server-components\.venv\Lib\site-packages\triton\runtime\jit.py", line 720, in run
kernel = self._do_compile(key, signature, device, constexprs, options, attrs, warmup)
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
File "C:\pinokio\api\overworld.pinokio.git\app\biome\server-components\.venv\Lib\site-packages\triton\runtime\jit.py", line 849, in _do_compile
kernel = self.compile(src, target=target, options=options.__dict__)
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
File "C:\pinokio\api\overworld.pinokio.git\app\biome\server-components\.venv\Lib\site-packages\triton\compiler\compiler.py", line 324, in compile
next_module = compile_ir(module, metadata)
^^^^^^^^^^^^^^^^^^^^^^^^^^^^
File "C:\pinokio\api\overworld.pinokio.git\app\biome\server-components\.venv\Lib\site-packages\triton\backends\nvidia\compiler.py", line 550, in <lambda>
stages["ttgir"] = lambda src, metadata: self.make_ttgir(src, metadata, options, capability)
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
File "C:\pinokio\api\overworld.pinokio.git\app\biome\server-components\.venv\Lib\site-packages\triton\backends\nvidia\compiler.py", line 325, in make_ttgir
pm.run(mod, 'make_ttgir')
RuntimeError: PassManager::run failed
logs/api/install.js/1787762056937
Source: api / install.js
Lines: 9 total, last 9 included
[api shell.run]
Microsoft Windows [Version 10.0.26200.9168]
(c) Microsoft Corporation. Tüm hakları saklıdır.
C:\pinokio\api\overworld.pinokio.git\app\biome>conda_hook & conda deactivate & conda deactivate & conda deactivate & conda activate base && git fetch origin main
From https://github.com/Overworldai/Biome
* branch main -> FETCH_HEAD
(base) C:\pinokio\api\overworld.pinokio.git\app\biome>
logs/api/install.js/1787762066388
Source: api / install.js
Lines: 7 total, last 7 included
[api shell.run]
Microsoft Windows [Version 10.0.26200.9168]
(c) Microsoft Corporation. Tüm hakları saklıdır.
C:\pinokio\api\overworld.pinokio.git\app\biome>conda_hook & conda deactivate & conda deactivate & conda deactivate & conda activate base && git sparse-checkout init --cone
(base) C:\pinokio\api\overworld.pinokio.git\app\biome>
logs/api/install.js/1787762073006
Source: api / install.js
Lines: 7 total, last 7 included
[api shell.run]
Microsoft Windows [Version 10.0.26200.9168]
(c) Microsoft Corporation. Tüm hakları saklıdır.
C:\pinokio\api\overworld.pinokio.git\app\biome>conda_hook & conda deactivate & conda deactivate & conda deactivate & conda activate base && git sparse-checkout set server-components seeds
(base) C:\pinokio\api\overworld.pinokio.git\app\biome>
logs/api/install.js/1787762079555
Source: api / install.js
Lines: 8 total, last 8 included
[api shell.run]
Microsoft Windows [Version 10.0.26200.9168]
(c) Microsoft Corporation. Tüm hakları saklıdır.
C:\pinokio\api\overworld.pinokio.git\app\biome>conda_hook & conda deactivate & conda deactivate & conda deactivate & conda activate base && git checkout a1ae46a
HEAD is now at a1ae46a Merge pull request #104 from Overworldai/torch-2.11
(base) C:\pinokio\api\overworld.pinokio.git\app\biome>
logs/api/install.js/1787762086080
Source: api / install.js
Lines: 12 total, last 12 included
[api shell.run]
Microsoft Windows [Version 10.0.26200.9168]
(c) Microsoft Corporation. Tüm hakları saklıdır.
C:\pinokio\api\overworld.pinokio.git\app\biome\server-components>conda_hook & conda deactivate & conda deactivate & conda deactivate & conda activate base & C:\pinokio\api\overworld.pinokio.git\app\biome\server-components\.venv\Scripts\activate C:\pinokio\api\overworld.pinokio.git\app\biome\server-components\.venv && uv sync
Resolved 96 packages in 2.65s
Uninstalled 1 package in 100ms
Installed 1 package in 68ms
- llama-cpp-python==0.3.36 (from https://github.com/JamePeng/llama-cpp-python/releases/download/v0.3.36-cu128-Basic-win-20260417/llama_cpp_python-0.3.36%2Bcu128.basic-cp312-cp312-win_amd64.whl)
+ llama-cpp-python==0.3.36+cu128.basic (from https://github.com/JamePeng/llama-cpp-python/releases/download/v0.3.36-cu128-Basic-win-20260417/llama_cpp_python-0.3.36%2Bcu128.basic-cp312-cp312-win_amd64.whl)
(biome-server) (base) C:\pinokio\api\overworld.pinokio.git\app\biome\server-components>
logs/api/install.js/1787762095937
Source: api / install.js
Lines: 158 total, last 158 included
[api shell.run]
Microsoft Windows [Version 10.0.26200.9168]
(c) Microsoft Corporation. Tüm hakları saklıdır.
C:\pinokio\api\overworld.pinokio.git\app\biome\server-components>conda_hook & conda deactivate & conda deactivate & conda deactivate & conda activate base & C:\pinokio\api\overworld.pinokio.git\app\biome\server-components\.venv\Scripts\activate C:\pinokio\api\overworld.pinokio.git\app\biome\server-components\.venv && uv run python ../../prepare_profile.py --profile low
Uninstalled 1 package in 104ms
Installed 1 package in 70ms
[prepare] Profile: Low VRAM 360p INT8
[prepare] Model: Overworld/Waypoint-1.5-1B-360P quant=intw8a8
[prepare] Seed: C:\pinokio\api\overworld.pinokio.git\app\biome\seeds\default.jpg
[prepare] Downloading/loading model files
C:\pinokio\api\overworld.pinokio.git\app\biome\server-components\.venv\Lib\site-packages\torch\cuda\__init__.py:209: UserWarning: expandable_segments not supported on this platform (Triggered internally at C:\actions-runner\_work\pytorch\pytorch\pytorch\c10/cuda/CUDAAllocatorConfig.h:39.)
torch.tensor([1.0], dtype=torch.bfloat16, device=device)
[prepare] Loading WorldEngine code
[prepare] Loading model
Warning: You are sending unauthenticated requests to the HF Hub. Please set a HF_TOKEN to enable higher rate limits and faster downloads.
Fetching 3 files: 100%|█████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████| 3/3 [00:00<00:00, 865.70it/s]
Download complete: : 0.00B [00:00, ?B/s] | 0/3 [00:00<?, ?it/s]
Fetching 2 files: 100%|█████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████| 2/2 [00:00<00:00, 682.61it/s]
Download complete: : 0.00B [00:00, ?B/s] | 0/2 [00:00<?, ?it/s]
C:\pinokio\api\overworld.pinokio.git\app\biome\server-components\.venv\Lib\site-packages\torch\nn\modules\module.py:934: UserWarning: NoiseConditioner: requested dtype cast ignored; keeping torch.float32.
module._apply(fn)
C:\pinokio\api\overworld.pinokio.git\app\biome\server-components\.venv\Lib\site-packages\world_engine\model\nn.py:24: UserWarning: NoiseConditioner: requested dtype cast ignored; keeping torch.float32.
return super()._apply(keep_dtype)
C:\pinokio\api\overworld.pinokio.git\app\biome\server-components\.venv\Lib\site-packages\world_engine\model\nn.py:24: UserWarning: OrthoRoPEAngles: requested dtype cast ignored; keeping torch.float32.
return super()._apply(keep_dtype)
[prepare] Model loaded in 11.23s
[prepare] Loading seed image
[prepare] Instantiating model weights
[prepare] Model load complete
[prepare] Starting warmup compile
[prepare] Warmup 1/4: reset engine state
[prepare] Warmup 2/4: append seed frame and compile seed path
C:\pinokio\api\overworld.pinokio.git\app\biome\server-components\.venv\Lib\site-packages\torch\_inductor\compile_fx.py:2941: UserWarning: NVIDIA GeForce RTX 2080 does not support bfloat16 compilation natively, skipping
warnings.warn(
[prepare] Still working: Warmup 2/4: append seed frame and compile seed path (19s in this step, 30s total). Torch/Triton compile can take several minutes.
W0826 19:35:42.678000 3852 .venv\Lib\site-packages\torch\_inductor\utils.py:1731] [1/0] Not enough SMs to use max_autotune_gemm mode
C:\pinokio\api\overworld.pinokio.git\app\biome\server-components\.venv\Lib\site-packages\torch\_inductor\compile_fx.py:2941: UserWarning: NVIDIA GeForce RTX 2080 does not support bfloat16 compilation natively, skipping
warnings.warn(
[prepare] Still working: Warmup 2/4: append seed frame and compile seed path (49s in this step, 60s total). Torch/Triton compile can take several minutes.
C:\pinokio\api\overworld.pinokio.git\app\biome\server-components\.venv\Lib\site-packages\torch\_inductor\compile_fx.py:2941: UserWarning: NVIDIA GeForce RTX 2080 does not support bfloat16 compilation natively, skipping
warnings.warn(
C:\pinokio\api\overworld.pinokio.git\app\biome\server-components\.venv\Lib\site-packages\torch\_inductor\compile_fx.py:2941: UserWarning: NVIDIA GeForce RTX 2080 does not support bfloat16 compilation natively, skipping
warnings.warn(
C:\pinokio\api\overworld.pinokio.git\app\biome\server-components\.venv\Lib\site-packages\torch\_inductor\compile_fx.py:2941: UserWarning: NVIDIA GeForce RTX 2080 does not support bfloat16 compilation natively, skipping
warnings.warn(
C:\pinokio\api\overworld.pinokio.git\app\biome\server-components\.venv\Lib\site-packages\torch\_inductor\compile_fx.py:2941: UserWarning: NVIDIA GeForce RTX 2080 does not support bfloat16 compilation natively, skipping
warnings.warn(
C:\pinokio\api\overworld.pinokio.git\app\biome\server-components\.venv\Lib\site-packages\torch\_inductor\compile_fx.py:2941: UserWarning: NVIDIA GeForce RTX 2080 does not support bfloat16 compilation natively, skipping
warnings.warn(
C:\pinokio\api\overworld.pinokio.git\app\biome\server-components\.venv\Lib\site-packages\torch\_inductor\compile_fx.py:2941: UserWarning: NVIDIA GeForce RTX 2080 does not support bfloat16 compilation natively, skipping
warnings.warn(
C:\pinokio\api\overworld.pinokio.git\app\biome\server-components\.venv\Lib\site-packages\torch\_dynamo\variables\functions.py:2202: UserWarning: Dynamo detected a call to a `functools.lru_cache`-wrapped function at 'einops.py:539'. Dynamo ignores the cache wrapper and directly traces the wrapped function. Silent incorrectness is only a *potential* risk, not something we have observed. Enable TORCH_LOGS=+dynamo for a DEBUG stack trace.
This call originates from:
File "C:\pinokio\api\overworld.pinokio.git\app\biome\server-components\.venv\Lib\site-packages\einops\einops.py", line 539, in reduce
recipe = _prepare_transformation_recipe(pattern, reduction, axes_names=tuple(axes_lengths), ndim=len(shape))
torch._dynamo.utils.warn_once(msg)
C:\pinokio\api\overworld.pinokio.git\app\biome\server-components\.venv\Lib\site-packages\torch\_dynamo\variables\functions.py:2202: UserWarning: Dynamo detected a call to a `functools.lru_cache`-wrapped function at 'einops.py:235'. Dynamo ignores the cache wrapper and directly traces the wrapped function. Silent incorrectness is only a *potential* risk, not something we have observed. Enable TORCH_LOGS=+dynamo for a DEBUG stack trace.
This call originates from:
File "C:\pinokio\api\overworld.pinokio.git\app\biome\server-components\.venv\Lib\site-packages\einops\einops.py", line 235, in _apply_recipe
init_shapes, axes_reordering, reduced_axes, added_axes, final_shapes, n_axes_w_added = _reconstruct_from_shape(
torch._dynamo.utils.warn_once(msg)
C:\pinokio\api\overworld.pinokio.git\app\biome\server-components\.venv\Lib\site-packages\torch\_inductor\compile_fx.py:2941: UserWarning: NVIDIA GeForce RTX 2080 does not support bfloat16 compilation natively, skipping
warnings.warn(
C:\pinokio\api\overworld.pinokio.git\app\biome\server-components\.venv\Lib\site-packages\torch\_inductor\compile_fx.py:2941: UserWarning: NVIDIA GeForce RTX 2080 does not support bfloat16 compilation natively, skipping
warnings.warn(
C:\pinokio\api\overworld.pinokio.git\app\biome\server-components\.venv\Lib\site-packages\torch\_inductor\compile_fx.py:2941: UserWarning: NVIDIA GeForce RTX 2080 does not support bfloat16 compilation natively, skipping
warnings.warn(
C:\pinokio\api\overworld.pinokio.git\app\biome\server-components\.venv\Lib\site-packages\torch\_inductor\compile_fx.py:2941: UserWarning: NVIDIA GeForce RTX 2080 does not support bfloat16 compilation natively, skipping
warnings.warn(
C:\pinokio\api\overworld.pinokio.git\app\biome\server-components\.venv\Lib\site-packages\torch\_inductor\compile_fx.py:2941: UserWarning: NVIDIA GeForce RTX 2080 does not support bfloat16 compilation natively, skipping
warnings.warn(
C:\pinokio\api\overworld.pinokio.git\app\biome\server-components\.venv\Lib\site-packages\torch\_inductor\compile_fx.py:2941: UserWarning: NVIDIA GeForce RTX 2080 does not support bfloat16 compilation natively, skipping
warnings.warn(
C:\pinokio\api\overworld.pinokio.git\app\biome\server-components\.venv\Lib\site-packages\torch\_inductor\compile_fx.py:2941: UserWarning: NVIDIA GeForce RTX 2080 does not support bfloat16 compilation natively, skipping
warnings.warn(
C:\pinokio\api\overworld.pinokio.git\app\biome\server-components\.venv\Lib\site-packages\torch\_inductor\compile_fx.py:2941: UserWarning: NVIDIA GeForce RTX 2080 does not support bfloat16 compilation natively, skipping
warnings.warn(
C:\pinokio\api\overworld.pinokio.git\app\biome\server-components\.venv\Lib\site-packages\torch\_inductor\compile_fx.py:2941: UserWarning: NVIDIA GeForce RTX 2080 does not support bfloat16 compilation natively, skipping
warnings.warn(
[prepare] Still working: Warmup 2/4: append seed frame and compile seed path (79s in this step, 90s total). Torch/Triton compile can take several minutes.
C:\pinokio\api\overworld.pinokio.git\app\biome\server-components\.venv\Lib\site-packages\torch\_inductor\compile_fx.py:2941: UserWarning: NVIDIA GeForce RTX 2080 does not support bfloat16 compilation natively, skipping
warnings.warn(
C:\pinokio\api\overworld.pinokio.git\app\biome\server-components\.venv\Lib\site-packages\torch\_inductor\compile_fx.py:2941: UserWarning: NVIDIA GeForce RTX 2080 does not support bfloat16 compilation natively, skipping
warnings.warn(
C:\pinokio\api\overworld.pinokio.git\app\biome\server-components\.venv\Lib\site-packages\torch\_inductor\compile_fx.py:2941: UserWarning: NVIDIA GeForce RTX 2080 does not support bfloat16 compilation natively, skipping
warnings.warn(
C:\pinokio\api\overworld.pinokio.git\app\biome\server-components\.venv\Lib\site-packages\gemlite\triton_kernels\gemm_kernels.py:384:24: error: 'arith.extf' op operand #0 must be floating-point-like, but got 'tensor<64x32xi8, #ttg.dot_op<{opIdx = 0, parent = #ttg.blocked<{sizePerThread = [4, 4], threadsPerWarp = [4, 8], warpsPerCTA = [4, 1], order = [1, 0]}>}>>'
acc = tl.dot(a, b.to(input_dtype), acc=acc, out_dtype=acc_dtype)
^
module {
tt.func public @gemm_INT_kernel(%arg0: !tt.ptr<i8> {tt.divisibility = 16 : i32}, %arg1: !tt.ptr<i8> {tt.divisibility = 16 : i32}, %arg2: !tt.ptr<bf16> {tt.divisibility = 16 : i32}, %arg3: !tt.ptr<f32> {tt.divisibility = 16 : i32}, %arg4: !tt.ptr<i32> {tt.divisibility = 16 : i32}, %arg5: !tt.ptr<f32> {tt.divisibility = 16 : i32}, %arg6: i32 {tt.divisibility = 16 : i32}, %arg7: i32 {tt.divisibility = 16 : i32}, %arg8: i32 {tt.divisibility = 16 : i32}, %arg9: i32 {tt.divisibility = 16 : i32}, %arg10: i32 {tt.divisibility = 16 : i32}, %arg11: i32 {tt.divisibility = 16 : i32}, %arg12: i32 {tt.divisibility = 16 : i32}, %arg13: i1) attributes {noinline = false} {
%cst = arith.constant dense<0> : tensor<64x32xi32>
%c63_i32 = arith.constant 63 : i32
%c8_i32 = arith.constant 8 : i32
%c31_i32 = arith.constant 31 : i32
%cst_0 = arith.constant dense<1.000000e+00> : tensor<32xf32>
%cst_1 = arith.constant dense<1.000000e+00> : tensor<64xf32>
%cst_2 = arith.constant dense<0> : tensor<64x32xi8>
%c1_i32 = arith.constant 1 : i32
%c0_i32 = arith.constant 0 : i32
%cst_3 = arith.constant dense<32> : tensor<32x32xi32>
%cst_4 = arith.constant dense<32> : tensor<64x32xi32>
%c32_i32 = arith.constant 32 : i32
%c64_i32 = arith.constant 64 : i32
%0 = tt.get_program_id x : i32
%1 = arith.addi %arg6, %c63_i32 : i32
%2 = arith.divsi %1, %c64_i32 : i32
%3 = arith.addi %arg7, %c31_i32 : i32
%4 = arith.divsi %3, %c32_i32 : i32
%5 = arith.muli %4, %c8_i32 : i32
%6 = arith.divsi %0, %5 : i32
%7 = arith.muli %6, %c8_i32 : i32
%8 = arith.subi %2, %7 : i32
%9 = arith.minsi %8, %c8_i32 : i32
%10 = arith.remsi %0, %9 : i32
%11 = arith.addi %7, %10 : i32
%12 = arith.remsi %0, %5: i32
%13 = arith.divsi %12, %9 : i32
%14 = arith.addi %arg8, %c31_i32 : i32
%15 = arith.divsi %14, %c32_i32 : i32
%16 = arith.muli %11, %c64_i32 : i32
%17 = tt.make_range {end = 64 : i32, start = 0 : i32} : tensor<64xi32>
%18 = tt.splat %16 : i32 -> tensor<64xi32>
%19 = arith.addi %18, %17 {tt.contiguity = dense<64> : tensor<1xi32>, tt.divisibility = dense<64> : tensor<1xi32>} : tensor<64xi32>
%20 = arith.muli %13, %c32_i32 : i32
%21 = tt.make_range {end = 32 : i32, start = 0 : i32} : tensor<32xi32>
%22 = tt.splat %20 : i32 -> tensor<32xi32>
%23 = arith.addi %22, %21 : tensor<32xi32>
%24 = tt.expand_dims %21 {axis = 1 : i32} : tensor<32xi32> -> tensor<32x1xi32>
%25 = tt.expand_dims %23 {axis = 0 : i32} : tensor<32xi32> -> tensor<1x32xi32>
%26 = tt.splat %arg11 : i32 -> tensor<1x32xi32>
%27 = arith.muli %25, %26 : tensor<1x32xi32>
%28 = tt.broadcast %24 : tensor<32x1xi32> -> tensor<32x32xi32>
%29 = tt.broadcast %27 : tensor<1x32xi32> -> tensor<32x32xi32>
%30 = arith.addi %28, %29 : tensor<32x32xi32>
%31 = tt.splat %arg1 : !tt.ptr<i8> -> tensor<32x32x!tt.ptr<i8>>
%32 = tt.addptr %31, %30 : tensor<32x32x!tt.ptr<i8>>, tensor<32x32xi32>
%33 = tt.expand_dims %19 {axis = 1 : i32} : tensor<64xi32> -> tensor<64x1xi32>
%34 = tt.splat %arg10 : i32 -> tensor<64x1xi32>
%35 = arith.muli %33, %34 : tensor<64x1xi32>
%36 = tt.expand_dims %21 {axis = 0 : i32} : tensor<32xi32> -> tensor<1x32xi32>
%37 = tt.broadcast %35 : tensor<64x1xi32> -> tensor<64x32xi32>
%38 = tt.broadcast %36 : tensor<1x32xi32> -> tensor<64x32xi32>
%39 = arith.addi %37, %38 : tensor<64x32xi32>
%40 = tt.splat %arg0 : !tt.ptr<i8> -> tensor<64x32x!tt.ptr<i8>>
%41 = tt.addptr %40, %39 : tensor<64x32x!tt.ptr<i8>>, tensor<64x32xi32>
%42 = tt.splat %arg6 : i32 -> tensor<64x1xi32>
%43 = arith.cmpi slt, %33, %42 : tensor<64x1xi32>
%44 = tt.splat %arg8 : i32 -> tensor<1x32xi32>
%45 = arith.cmpi slt, %36, %44 : tensor<1x32xi32>
%46 = tt.broadcast %43 : tensor<64x1xi1> -> tensor<64x32xi1>
%47 = tt.broadcast %45 : tensor<1x32xi1> -> tensor<64x32xi1>
%48 = arith.andi %46, %47 : tensor<64x32xi1>
