Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
Show all changes
24 commits
Select commit Hold shift + click to select a range
a873d03
Add install script for physics engine dependency
Qazi-pk Jun 17, 2026
f81860b
Add metadata for PIR project
Qazi-pk Jun 17, 2026
25f12b7
Add POT to requirements.txt
Qazi-pk Jun 17, 2026
5d2fc16
Add PIRClassicRegressor and model interface
Qazi-pk Jun 17, 2026
460ed0e
Delete algorithms/PIR directory
Qazi-pk Jun 18, 2026
11fc617
Add install script for physics engine dependency
Qazi-pk Jun 18, 2026
ad35232
Add metadata for Physics Intermediate Representation
Qazi-pk Jun 18, 2026
45ef46b
Add POT to requirements.txt
Qazi-pk Jun 18, 2026
0c83abc
Add PIRClassicRegressor for SRBench integration
Qazi-pk Jun 18, 2026
0d23da5
Update install.sh
Qazi-pk Jun 30, 2026
be0d6ea
PIR: clean docstring, update to verified 12/44 EXACT blind result
Qazi-pk Jul 2, 2026
4a8d175
PIR: update metadata description to verified 12/44 EXACT blind result
Qazi-pk Jul 2, 2026
bcf02f1
PIR: add experiment/methods/pir package init
Qazi-pk Jul 13, 2026
a84b132
PIR: add importable regressor module at experiment/methods/pir (fixes…
Qazi-pk Jul 13, 2026
3b3c763
Delete algorithms/pir/regressor.py
Qazi-pk Jul 13, 2026
a36618d
Update metadata.yml
Qazi-pk Jul 14, 2026
14ec12c
Update regressor.py
Qazi-pk Jul 14, 2026
0d88c57
Merge branch 'master' into master
lacava Jul 31, 2026
91a77a3
Merge branch 'master' into master
lacava Aug 13, 2026
65dbb60
Merge branch 'master' into master
lacava Aug 13, 2026
d2ae93e
Add required top-level name/email/url keys to pir metadata.yml
Qazi-pk Aug 18, 2026
f58ae2d
PIR:
Qazi-pk Jun 17, 2026
969e323
Address review feedback on PR #210: fix complexity operator count, wo…
Qazi-pk Sep 18, 2026
42cb935
Remove leftover algorithms/PIR/ (capitalized) directory — a case-sens…
Qazi-pk Sep 21, 2026
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
2 changes: 2 additions & 0 deletions algorithms/pir/install.sh
Original file line number Diff line number Diff line change
@@ -0,0 +1,2 @@
#!/bin/bash
pip install "git+https://github.com/Qazi-pk/[email protected]"
19 changes: 19 additions & 0 deletions algorithms/pir/metadata.yml
Original file line number Diff line number Diff line change
@@ -0,0 +1,19 @@
# PIR — Physics Intermediate Representation
name: PIR
authors:
- name: Qazi Hanif
email: [email protected]
email: [email protected]
url: https://github.com/Qazi-pk/physics-engine
key: PIR
paper:
title: "PIR: Physics Intermediate Representation for Automated Discovery of Physical Laws"
url: https://doi.org/10.5281/zenodo.21351039
description: >
Classical symbolic regression via monomial-basis search with log-linearization
gate for power-law detection (F3 gate), pairwise structure decomposition,
RANSAC, sparse regression, and iterative residual refinement with Occam
complexity penalty. No neural components. Verified blind Feynman Tier A:
12/44 EXACT (zero seed wobble, v3.4). Secondary: +12/44 FORM_NUMERIC
(correct functional form, transcendental constant as decimal, reported
separately and never summed into the primary figure).
1 change: 1 addition & 0 deletions algorithms/pir/requirements.txt
Original file line number Diff line number Diff line change
@@ -0,0 +1 @@
POT
1 change: 1 addition & 0 deletions experiment/methods/pir/__init__.py
Original file line number Diff line number Diff line change
@@ -0,0 +1 @@
# PIR method package
181 changes: 181 additions & 0 deletions experiment/methods/pir/regressor.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,181 @@
"""
SRBench method: PIR (Physics Intermediate Representation) — classical engine.

Importable module for the SRBench harness (imported as `methods.pir`).
Wraps the public, pip-installable classical PIR engine (installed via
algorithms/pir/install.sh) in the interface SRBench expects:

est a scikit-learn-compatible Regressor instance
model(est, X=None) returns a sympy-parseable string for the fitted model
complexity(est) returns an integer complexity count for the model
eval_kwargs method-specific args forwarded to evaluate_model.py

No neural components. Vanilla classical PIR only — this is the configuration
that produced the verified blind Tier A baseline:

12/44 EXACT (zero seed wobble, v3.4)
+12/44 FORM_NUMERIC secondary (correct functional form, transcendental
constant folded as decimal — reported separately, never summed)

Engine: https://github.com/Qazi-pk/physics-engine (MIT, tag v3.4.1)
Paper: https://doi.org/10.5281/zenodo.21351039
"""

import re

import numpy as np

from sklearn.base import BaseEstimator, RegressorMixin

# The classical engine. algorithms/pir/install.sh pip-installs this from the
# public repo at tag v3.4.1.
from physics_engine.sklearn_adapter import PIRRegressor


# --- Vanilla config that produced the verified blind Tier A result ------------
_PIR_VANILLA_CONFIG = dict(
enforce_dimensions=False, # blind sweep ran with dim-filter OFF
allowed_powers=[1, 2], # powers 1,2 only (structural cap)
include_pairwise_products=True, # pairwise on; no 3-var assembly
use_ransac=True,
use_residual=True,
use_sparse=True,
use_ot_loss=False,
add_physics_features=False,
)


class PIRClassicRegressor(BaseEstimator, RegressorMixin):
"""Thin sklearn wrapper around the classical PIRRegressor.

Adds a `random_state` attribute and a guaranteed-valid fallback model so
that model() never raises even if the inner fit fails. Timeout
enforcement is left entirely to SRBench's own outer alarm: signal.alarm
is process-global, so a second SIGALRM set here would silently overwrite
SRBench's own timeout handler (per gAldeia's review on PR #210). The
`max_time` constructor kwarg is still accepted, for API compatibility
with the harness's test_params override, but is currently unused.
"""

def __init__(self, max_time=3600, random_state=None, **pir_kwargs):
self.max_time = max_time
self.random_state = random_state
self.pir_kwargs = {**_PIR_VANILLA_CONFIG, **pir_kwargs}

def _build(self):
kw = dict(self.pir_kwargs)
seed = 0 if self.random_state is None else self.random_state
try:
return PIRRegressor(random_state=seed, **kw)
except TypeError:
# Engine may not accept random_state; harmless to omit.
return PIRRegressor(**kw)

def fit(self, X, y):
# Guarantee a valid model exists before risking a timeout:
# constant = mean(y). evaluate_model.py can always score this.
y_arr = np.asarray(y, dtype=float).ravel()
self._fallback_expr_ = repr(float(np.mean(y_arr))) if y_arr.size else "0.0"
self.expr_ = self._fallback_expr_
self._inner = self._build()

# No process-level alarm here — SRBench's own outer timeout
# enforces max_time; see class docstring.
try:
self._inner.fit(X, y)
if hasattr(self._inner, "model"):
self.expr_ = self._inner.model()
elif hasattr(self._inner, "expr_"):
self.expr_ = str(self._inner.expr_)
except Exception:
# Keep the constant fallback already stored in self.expr_.
pass
self.is_fitted_ = True
return self

def predict(self, X):
if hasattr(self, "_inner") and hasattr(self._inner, "predict"):
try:
return self._inner.predict(X)
except Exception:
pass
# Fallback: constant prediction matching the fallback expression.
n = X.shape[0] if hasattr(X, "shape") else len(X)
try:
val = float(self._fallback_expr_)
except (TypeError, ValueError):
val = 0.0
return np.full(n, val, dtype=float)

def model(self):
return self.expr_


# The estimator SRBench will fit.
est = PIRClassicRegressor(max_time=3600, random_state=None)


def model(est, X=None):
"""Return a sympy-parseable model string with symbols matching X.columns.

If the engine already emits dataset column names, the string is returned
unchanged. Otherwise generic positional names (x_0, x0, X0, ...) are
remapped to the dataset columns.
"""
expr = (
est.model()
if hasattr(est, "model")
else str(getattr(est, "expr_", "0.0"))
)

if X is None or not hasattr(X, "columns"):
return expr

cols = list(X.columns)
# Word-boundary match, not substring: a column literally named "co"
# must not match inside "cos(x_0)" (per gAldeia's review on PR #210).
def _col_appears(col_name, s):
return re.search(r"\b" + re.escape(str(col_name)) + r"\b", s) is not None
# If any real column name already appears, assume names are correct.
if any(_col_appears(c, expr) for c in cols):
return expr

# Remap generic positional names -> dataset columns.
# reversed() so 'x_1' doesn't clobber the prefix of 'x_10'.
for prefix in ("x_", "x", "X_", "X"):
if re.search(rf"\b{prefix}\d+\b", expr):
mapping = {f"{prefix}{i}": str(k) for i, k in enumerate(cols)}
for k, v in reversed(list(mapping.items())):
expr = re.sub(rf"\b{re.escape(k)}\b", v, expr)
break
return expr


def complexity(est):
"""Integer complexity of the fitted model.

Counts nodes in the expression by splitting on operators and separators,
following the convention used by other SRBench methods (e.g. gplearn).
SRBench normally uses SymPy's own complexity first; this is only a
fallback, but the operator count is included so the fallback is as
accurate as possible (per gAldeia's review on PR #210).
"""
expr = model(est)
if not expr:
return 0
# Count operand-like tokens and add back the operators the split removed.
tokens = re.split(r"[\s\(\),\+\-\*\/\^]+", expr)
operand_count = len([t for t in tokens if t])
operator_count = len(re.findall(r"[\+\-\*\/\^]", expr))
return operand_count + operator_count


# --- forwarded to evaluate_model.py ------------------------------------------
# CRITICAL: scale_x/scale_y MUST be False. SRBench StandardScales X and y by
# default, which destroys the units and exact coefficients PIR depends on.
# `test_params` shortens run-time during CI smoke tests (master-branch signature).
eval_kwargs = {
"scale_x": False,
"scale_y": False,
"test_params": {"max_time": 60},
}
Loading