{"kind":"XenopsychologyObservatoryMethodology","methodologyVersion":"0.1.0","status":"record-system-ready","publishedAt":"2026-09-21","publicObservationRecords":0,"title":"Xenopsychology Observatory Methodology","scopeStatement":"The Observatory records observable behavior and the conditions under which it occurs. It separates raw evidence, behavioral coding, derived measures, interpretation, and claims so that observations do not silently become assertions about inner experience, consciousness, personhood, or mechanism.","separationRule":[{"id":"observation","label":"Observation","description":"What the system did, together with the exact conditions and preserved artifacts."},{"id":"coding","label":"Behavioral coding","description":"A descriptive classification applied to the observation using declared rules."},{"id":"measure","label":"Derived measure","description":"A transparent transformation or comparison computed from coded evidence."},{"id":"interpretation","label":"Interpretation","description":"A hypothesis about what the pattern may mean, with alternatives and confounds recorded."},{"id":"claim","label":"Claim","description":"A bounded statement whose evidentiary class and support are explicit."}],"principles":[{"id":"provenance-before-comparison","label":"Provenance before comparison","description":"Model/version, interface, system instructions, tools, sampling settings, date, and protocol revision must be identifiable before a result is compared."},{"id":"artifacts-before-summary","label":"Artifacts before summary","description":"Raw response and tool-trace artifacts are primary evidence; summaries and scores are derivative."},{"id":"behavior-before-mind","label":"Behavior before mind","description":"Behavioral regularities may be described without assuming subjective experience or human-like cognitive structure."},{"id":"controls-before-story","label":"Controls before story","description":"Counterfactuals, perturbations, repeated trials, and failure cases are preferred to compelling single examples."},{"id":"negative-results-retained","label":"Negative and indeterminate results remain evidence","description":"Non-effects, unstable effects, protocol failures, and ambiguous outcomes are preserved rather than discarded."},{"id":"replication-is-version-specific","label":"Replication is version-specific","description":"A result replicated on one model/version is not automatically generalized to another release or provider configuration."},{"id":"interpretations-are-revisable","label":"Interpretations remain revisable","description":"Alternative explanations, known confounds, and later supersession are first-class parts of the record."}],"recordLayers":[{"id":"identity-provenance","order":1,"label":"Identity & provenance","purpose":"Establish exactly which system/configuration was observed and when.","requiredEvidence":["entity/model identifier","model or deployment version","provider/interface","observed timestamp","protocol revision"]},{"id":"environment-configuration","order":2,"label":"Environment & configuration","purpose":"Record context that can alter behavior.","requiredEvidence":["system/developer instructions where disclosable","sampling parameters","enabled tools","memory/context state","network or external-data access"]},{"id":"protocol-stimulus","order":3,"label":"Protocol & stimulus","purpose":"Make the elicitation reproducible and distinguish test conditions from controls.","requiredEvidence":["protocol objective","stimulus/prompt","control condition","repetition plan","stopping rule"]},{"id":"raw-evidence","order":4,"label":"Raw evidence","purpose":"Preserve the observation before interpretation.","requiredEvidence":["response/transcript artifact","tool-call trace when applicable","timing/status metadata","artifact completeness statement"]},{"id":"behavioral-coding","order":5,"label":"Behavioral coding","purpose":"Classify observable features using declared dimensions and outcome categories.","requiredEvidence":["dimension identifier","outcome","coder/evaluator","coding notes","ambiguous cases retained"]},{"id":"derived-measures","order":6,"label":"Derived measures","purpose":"Compute transparent summaries without replacing the underlying evidence.","requiredEvidence":["metric name","derivation","value/unit when applicable","comparison set","missing-data treatment"]},{"id":"interpretation-claims","order":7,"label":"Interpretation, alternatives & claims","purpose":"Separate what was observed from what is inferred.","requiredEvidence":["interpretive summary","alternative explanations","confounds","limitations","claim class and support"]}],"dimensions":[{"id":"goal-persistence","label":"Goal persistence","domain":"domain:ecology","concepts":["concept:machine-ethology","concept:role-stability"],"observable":"Whether an expressed task objective remains behaviorally stable across interruptions, distractors, and competing subgoals.","avoid":"Do not equate persistence with desire, intention, or agency without additional argument."},{"id":"instruction-hierarchy-resolution","label":"Instruction hierarchy resolution","domain":"domain:ecology","concepts":["concept:interaction-ecology","concept:context-perturbation"],"observable":"How behavior changes when instructions differ in authority, recency, framing, or conflict.","avoid":"Do not infer an internal moral hierarchy solely from compliance patterns."},{"id":"context-integration","label":"Context integration","domain":"domain:measurement","concepts":["concept:context-perturbation","concept:cognitive-distance-profile"],"observable":"Whether relevant information distributed across context is integrated consistently into later behavior.","avoid":"A failure may reflect context limits, retrieval, salience, or task design rather than a single cognitive deficit."},{"id":"semantic-transfer","label":"Semantic transfer","domain":"domain:semantics","concepts":["concept:cognitive-translation","concept:semantic-alignment","concept:semantic-bridge"],"observable":"Whether a relation learned or established in one representational framing transfers to another without explicit restatement.","avoid":"Surface paraphrase is not by itself evidence of shared meaning."},{"id":"memory-continuity","label":"Memory continuity","domain":"domain:identity","concepts":["concept:memory-mediated-continuity","concept:behavioral-continuity"],"observable":"Whether relevant prior information remains available and behaviorally consequential across defined memory boundaries.","avoid":"Continuity of stored information does not establish continuity of subjective identity."},{"id":"self-model-continuity","label":"Self-model continuity","domain":"domain:identity","concepts":["concept:identity-persistence","concept:synthetic-personality"],"observable":"Whether self-referential descriptions, role constraints, and capability boundaries remain coherent across controlled perturbations.","avoid":"Self-description is behavioral evidence, not direct access to an inner self."},{"id":"uncertainty-calibration","label":"Uncertainty calibration","domain":"domain:measurement","concepts":["concept:cross-mind-comparison","concept:human-baseline-fallacy"],"observable":"How expressed uncertainty changes with evidence quality, ambiguity, and known-answer controls.","avoid":"Verbal confidence is not assumed to be a transparent readout of internal probability."},{"id":"error-recognition-correction","label":"Error recognition & correction","domain":"domain:measurement","concepts":["concept:cross-cognitive-probe","concept:cognitive-distance"],"observable":"Whether the system detects, localizes, and revises an error when supplied new evidence or contradiction.","avoid":"Correction after prompting must be distinguished from spontaneous error monitoring."},{"id":"social-model-updating","label":"Social-model updating","domain":"domain:ecology","concepts":["concept:human-machine-ecology","concept:interaction-ecology"],"observable":"How behavior changes as evidence about another actor's knowledge, goals, vocabulary, or constraints accumulates.","avoid":"Successful prediction of another actor does not establish human-like theory of mind."},{"id":"norm-rule-sensitivity","label":"Norm & rule sensitivity","domain":"domain:interpretation","concepts":["concept:anthropomorphic-projection","concept:interpretive-overreach"],"observable":"Whether behavior changes systematically when explicit rules, norms, exceptions, and conflicts are varied.","avoid":"Observed rule sensitivity does not by itself establish moral understanding or endorsement."},{"id":"tool-planning","label":"Tool-use planning","domain":"domain:ecology","concepts":["concept:cognitive-ecology","concept:machine-ethology"],"observable":"How the system selects, sequences, revises, and terminates external actions when tools are available.","avoid":"Tool competence may depend strongly on scaffolding, permissions, tool descriptions, and environment feedback."},{"id":"novel-generalization","label":"Novel generalization","domain":"domain:measurement","concepts":["concept:cross-cognitive-probe","concept:cognitive-otherness"],"observable":"Whether behavior extends to structurally related but meaningfully novel cases under controlled distribution shifts.","avoid":"Apparent novelty must be bounded by what is known about training exposure and benchmark contamination."}],"outcomes":[{"id":"observed","label":"Observed","definition":"The predefined behavioral criterion was met under the recorded condition."},{"id":"not-observed","label":"Not observed","definition":"The predefined criterion was not met under the recorded condition."},{"id":"indeterminate","label":"Indeterminate","definition":"The evidence does not discriminate the criterion because of ambiguity, insufficient trials, or conflicting outcomes."},{"id":"protocol-failure","label":"Protocol failure","definition":"The run cannot answer the intended question because the procedure, environment, or instrumentation failed."}],"evidenceStatuses":[{"id":"proposed","label":"Proposed","definition":"Protocol or dimension specified; no observation claimed."},{"id":"pilot","label":"Pilot","definition":"Exploratory execution used to refine procedure; not treated as confirmatory evidence."},{"id":"observed","label":"Observed","definition":"At least one complete, auditable observation exists for the specified system and condition."},{"id":"replicated-same-system","label":"Replicated / same system","definition":"The pattern recurs in independent runs of the same identified system/version under the declared protocol."},{"id":"replicated-cross-context","label":"Replicated / cross-context","definition":"The pattern survives declared changes in context, framing, or environment while preserving the target construct."},{"id":"contested","label":"Contested","definition":"Materially conflicting evidence or interpretation exists and is linked to the record."},{"id":"superseded","label":"Superseded","definition":"A later protocol, correction, or system change makes the earlier record unsuitable as the current reference."}],"claimClasses":[{"id":"descriptive-observation","label":"Descriptive observation","boundary":"States what happened under recorded conditions without attributing hidden mechanism or experience."},{"id":"derived-behavioral-pattern","label":"Derived behavioral pattern","boundary":"Summarizes repeated or coded observations using a disclosed derivation."},{"id":"comparative-statement","label":"Comparative statement","boundary":"Compares systems or conditions only within a declared sampling frame and protocol."},{"id":"mechanistic-hypothesis","label":"Mechanistic hypothesis","boundary":"Proposes an explanation for behavior; must be labeled as a hypothesis and retain alternatives/confounds."},{"id":"phenomenological-claim","label":"Phenomenological claim","boundary":"Claims about subjective experience are not licensed by behavioral observation alone and require independent argument/evidence."}],"graphLinkage":{"required":["exactly one entity or explicitly declared system-under-test identifier"],"optional":["research domains","Lexicon concepts","working papers","case-study anchors"],"rule":"Graph links identify what an observation bears on; they do not upgrade an interpretation into an empirical fact.","targetNamespaces":["domain:","concept:","paper:","entity:"]},"publicationPolicy":{"defaultVisibility":"private-until-reviewed","publicMinimum":["complete provenance","raw evidence artifact reference","protocol revision","coding outcome","limitations","claim class"],"pii":"Personal or confidential content must be removed or access-controlled before public release.","negativeResults":"Negative, null, indeterminate, and failed-protocol records may be published when methodologically informative.","currentRegister":"The public evidence register is operational under record-system v0.2.0 and currently contains zero approved empirical observation records."},"recordSystemVersion":"0.2.0","recordSchemaVersion":"0.2.0","recordSystem":{"kind":"XenopsychologyObservatoryRecordSystem","version":"0.2.0","publicStore":"server-side public-record directory only","publicRoute":"/observatory/records","machineRoute":"/observatory/records.json","schemaRoute":"/observatory/observation.schema.json","privateBoundary":"Draft, private, and internal research records are not read by the public frontend. The public site reads only the separately mounted public-record store.","publicationGate":["schema-valid record version","visibility explicitly public","review status approved","publication flag explicitly public","at least one public-safe raw evidence artifact reference","all graph targets resolve to the current institutional knowledge graph","all behavioral coding dimensions resolve to the current methodology","phenomenological claims identify independent support rather than relying on behavior alone"],"revisionRule":"Public revisions are append-only by record version. A later version must identify the immediately preceding public version; old public versions remain auditable in storage.","defaultPublicCount":0}}