Publish party-position estimates and validation materials
This commit is contained in:
@@ -0,0 +1,10 @@
|
|||||||
|
# party2d_estimates agent notes
|
||||||
|
|
||||||
|
- For repositories on `git.seimel.app`, use the existing Tea/Gitea token from `~/.config/tea/config.yml` for authenticated `git push`/`git fetch` operations.
|
||||||
|
- Do not print, log, or expose the token.
|
||||||
|
- Prefer token-authenticated HTTPS over unauthenticated HTTPS or SSH when pushing to `git.seimel.app`.
|
||||||
|
- Python is forbidden for this project. Public workflow code and documentation must use R, Julia, Stan, and shell only; do not add Python scripts or Python references.
|
||||||
|
- `data-setup/run_data_setup.sh` is intentionally a no-option public entry point. It downloads script-accessible sources, checks local raw files, rebuilds generated inputs under `_local/`, compares them to committed `data/`, and never replaces committed inputs automatically.
|
||||||
|
- This is a public release repository. Commit only publication-ready code, data, documentation, figures, and release metadata. Keep working notes, private correspondence, assessment records, decision logs, draft response material, and non-public planning outside this repository.
|
||||||
|
- Before every public push, run `bash scripts/check_public_content.sh`. Install the version-controlled pre-push guard with `bash scripts/install_public_guard.sh` in each local clone.
|
||||||
|
- If material was committed here in error, stop publication work. Removing it in a later commit is insufficient: create a clean replacement history or repository before resuming publication.
|
||||||
@@ -0,0 +1,19 @@
|
|||||||
|
CC BY 4.0
|
||||||
|
|
||||||
|
Creative Commons Attribution 4.0 International
|
||||||
|
|
||||||
|
This repository is released under the Creative Commons Attribution 4.0
|
||||||
|
International License (CC BY 4.0), except where third-party source terms
|
||||||
|
require otherwise.
|
||||||
|
|
||||||
|
You are free to share and adapt the licensed material for any purpose,
|
||||||
|
including commercial use, provided that appropriate credit is given, a link to
|
||||||
|
the license is provided, and any changes are indicated.
|
||||||
|
|
||||||
|
This license applies to the dataset, metadata, documentation, diagnostics, and
|
||||||
|
repository workflow materials created for this release. It does not relicense
|
||||||
|
third-party source data used to construct the release inputs; those sources
|
||||||
|
remain governed by their own terms.
|
||||||
|
|
||||||
|
License deed: https://creativecommons.org/licenses/by/4.0/
|
||||||
|
Legal code: https://creativecommons.org/licenses/by/4.0/legalcode
|
||||||
+663
@@ -0,0 +1,663 @@
|
|||||||
|
# This file is machine-generated - editing it directly is not advised
|
||||||
|
|
||||||
|
julia_version = "1.11.3"
|
||||||
|
manifest_format = "2.0"
|
||||||
|
project_hash = "5d7dd35b02e3ab85cb2ccc99d83afe86c825420e"
|
||||||
|
|
||||||
|
[[deps.AbstractFFTs]]
|
||||||
|
deps = ["LinearAlgebra"]
|
||||||
|
git-tree-sha1 = "d92ad398961a3ed262d8bf04a1a2b8340f915fef"
|
||||||
|
uuid = "621f4979-c628-5d54-868e-fcf4e3e8185c"
|
||||||
|
version = "1.5.0"
|
||||||
|
|
||||||
|
[deps.AbstractFFTs.extensions]
|
||||||
|
AbstractFFTsChainRulesCoreExt = "ChainRulesCore"
|
||||||
|
AbstractFFTsTestExt = "Test"
|
||||||
|
|
||||||
|
[deps.AbstractFFTs.weakdeps]
|
||||||
|
ChainRulesCore = "d360d2e6-b24c-11e9-a2a3-2a2ae2dbcce4"
|
||||||
|
Test = "8dfed614-e22c-5e08-85e1-65c5234f0b40"
|
||||||
|
|
||||||
|
[[deps.AliasTables]]
|
||||||
|
deps = ["PtrArrays", "Random"]
|
||||||
|
git-tree-sha1 = "9876e1e164b144ca45e9e3198d0b689cadfed9ff"
|
||||||
|
uuid = "66dad0bd-aa9a-41b7-9441-69ab47430ed8"
|
||||||
|
version = "1.1.3"
|
||||||
|
|
||||||
|
[[deps.ArgTools]]
|
||||||
|
uuid = "0dad84c5-d112-42e6-8d28-ef12dabb789f"
|
||||||
|
version = "1.1.2"
|
||||||
|
|
||||||
|
[[deps.Artifacts]]
|
||||||
|
uuid = "56f22d72-fd6d-98f1-02f0-08ddc0907c33"
|
||||||
|
version = "1.11.0"
|
||||||
|
|
||||||
|
[[deps.Base64]]
|
||||||
|
uuid = "2a0f44e3-6c83-55bd-87e4-b1978d98bd5f"
|
||||||
|
version = "1.11.0"
|
||||||
|
|
||||||
|
[[deps.CSV]]
|
||||||
|
deps = ["CodecZlib", "Dates", "FilePathsBase", "InlineStrings", "Mmap", "Parsers", "PooledArrays", "PrecompileTools", "SentinelArrays", "Tables", "Unicode", "WeakRefStrings", "WorkerUtilities"]
|
||||||
|
git-tree-sha1 = "8d8e0b0f350b8e1c91420b5e64e5de774c2f0f4d"
|
||||||
|
uuid = "336ed68f-0bac-5ca0-87d4-7b16caf5d00b"
|
||||||
|
version = "0.10.16"
|
||||||
|
|
||||||
|
[[deps.CategoricalArrays]]
|
||||||
|
deps = ["Compat", "DataAPI", "Future", "Missings", "Printf", "Requires", "Statistics", "Unicode"]
|
||||||
|
git-tree-sha1 = "20ff1463035a170b25eba2ef9823bb9ad51635e4"
|
||||||
|
uuid = "324d7699-5711-5eae-9e2f-1d82baa6b597"
|
||||||
|
version = "1.1.1"
|
||||||
|
|
||||||
|
[deps.CategoricalArrays.extensions]
|
||||||
|
CategoricalArraysArrowExt = "Arrow"
|
||||||
|
CategoricalArraysJSONExt = "JSON"
|
||||||
|
CategoricalArraysRecipesBaseExt = "RecipesBase"
|
||||||
|
CategoricalArraysSentinelArraysExt = "SentinelArrays"
|
||||||
|
CategoricalArraysStatsBaseExt = "StatsBase"
|
||||||
|
CategoricalArraysStructTypesExt = "StructTypes"
|
||||||
|
|
||||||
|
[deps.CategoricalArrays.weakdeps]
|
||||||
|
Arrow = "69666777-d1a9-59fb-9406-91d4454c9d45"
|
||||||
|
JSON = "682c06a0-de6a-54ab-a142-c8b1cf79cde6"
|
||||||
|
RecipesBase = "3cdcf5f2-1ef4-517c-9805-6587b60abb01"
|
||||||
|
SentinelArrays = "91c51154-3ec4-41a3-a24f-3f23e20d615c"
|
||||||
|
StatsBase = "2913bbd2-ae8a-5f71-8c99-4fb6c76f3a91"
|
||||||
|
StructTypes = "856f2bd8-1eba-4b0a-8007-ebc267875bd4"
|
||||||
|
|
||||||
|
[[deps.CodecZlib]]
|
||||||
|
deps = ["TranscodingStreams", "Zlib_jll"]
|
||||||
|
git-tree-sha1 = "962834c22b66e32aa10f7611c08c8ca4e20749a9"
|
||||||
|
uuid = "944b1d66-785c-5afd-91f1-9de20f533193"
|
||||||
|
version = "0.7.8"
|
||||||
|
|
||||||
|
[[deps.Compat]]
|
||||||
|
deps = ["TOML", "UUIDs"]
|
||||||
|
git-tree-sha1 = "9d8a54ce4b17aa5bdce0ea5c34bc5e7c340d16ad"
|
||||||
|
uuid = "34da2185-b29b-5c13-b0c7-acf172513d20"
|
||||||
|
version = "4.18.1"
|
||||||
|
weakdeps = ["Dates", "LinearAlgebra"]
|
||||||
|
|
||||||
|
[deps.Compat.extensions]
|
||||||
|
CompatLinearAlgebraExt = "LinearAlgebra"
|
||||||
|
|
||||||
|
[[deps.CompatHelperLocal]]
|
||||||
|
deps = ["Pkg"]
|
||||||
|
git-tree-sha1 = "f239f702063ca2fc50844c6cf27b4d497652b474"
|
||||||
|
uuid = "5224ae11-6099-4aaa-941d-3aab004bd678"
|
||||||
|
version = "0.1.29"
|
||||||
|
|
||||||
|
[[deps.CompilerSupportLibraries_jll]]
|
||||||
|
deps = ["Artifacts", "Libdl"]
|
||||||
|
uuid = "e66e0078-7015-5450-92f7-15fbd957f2ae"
|
||||||
|
version = "1.1.1+0"
|
||||||
|
|
||||||
|
[[deps.Crayons]]
|
||||||
|
git-tree-sha1 = "249fe38abf76d48563e2f4556bebd215aa317e15"
|
||||||
|
uuid = "a8cc5b0e-0ffa-5ad4-8c14-923d3ee1735f"
|
||||||
|
version = "4.1.1"
|
||||||
|
|
||||||
|
[[deps.DataAPI]]
|
||||||
|
git-tree-sha1 = "abe83f3a2f1b857aac70ef8b269080af17764bbe"
|
||||||
|
uuid = "9a962f9c-6df0-11e9-0e5d-c546b8b5ee8a"
|
||||||
|
version = "1.16.0"
|
||||||
|
|
||||||
|
[[deps.DataFrames]]
|
||||||
|
deps = ["Compat", "DataAPI", "DataStructures", "Future", "InlineStrings", "InvertedIndices", "IteratorInterfaceExtensions", "LinearAlgebra", "Markdown", "Missings", "PooledArrays", "PrecompileTools", "PrettyTables", "Printf", "Random", "Reexport", "SentinelArrays", "SortingAlgorithms", "Statistics", "TableTraits", "Tables", "Unicode"]
|
||||||
|
git-tree-sha1 = "5fab31e2e01e70ad66e3e24c968c264d1cf166d6"
|
||||||
|
uuid = "a93c6f00-e57d-5684-b7b6-d8193f3e46c0"
|
||||||
|
version = "1.8.2"
|
||||||
|
|
||||||
|
[[deps.DataStructures]]
|
||||||
|
deps = ["OrderedCollections"]
|
||||||
|
git-tree-sha1 = "6fb53a69613a0b2b68a0d12671717d307ab8b24e"
|
||||||
|
uuid = "864edb3b-99cc-5e75-8d2d-829cb0a9cfe8"
|
||||||
|
version = "0.19.5"
|
||||||
|
|
||||||
|
[[deps.DataValueInterfaces]]
|
||||||
|
git-tree-sha1 = "bfc1187b79289637fa0ef6d4436ebdfe6905cbd6"
|
||||||
|
uuid = "e2d170a0-9d28-54be-80f0-106bbe20a464"
|
||||||
|
version = "1.0.0"
|
||||||
|
|
||||||
|
[[deps.Dates]]
|
||||||
|
deps = ["Printf"]
|
||||||
|
uuid = "ade2ca70-3891-5945-98fb-dc099432e06a"
|
||||||
|
version = "1.11.0"
|
||||||
|
|
||||||
|
[[deps.DelimitedFiles]]
|
||||||
|
deps = ["Mmap"]
|
||||||
|
git-tree-sha1 = "9e2f36d3c96a820c678f2f1f1782582fcf685bae"
|
||||||
|
uuid = "8bb1440f-4735-579b-a4ab-409b98df4dab"
|
||||||
|
version = "1.9.1"
|
||||||
|
|
||||||
|
[[deps.Distributed]]
|
||||||
|
deps = ["Random", "Serialization", "Sockets"]
|
||||||
|
uuid = "8ba89e20-285c-5b6f-9357-94700520ee1b"
|
||||||
|
version = "1.11.0"
|
||||||
|
|
||||||
|
[[deps.DocStringExtensions]]
|
||||||
|
git-tree-sha1 = "7442a5dfe1ebb773c29cc2962a8980f47221d76c"
|
||||||
|
uuid = "ffbed154-4ef7-542d-bbb7-c09d3a79fcae"
|
||||||
|
version = "0.9.5"
|
||||||
|
|
||||||
|
[[deps.Downloads]]
|
||||||
|
deps = ["ArgTools", "FileWatching", "LibCURL", "NetworkOptions"]
|
||||||
|
uuid = "f43a241f-c20a-4ad4-852c-f6b1247861c6"
|
||||||
|
version = "1.6.0"
|
||||||
|
|
||||||
|
[[deps.FFTW]]
|
||||||
|
deps = ["AbstractFFTs", "FFTW_jll", "Libdl", "LinearAlgebra", "MKL_jll", "Preferences", "Reexport"]
|
||||||
|
git-tree-sha1 = "97f08406df914023af55ade2f843c39e99c5d969"
|
||||||
|
uuid = "7a1cc6ca-52ef-59f5-83cd-3a7055c09341"
|
||||||
|
version = "1.10.0"
|
||||||
|
|
||||||
|
[[deps.FFTW_jll]]
|
||||||
|
deps = ["Artifacts", "JLLWrappers", "Libdl"]
|
||||||
|
git-tree-sha1 = "6866aec60ef98e3164cd8d6855225684207e9dff"
|
||||||
|
uuid = "f5851436-0d7a-5f13-b9de-f02708fd171a"
|
||||||
|
version = "3.3.12+0"
|
||||||
|
|
||||||
|
[[deps.FilePathsBase]]
|
||||||
|
deps = ["Compat", "Dates"]
|
||||||
|
git-tree-sha1 = "3bab2c5aa25e7840a4b065805c0cdfc01f3068d2"
|
||||||
|
uuid = "48062228-2e41-5def-b9a4-89aafe57970f"
|
||||||
|
version = "0.9.24"
|
||||||
|
weakdeps = ["Mmap", "Test"]
|
||||||
|
|
||||||
|
[deps.FilePathsBase.extensions]
|
||||||
|
FilePathsBaseMmapExt = "Mmap"
|
||||||
|
FilePathsBaseTestExt = "Test"
|
||||||
|
|
||||||
|
[[deps.FileWatching]]
|
||||||
|
uuid = "7b1f6079-737a-58dc-b8bc-7a2ca5c1b5ee"
|
||||||
|
version = "1.11.0"
|
||||||
|
|
||||||
|
[[deps.Future]]
|
||||||
|
deps = ["Random"]
|
||||||
|
uuid = "9fa8497b-333b-5362-9e8d-4d0656e87820"
|
||||||
|
version = "1.11.0"
|
||||||
|
|
||||||
|
[[deps.HypergeometricFunctions]]
|
||||||
|
deps = ["LinearAlgebra", "OpenLibm_jll", "SpecialFunctions"]
|
||||||
|
git-tree-sha1 = "68c173f4f449de5b438ee67ed0c9c748dc31a2ec"
|
||||||
|
uuid = "34004b35-14d8-5ef3-9330-4cdb6864b03a"
|
||||||
|
version = "0.3.28"
|
||||||
|
|
||||||
|
[[deps.InlineStrings]]
|
||||||
|
git-tree-sha1 = "8f3d257792a522b4601c24a577954b0a8cd7334d"
|
||||||
|
uuid = "842dd82b-1e85-43dc-bf29-5d0ee9dffc48"
|
||||||
|
version = "1.4.5"
|
||||||
|
|
||||||
|
[deps.InlineStrings.extensions]
|
||||||
|
ArrowTypesExt = "ArrowTypes"
|
||||||
|
ParsersExt = "Parsers"
|
||||||
|
|
||||||
|
[deps.InlineStrings.weakdeps]
|
||||||
|
ArrowTypes = "31f734f8-188a-4ce0-8406-c8a06bd891cd"
|
||||||
|
Parsers = "69de0a69-1ddd-5017-9359-2bf0b02dc9f0"
|
||||||
|
|
||||||
|
[[deps.IntelOpenMP_jll]]
|
||||||
|
deps = ["Artifacts", "JLLWrappers", "LazyArtifacts", "Libdl"]
|
||||||
|
git-tree-sha1 = "ec1debd61c300961f98064cfb21287613ad7f303"
|
||||||
|
uuid = "1d5cc7b8-4909-519e-a0f8-d0f5ad9712d0"
|
||||||
|
version = "2025.2.0+0"
|
||||||
|
|
||||||
|
[[deps.InteractiveUtils]]
|
||||||
|
deps = ["Markdown"]
|
||||||
|
uuid = "b77e0a4c-d291-57a0-90e8-8db25a27a240"
|
||||||
|
version = "1.11.0"
|
||||||
|
|
||||||
|
[[deps.InvertedIndices]]
|
||||||
|
git-tree-sha1 = "6da3c4316095de0f5ee2ebd875df8721e7e0bdbe"
|
||||||
|
uuid = "41ab1584-1d38-5bbf-9106-f11c6c58b48f"
|
||||||
|
version = "1.3.1"
|
||||||
|
|
||||||
|
[[deps.IrrationalConstants]]
|
||||||
|
git-tree-sha1 = "b2d91fe939cae05960e760110b328288867b5758"
|
||||||
|
uuid = "92d709cd-6900-40b7-9082-c6be49f344b6"
|
||||||
|
version = "0.2.6"
|
||||||
|
|
||||||
|
[[deps.IteratorInterfaceExtensions]]
|
||||||
|
git-tree-sha1 = "a3f24677c21f5bbe9d2a714f95dcd58337fb2856"
|
||||||
|
uuid = "82899510-4779-5014-852e-03e436cf321d"
|
||||||
|
version = "1.0.0"
|
||||||
|
|
||||||
|
[[deps.JLLWrappers]]
|
||||||
|
deps = ["Artifacts", "Preferences"]
|
||||||
|
git-tree-sha1 = "7204148362dafe5fe6a273f855b8ccbe4df8173e"
|
||||||
|
uuid = "692b3bcd-3c85-4b1f-b108-f13ce0eb3210"
|
||||||
|
version = "1.8.0"
|
||||||
|
|
||||||
|
[[deps.JSON]]
|
||||||
|
deps = ["Dates", "Mmap", "Parsers", "Unicode"]
|
||||||
|
git-tree-sha1 = "31e996f0a15c7b280ba9f76636b3ff9e2ae58c9a"
|
||||||
|
uuid = "682c06a0-de6a-54ab-a142-c8b1cf79cde6"
|
||||||
|
version = "0.21.4"
|
||||||
|
|
||||||
|
[[deps.LaTeXStrings]]
|
||||||
|
git-tree-sha1 = "dda21b8cbd6a6c40d9d02a73230f9d70fed6918c"
|
||||||
|
uuid = "b964fa9f-0449-5b57-a5c2-d3ea65f4040f"
|
||||||
|
version = "1.4.0"
|
||||||
|
|
||||||
|
[[deps.LazyArtifacts]]
|
||||||
|
deps = ["Artifacts", "Pkg"]
|
||||||
|
uuid = "4af54fe1-eca0-43a8-85a7-787d91b784e3"
|
||||||
|
version = "1.11.0"
|
||||||
|
|
||||||
|
[[deps.LibCURL]]
|
||||||
|
deps = ["LibCURL_jll", "MozillaCACerts_jll"]
|
||||||
|
uuid = "b27032c2-a3e7-50c8-80cd-2d36dbcbfd21"
|
||||||
|
version = "0.6.4"
|
||||||
|
|
||||||
|
[[deps.LibCURL_jll]]
|
||||||
|
deps = ["Artifacts", "LibSSH2_jll", "Libdl", "MbedTLS_jll", "Zlib_jll", "nghttp2_jll"]
|
||||||
|
uuid = "deac9b47-8bc7-5906-a0fe-35ac56dc84c0"
|
||||||
|
version = "8.6.0+0"
|
||||||
|
|
||||||
|
[[deps.LibGit2]]
|
||||||
|
deps = ["Base64", "LibGit2_jll", "NetworkOptions", "Printf", "SHA"]
|
||||||
|
uuid = "76f85450-5226-5b5a-8eaa-529ad045b433"
|
||||||
|
version = "1.11.0"
|
||||||
|
|
||||||
|
[[deps.LibGit2_jll]]
|
||||||
|
deps = ["Artifacts", "LibSSH2_jll", "Libdl", "MbedTLS_jll"]
|
||||||
|
uuid = "e37daf67-58a4-590a-8e99-b0245dd2ffc5"
|
||||||
|
version = "1.7.2+0"
|
||||||
|
|
||||||
|
[[deps.LibSSH2_jll]]
|
||||||
|
deps = ["Artifacts", "Libdl", "MbedTLS_jll"]
|
||||||
|
uuid = "29816b5a-b9ab-546f-933c-edad1886dfa8"
|
||||||
|
version = "1.11.0+1"
|
||||||
|
|
||||||
|
[[deps.Libdl]]
|
||||||
|
uuid = "8f399da3-3557-5675-b5ff-fb832c97cbdb"
|
||||||
|
version = "1.11.0"
|
||||||
|
|
||||||
|
[[deps.LinearAlgebra]]
|
||||||
|
deps = ["Libdl", "OpenBLAS_jll", "libblastrampoline_jll"]
|
||||||
|
uuid = "37e2e46d-f89d-539d-b4ee-838fcccc9c8e"
|
||||||
|
version = "1.11.0"
|
||||||
|
|
||||||
|
[[deps.LogExpFunctions]]
|
||||||
|
deps = ["DocStringExtensions", "IrrationalConstants", "LinearAlgebra"]
|
||||||
|
git-tree-sha1 = "bba2d9aa057d8f126415de240573e86a8f39d2a1"
|
||||||
|
uuid = "2ab3a3ac-af41-5b50-aa03-7779005ae688"
|
||||||
|
version = "1.0.1"
|
||||||
|
|
||||||
|
[deps.LogExpFunctions.extensions]
|
||||||
|
LogExpFunctionsChainRulesCoreExt = "ChainRulesCore"
|
||||||
|
LogExpFunctionsChangesOfVariablesExt = "ChangesOfVariables"
|
||||||
|
LogExpFunctionsInverseFunctionsExt = "InverseFunctions"
|
||||||
|
|
||||||
|
[deps.LogExpFunctions.weakdeps]
|
||||||
|
ChainRulesCore = "d360d2e6-b24c-11e9-a2a3-2a2ae2dbcce4"
|
||||||
|
ChangesOfVariables = "9e997f8a-9a97-42d5-a9f1-ce6bfc15e2c0"
|
||||||
|
InverseFunctions = "3587e190-3f89-42d0-90ee-14403ec27112"
|
||||||
|
|
||||||
|
[[deps.Logging]]
|
||||||
|
uuid = "56ddb016-857b-54e1-b83d-db4d58db5568"
|
||||||
|
version = "1.11.0"
|
||||||
|
|
||||||
|
[[deps.MKL_jll]]
|
||||||
|
deps = ["Artifacts", "IntelOpenMP_jll", "JLLWrappers", "LazyArtifacts", "Libdl", "oneTBB_jll"]
|
||||||
|
git-tree-sha1 = "282cadc186e7b2ae0eeadbd7a4dffed4196ae2aa"
|
||||||
|
uuid = "856f044c-d86e-5d09-b602-aeab76dc8ba7"
|
||||||
|
version = "2025.2.0+0"
|
||||||
|
|
||||||
|
[[deps.Markdown]]
|
||||||
|
deps = ["Base64"]
|
||||||
|
uuid = "d6f4376e-aef5-505a-96c1-9c027394607a"
|
||||||
|
version = "1.11.0"
|
||||||
|
|
||||||
|
[[deps.MbedTLS_jll]]
|
||||||
|
deps = ["Artifacts", "Libdl"]
|
||||||
|
uuid = "c8ffd9c3-330d-5841-b78e-0817d7145fa1"
|
||||||
|
version = "2.28.6+0"
|
||||||
|
|
||||||
|
[[deps.Missings]]
|
||||||
|
deps = ["DataAPI"]
|
||||||
|
git-tree-sha1 = "ec4f7fbeab05d7747bdf98eb74d130a2a2ed298d"
|
||||||
|
uuid = "e1d29d7a-bbdc-5cf2-9ac0-f12de2c33e28"
|
||||||
|
version = "1.2.0"
|
||||||
|
|
||||||
|
[[deps.Mmap]]
|
||||||
|
uuid = "a63ad114-7e13-5084-954f-fe012c677804"
|
||||||
|
version = "1.11.0"
|
||||||
|
|
||||||
|
[[deps.MozillaCACerts_jll]]
|
||||||
|
uuid = "14a3606d-f60d-562e-9121-12d972cd8159"
|
||||||
|
version = "2023.12.12"
|
||||||
|
|
||||||
|
[[deps.NamedTupleTools]]
|
||||||
|
git-tree-sha1 = "90914795fc59df44120fe3fff6742bb0d7adb1d0"
|
||||||
|
uuid = "d9ec5142-1e00-5aa0-9d6a-321866360f50"
|
||||||
|
version = "0.14.3"
|
||||||
|
|
||||||
|
[[deps.NetworkOptions]]
|
||||||
|
uuid = "ca575930-c2e3-43a9-ace4-1e988b2c1908"
|
||||||
|
version = "1.2.0"
|
||||||
|
|
||||||
|
[[deps.OpenBLAS_jll]]
|
||||||
|
deps = ["Artifacts", "CompilerSupportLibraries_jll", "Libdl"]
|
||||||
|
uuid = "4536629a-c528-5b80-bd46-f80d51c5b363"
|
||||||
|
version = "0.3.27+1"
|
||||||
|
|
||||||
|
[[deps.OpenLibm_jll]]
|
||||||
|
deps = ["Artifacts", "Libdl"]
|
||||||
|
uuid = "05823500-19ac-5b8b-9628-191a04bc5112"
|
||||||
|
version = "0.8.1+2"
|
||||||
|
|
||||||
|
[[deps.OpenSpecFun_jll]]
|
||||||
|
deps = ["Artifacts", "CompilerSupportLibraries_jll", "JLLWrappers", "Libdl"]
|
||||||
|
git-tree-sha1 = "1346c9208249809840c91b26703912dff463d335"
|
||||||
|
uuid = "efe28fd5-8261-553b-a9e1-b2916fc3738e"
|
||||||
|
version = "0.5.6+0"
|
||||||
|
|
||||||
|
[[deps.OrderedCollections]]
|
||||||
|
git-tree-sha1 = "94ba93778373a53bfd5a0caaf7d809c445292ff4"
|
||||||
|
uuid = "bac558e1-5e72-5ebc-8fee-abe8a469f55d"
|
||||||
|
version = "1.8.2"
|
||||||
|
|
||||||
|
[[deps.Parameters]]
|
||||||
|
deps = ["OrderedCollections", "UnPack"]
|
||||||
|
git-tree-sha1 = "34c0e9ad262e5f7fc75b10a9952ca7692cfc5fbe"
|
||||||
|
uuid = "d96e819e-fc66-5662-9728-84c9c7592b0a"
|
||||||
|
version = "0.12.3"
|
||||||
|
|
||||||
|
[[deps.Parsers]]
|
||||||
|
deps = ["Dates", "PrecompileTools", "UUIDs"]
|
||||||
|
git-tree-sha1 = "468dbe2b510c876dc091b2c74ed52c7c34f48b9b"
|
||||||
|
uuid = "69de0a69-1ddd-5017-9359-2bf0b02dc9f0"
|
||||||
|
version = "2.8.5"
|
||||||
|
|
||||||
|
[[deps.Pkg]]
|
||||||
|
deps = ["Artifacts", "Dates", "Downloads", "FileWatching", "LibGit2", "Libdl", "Logging", "Markdown", "Printf", "Random", "SHA", "TOML", "Tar", "UUIDs", "p7zip_jll"]
|
||||||
|
uuid = "44cfe95a-1eb2-52ea-b672-e2afdf69b78f"
|
||||||
|
version = "1.11.0"
|
||||||
|
weakdeps = ["REPL"]
|
||||||
|
|
||||||
|
[deps.Pkg.extensions]
|
||||||
|
REPLExt = "REPL"
|
||||||
|
|
||||||
|
[[deps.PooledArrays]]
|
||||||
|
deps = ["DataAPI", "Future"]
|
||||||
|
git-tree-sha1 = "36d8b4b899628fb92c2749eb488d884a926614d3"
|
||||||
|
uuid = "2dfb63ee-cc39-5dd5-95bd-886bf059d720"
|
||||||
|
version = "1.4.3"
|
||||||
|
|
||||||
|
[[deps.PrecompileTools]]
|
||||||
|
deps = ["Preferences"]
|
||||||
|
git-tree-sha1 = "5aa36f7049a63a1528fe8f7c3f2113413ffd4e1f"
|
||||||
|
uuid = "aea7be01-6a6a-4083-8856-8a6e6704d82a"
|
||||||
|
version = "1.2.1"
|
||||||
|
|
||||||
|
[[deps.Preferences]]
|
||||||
|
deps = ["TOML"]
|
||||||
|
git-tree-sha1 = "8b770b60760d4451834fe79dd483e318eee709c4"
|
||||||
|
uuid = "21216c6a-2e73-6563-6e65-726566657250"
|
||||||
|
version = "1.5.2"
|
||||||
|
|
||||||
|
[[deps.PrettyTables]]
|
||||||
|
deps = ["Crayons", "LaTeXStrings", "Markdown", "PrecompileTools", "Printf", "REPL", "Reexport", "StringManipulation", "Tables"]
|
||||||
|
git-tree-sha1 = "624de6279ab7d94fc9f672f0068107eb6619732c"
|
||||||
|
uuid = "08abe8d2-0d0c-5749-adfa-8a2ac140af0d"
|
||||||
|
version = "3.3.2"
|
||||||
|
|
||||||
|
[deps.PrettyTables.extensions]
|
||||||
|
PrettyTablesTypstryExt = "Typstry"
|
||||||
|
|
||||||
|
[deps.PrettyTables.weakdeps]
|
||||||
|
Typstry = "f0ed7684-a786-439e-b1e3-3b82803b501e"
|
||||||
|
|
||||||
|
[[deps.Printf]]
|
||||||
|
deps = ["Unicode"]
|
||||||
|
uuid = "de0858da-6303-5e67-8744-51eddeeeb8d7"
|
||||||
|
version = "1.11.0"
|
||||||
|
|
||||||
|
[[deps.ProgressMeter]]
|
||||||
|
deps = ["Distributed", "Printf"]
|
||||||
|
git-tree-sha1 = "fbb92c6c56b34e1a2c4c36058f68f332bec840e7"
|
||||||
|
uuid = "92933f4c-e287-5a05-a399-4b506db050ca"
|
||||||
|
version = "1.11.0"
|
||||||
|
|
||||||
|
[[deps.PtrArrays]]
|
||||||
|
git-tree-sha1 = "4fbbafbc6251b883f4d2705356f3641f3652a7fe"
|
||||||
|
uuid = "43287f4e-b6f4-7ad1-bb20-aadabca52c3d"
|
||||||
|
version = "1.4.0"
|
||||||
|
|
||||||
|
[[deps.REPL]]
|
||||||
|
deps = ["InteractiveUtils", "Markdown", "Sockets", "StyledStrings", "Unicode"]
|
||||||
|
uuid = "3fa0cd96-eef1-5676-8a61-b3b8758bbffb"
|
||||||
|
version = "1.11.0"
|
||||||
|
|
||||||
|
[[deps.Random]]
|
||||||
|
deps = ["SHA"]
|
||||||
|
uuid = "9a3f8284-a2c9-5f02-9a11-845980a1fd5c"
|
||||||
|
version = "1.11.0"
|
||||||
|
|
||||||
|
[[deps.Reexport]]
|
||||||
|
git-tree-sha1 = "45e428421666073eab6f2da5c9d310d99bb12f9b"
|
||||||
|
uuid = "189a3867-3050-52da-a836-e630ba90ab69"
|
||||||
|
version = "1.2.2"
|
||||||
|
|
||||||
|
[[deps.Requires]]
|
||||||
|
deps = ["UUIDs"]
|
||||||
|
git-tree-sha1 = "62389eeff14780bfe55195b7204c0d8738436d64"
|
||||||
|
uuid = "ae029012-a4dd-5104-9daa-d747884805df"
|
||||||
|
version = "1.3.1"
|
||||||
|
|
||||||
|
[[deps.Rmath]]
|
||||||
|
deps = ["Random", "Rmath_jll"]
|
||||||
|
git-tree-sha1 = "5b3d50eb374cea306873b371d3f8d3915a018f0b"
|
||||||
|
uuid = "79098fc4-a85e-5d69-aa6a-4863f24498fa"
|
||||||
|
version = "0.9.0"
|
||||||
|
|
||||||
|
[[deps.Rmath_jll]]
|
||||||
|
deps = ["Artifacts", "JLLWrappers", "Libdl"]
|
||||||
|
git-tree-sha1 = "58cdd8fb2201a6267e1db87ff148dd6c1dbd8ad8"
|
||||||
|
uuid = "f50d1b31-88e8-58de-be2c-1cc44531875f"
|
||||||
|
version = "0.5.1+0"
|
||||||
|
|
||||||
|
[[deps.SHA]]
|
||||||
|
uuid = "ea8e919c-243c-51af-8825-aaa63cd721ce"
|
||||||
|
version = "0.7.0"
|
||||||
|
|
||||||
|
[[deps.SentinelArrays]]
|
||||||
|
deps = ["Dates", "Random"]
|
||||||
|
git-tree-sha1 = "084c47c7c5ce5cfecefa0a98dff69eb3646b5a80"
|
||||||
|
uuid = "91c51154-3ec4-41a3-a24f-3f23e20d615c"
|
||||||
|
version = "1.4.10"
|
||||||
|
|
||||||
|
[[deps.Serialization]]
|
||||||
|
uuid = "9e88b42a-f829-5b0c-bbe9-9e923198166b"
|
||||||
|
version = "1.11.0"
|
||||||
|
|
||||||
|
[[deps.Sockets]]
|
||||||
|
uuid = "6462fe0b-24de-5631-8697-dd941f90decc"
|
||||||
|
version = "1.11.0"
|
||||||
|
|
||||||
|
[[deps.SortingAlgorithms]]
|
||||||
|
deps = ["DataStructures"]
|
||||||
|
git-tree-sha1 = "64d974c2e6fdf07f8155b5b2ca2ffa9069b608d9"
|
||||||
|
uuid = "a2af1166-a08f-5f64-846c-94a0d3cef48c"
|
||||||
|
version = "1.2.2"
|
||||||
|
|
||||||
|
[[deps.SparseArrays]]
|
||||||
|
deps = ["Libdl", "LinearAlgebra", "Random", "Serialization", "SuiteSparse_jll"]
|
||||||
|
uuid = "2f01184e-e22b-5df5-ae63-d93ebab69eaf"
|
||||||
|
version = "1.11.0"
|
||||||
|
|
||||||
|
[[deps.SpecialFunctions]]
|
||||||
|
deps = ["IrrationalConstants", "LogExpFunctions", "OpenLibm_jll", "OpenSpecFun_jll"]
|
||||||
|
git-tree-sha1 = "6547cbdd8ce32efba0d21c5a40fa96d1a3548f9f"
|
||||||
|
uuid = "276daf66-3868-5448-9aa4-cd146d93841b"
|
||||||
|
version = "2.8.0"
|
||||||
|
|
||||||
|
[deps.SpecialFunctions.extensions]
|
||||||
|
SpecialFunctionsChainRulesCoreExt = "ChainRulesCore"
|
||||||
|
|
||||||
|
[deps.SpecialFunctions.weakdeps]
|
||||||
|
ChainRulesCore = "d360d2e6-b24c-11e9-a2a3-2a2ae2dbcce4"
|
||||||
|
|
||||||
|
[[deps.StanBase]]
|
||||||
|
deps = ["CSV", "DataFrames", "DelimitedFiles", "Distributed", "DocStringExtensions", "JSON", "NamedTupleTools", "OrderedCollections", "Parameters", "Random", "Unicode"]
|
||||||
|
git-tree-sha1 = "fac478e96c6a32f08af58f709b07f5ee722b3140"
|
||||||
|
uuid = "d0ee94f6-a23d-54aa-bbe9-7f572d6da7f5"
|
||||||
|
version = "4.12.4"
|
||||||
|
|
||||||
|
[[deps.StanSample]]
|
||||||
|
deps = ["CSV", "CompatHelperLocal", "DataFrames", "DelimitedFiles", "DocStringExtensions", "JSON", "LazyArtifacts", "NamedTupleTools", "OrderedCollections", "Parameters", "Random", "Reexport", "Requires", "Serialization", "StanBase", "TableOperations", "Tables", "Unicode"]
|
||||||
|
git-tree-sha1 = "0e6f7c729879a416fd7b27cfd81b2b1f6e08586e"
|
||||||
|
uuid = "c1514b29-d3a0-5178-b312-660c88baa699"
|
||||||
|
version = "7.10.3"
|
||||||
|
|
||||||
|
[deps.StanSample.extensions]
|
||||||
|
AxisKeysExt = "AxisKeys"
|
||||||
|
InferenceObjectsExt = "InferenceObjects"
|
||||||
|
MCMCChainsExt = "MCMCChains"
|
||||||
|
MonteCarloMeasurementsExt = "MonteCarloMeasurements"
|
||||||
|
|
||||||
|
[deps.StanSample.weakdeps]
|
||||||
|
AxisKeys = "94b1ba4f-4ee9-5380-92f1-94cde586c3c5"
|
||||||
|
InferenceObjects = "b5cf5a8d-e756-4ee3-b014-01d49d192c00"
|
||||||
|
MCMCChains = "c7f686f2-ff18-58e9-bc7b-31028e88f75d"
|
||||||
|
MonteCarloMeasurements = "0987c9cc-fe09-11e8-30f0-b96dd679fdca"
|
||||||
|
|
||||||
|
[[deps.Statistics]]
|
||||||
|
deps = ["LinearAlgebra"]
|
||||||
|
git-tree-sha1 = "ae3bb1eb3bba077cd276bc5cfc337cc65c3075c0"
|
||||||
|
uuid = "10745b16-79ce-11e8-11f9-7d13ad32a3b2"
|
||||||
|
version = "1.11.1"
|
||||||
|
weakdeps = ["SparseArrays"]
|
||||||
|
|
||||||
|
[deps.Statistics.extensions]
|
||||||
|
SparseArraysExt = ["SparseArrays"]
|
||||||
|
|
||||||
|
[[deps.StatsAPI]]
|
||||||
|
deps = ["LinearAlgebra"]
|
||||||
|
git-tree-sha1 = "178ed29fd5b2a2cfc3bd31c13375ae925623ff36"
|
||||||
|
uuid = "82ae8749-77ed-4fe6-ae5f-f523153014b0"
|
||||||
|
version = "1.8.0"
|
||||||
|
|
||||||
|
[[deps.StatsBase]]
|
||||||
|
deps = ["AliasTables", "DataAPI", "DataStructures", "IrrationalConstants", "LinearAlgebra", "LogExpFunctions", "Missings", "Printf", "Random", "SortingAlgorithms", "SparseArrays", "Statistics", "StatsAPI"]
|
||||||
|
git-tree-sha1 = "c6f18e5a52a176a383f6f6c635e0f81feed1d6d4"
|
||||||
|
uuid = "2913bbd2-ae8a-5f71-8c99-4fb6c76f3a91"
|
||||||
|
version = "0.34.11"
|
||||||
|
|
||||||
|
[[deps.StatsFuns]]
|
||||||
|
deps = ["HypergeometricFunctions", "IrrationalConstants", "LogExpFunctions", "Reexport", "Rmath", "SpecialFunctions"]
|
||||||
|
git-tree-sha1 = "3f4e1d24289cd974e089c617b1472311a2b1feab"
|
||||||
|
uuid = "4c63d2b9-4356-54db-8cca-17b64c39e42c"
|
||||||
|
version = "2.1.0"
|
||||||
|
|
||||||
|
[deps.StatsFuns.extensions]
|
||||||
|
StatsFunsChainRulesCoreExt = "ChainRulesCore"
|
||||||
|
StatsFunsInverseFunctionsExt = "InverseFunctions"
|
||||||
|
|
||||||
|
[deps.StatsFuns.weakdeps]
|
||||||
|
ChainRulesCore = "d360d2e6-b24c-11e9-a2a3-2a2ae2dbcce4"
|
||||||
|
InverseFunctions = "3587e190-3f89-42d0-90ee-14403ec27112"
|
||||||
|
|
||||||
|
[[deps.StringManipulation]]
|
||||||
|
deps = ["PrecompileTools"]
|
||||||
|
git-tree-sha1 = "d05693d339e37d6ab134c5ab53c29fce5ee5d7d5"
|
||||||
|
uuid = "892a3eda-7b42-436c-8928-eab12a02cf0e"
|
||||||
|
version = "0.4.4"
|
||||||
|
|
||||||
|
[[deps.StyledStrings]]
|
||||||
|
uuid = "f489334b-da3d-4c2e-b8f0-e476e12c162b"
|
||||||
|
version = "1.11.0"
|
||||||
|
|
||||||
|
[[deps.SuiteSparse_jll]]
|
||||||
|
deps = ["Artifacts", "Libdl", "libblastrampoline_jll"]
|
||||||
|
uuid = "bea87d4a-7f5b-5778-9afe-8cc45184846c"
|
||||||
|
version = "7.7.0+0"
|
||||||
|
|
||||||
|
[[deps.TOML]]
|
||||||
|
deps = ["Dates"]
|
||||||
|
uuid = "fa267f1f-6049-4f14-aa54-33bafae1ed76"
|
||||||
|
version = "1.0.3"
|
||||||
|
|
||||||
|
[[deps.TableOperations]]
|
||||||
|
deps = ["SentinelArrays", "Tables", "Test"]
|
||||||
|
git-tree-sha1 = "e383c87cf2a1dc41fa30c093b2a19877c83e1bc1"
|
||||||
|
uuid = "ab02a1b2-a7df-11e8-156e-fb1833f50b87"
|
||||||
|
version = "1.2.0"
|
||||||
|
|
||||||
|
[[deps.TableTraits]]
|
||||||
|
deps = ["IteratorInterfaceExtensions"]
|
||||||
|
git-tree-sha1 = "c06b2f539df1c6efa794486abfb6ed2022561a39"
|
||||||
|
uuid = "3783bdb8-4a98-5b6b-af9a-565f29a5fe9c"
|
||||||
|
version = "1.0.1"
|
||||||
|
|
||||||
|
[[deps.Tables]]
|
||||||
|
deps = ["DataAPI", "DataValueInterfaces", "IteratorInterfaceExtensions", "OrderedCollections", "TableTraits"]
|
||||||
|
git-tree-sha1 = "f2c1efbc8f3a609aadf318094f8fc5204bdaf344"
|
||||||
|
uuid = "bd369af6-aec1-5ad0-b16a-f7cc5008161c"
|
||||||
|
version = "1.12.1"
|
||||||
|
|
||||||
|
[[deps.Tar]]
|
||||||
|
deps = ["ArgTools", "SHA"]
|
||||||
|
uuid = "a4e569a6-e804-4fa4-b0f3-eef7a1d5b13e"
|
||||||
|
version = "1.10.0"
|
||||||
|
|
||||||
|
[[deps.Test]]
|
||||||
|
deps = ["InteractiveUtils", "Logging", "Random", "Serialization"]
|
||||||
|
uuid = "8dfed614-e22c-5e08-85e1-65c5234f0b40"
|
||||||
|
version = "1.11.0"
|
||||||
|
|
||||||
|
[[deps.TranscodingStreams]]
|
||||||
|
git-tree-sha1 = "0c45878dcfdcfa8480052b6ab162cdd138781742"
|
||||||
|
uuid = "3bb67fe8-82b1-5028-8e26-92a6c54297fa"
|
||||||
|
version = "0.11.3"
|
||||||
|
|
||||||
|
[[deps.UUIDs]]
|
||||||
|
deps = ["Random", "SHA"]
|
||||||
|
uuid = "cf7118a7-6976-5b1a-9a39-7adc72f591a4"
|
||||||
|
version = "1.11.0"
|
||||||
|
|
||||||
|
[[deps.UnPack]]
|
||||||
|
git-tree-sha1 = "387c1f73762231e86e0c9c5443ce3b4a0a9a0c2b"
|
||||||
|
uuid = "3a884ed6-31ef-47d7-9d2a-63182c4928ed"
|
||||||
|
version = "1.0.2"
|
||||||
|
|
||||||
|
[[deps.Unicode]]
|
||||||
|
uuid = "4ec0a83e-493e-50e2-b9ac-8f72acf5a8f5"
|
||||||
|
version = "1.11.0"
|
||||||
|
|
||||||
|
[[deps.WeakRefStrings]]
|
||||||
|
deps = ["DataAPI", "InlineStrings", "Parsers"]
|
||||||
|
git-tree-sha1 = "0716e01c3b40413de5dedbc9c5c69f27cddfddfc"
|
||||||
|
uuid = "ea10d353-3f73-51f8-a26c-33c1cb351aa5"
|
||||||
|
version = "1.4.3"
|
||||||
|
|
||||||
|
[[deps.WorkerUtilities]]
|
||||||
|
git-tree-sha1 = "cd1659ba0d57b71a464a29e64dbc67cfe83d54e7"
|
||||||
|
uuid = "76eceee3-57b5-4d4a-8e66-0e911cebbf60"
|
||||||
|
version = "1.6.1"
|
||||||
|
|
||||||
|
[[deps.Zlib_jll]]
|
||||||
|
deps = ["Libdl"]
|
||||||
|
uuid = "83775a58-1f1d-513f-b197-d71354ab007a"
|
||||||
|
version = "1.2.13+1"
|
||||||
|
|
||||||
|
[[deps.libblastrampoline_jll]]
|
||||||
|
deps = ["Artifacts", "Libdl"]
|
||||||
|
uuid = "8e850b90-86db-534c-a0d3-1478176c7d93"
|
||||||
|
version = "5.11.0+0"
|
||||||
|
|
||||||
|
[[deps.nghttp2_jll]]
|
||||||
|
deps = ["Artifacts", "Libdl"]
|
||||||
|
uuid = "8e850ede-7688-5339-a07c-302acd2aaf8d"
|
||||||
|
version = "1.59.0+0"
|
||||||
|
|
||||||
|
[[deps.oneTBB_jll]]
|
||||||
|
deps = ["Artifacts", "JLLWrappers", "LazyArtifacts", "Libdl"]
|
||||||
|
git-tree-sha1 = "da8c1f6eee04831f14edcfa5dae611d309807e57"
|
||||||
|
uuid = "1317d2d5-d96f-522e-a858-c73665f53c3e"
|
||||||
|
version = "2022.3.0+0"
|
||||||
|
|
||||||
|
[[deps.p7zip_jll]]
|
||||||
|
deps = ["Artifacts", "Libdl"]
|
||||||
|
uuid = "3f19e933-33d8-53b3-aaab-bd5110c3b7a0"
|
||||||
|
version = "17.4.0+2"
|
||||||
@@ -0,0 +1,23 @@
|
|||||||
|
# Public repository policy
|
||||||
|
|
||||||
|
This repository is a public release surface. Every reachable commit, branch,
|
||||||
|
tag, file, attachment, issue, pull request, and release note must be safe for
|
||||||
|
unrestricted public access.
|
||||||
|
|
||||||
|
Allowed material is limited to publication-ready software, data, metadata,
|
||||||
|
documentation, figures, reproducibility outputs, and release records.
|
||||||
|
|
||||||
|
Do not place private correspondence, assessment records, decision logs,
|
||||||
|
working notes, draft response material, credentials, local paths, or other
|
||||||
|
non-public project context in this repository.
|
||||||
|
|
||||||
|
Before every push, run `bash scripts/check_public_content.sh`. Install the
|
||||||
|
pre-push guard in every clone with `bash scripts/install_public_guard.sh`.
|
||||||
|
|
||||||
|
If non-public material enters a public ref, stop publication work immediately.
|
||||||
|
A later deletion does not make the prior object private. Replace the affected
|
||||||
|
public history or repository with a clean snapshot, then verify that the old
|
||||||
|
object URLs no longer resolve before publishing again.
|
||||||
|
|
||||||
|
Branch names, tags, commit messages, release titles, release notes, issue
|
||||||
|
titles, and pull-request titles follow the same rule as file contents.
|
||||||
@@ -0,0 +1,10 @@
|
|||||||
|
[deps]
|
||||||
|
CSV = "336ed68f-0bac-5ca0-87d4-7b16caf5d00b"
|
||||||
|
CategoricalArrays = "324d7699-5711-5eae-9e2f-1d82baa6b597"
|
||||||
|
DataFrames = "a93c6f00-e57d-5684-b7b6-d8193f3e46c0"
|
||||||
|
FFTW = "7a1cc6ca-52ef-59f5-83cd-3a7055c09341"
|
||||||
|
JSON = "682c06a0-de6a-54ab-a142-c8b1cf79cde6"
|
||||||
|
ProgressMeter = "92933f4c-e287-5a05-a399-4b506db050ca"
|
||||||
|
StanSample = "c1514b29-d3a0-5178-b312-660c88baa699"
|
||||||
|
StatsBase = "2913bbd2-ae8a-5f71-8c99-4fb6c76f3a91"
|
||||||
|
StatsFuns = "4c63d2b9-4356-54db-8cca-17b64c39e42c"
|
||||||
@@ -0,0 +1,156 @@
|
|||||||
|
# party2d
|
||||||
|
|
||||||
|
Code and processed model inputs for generating two-dimensional party-position estimates from text and expert data. The model combines manifesto and media text indicators with expert survey placements in a Bayesian dynamic item-response framework.
|
||||||
|
|
||||||
|
## Repository contents
|
||||||
|
|
||||||
|
- `data-setup/` — source download, source-file checks, and rebuild workflow for raw files that cannot be redistributed here.
|
||||||
|
- `src/julia/` — Stan data preparation, model fitting, post-estimation, enrichment, and validation.
|
||||||
|
- `models/` — Stan model specification.
|
||||||
|
- `data/` — processed party-level inputs used by the Julia/Stan model.
|
||||||
|
- `metadata/` — data dictionary and source-support documentation.
|
||||||
|
- `diagnostics/` — repository diagnostics report regenerated after model estimation.
|
||||||
|
- `data/releases/` — release-ready data files, checksums, and diagnostics report.
|
||||||
|
|
||||||
|
Processed inputs needed by the model are included in `data/` so the estimation step can be reproduced from the model-ready data.
|
||||||
|
|
||||||
|
## Release assets
|
||||||
|
|
||||||
|
The public `v0` release is available from <https://git.seimel.app/armin/party2d/releases> and contains:
|
||||||
|
|
||||||
|
- `party_2d_election_year_panel_v0.zip` — primary election-year party-position panel.
|
||||||
|
- `party_2d_annual_model_output_v0.csv.xz` — secondary annual model output (XZ-compressed CSV).
|
||||||
|
- `party_2d_diagnostics_report_v0.pdf` — script-generated diagnostics report.
|
||||||
|
- `SHA256SUMS` — checksums for the release assets.
|
||||||
|
|
||||||
|
## Source data and redistribution
|
||||||
|
|
||||||
|
The public repository does **not** contain the original raw/source files. Several inputs are third-party datasets with their own terms of use, and the Morgan historical file is a local OCR/transcription source. Instead, this repository provides:
|
||||||
|
|
||||||
|
1. committed model-ready inputs in `data/`, sufficient for fitting the Julia/Stan model; and
|
||||||
|
2. `data-setup/`, an R/Shell workflow that downloads script-accessible sources, checks locally supplied raw sources, and rebuilds comparable model-ready inputs locally.
|
||||||
|
|
||||||
|
`data-setup/` downloads script-accessible sources and checks locally supplied sources for:
|
||||||
|
|
||||||
|
- PolDem
|
||||||
|
- PartyFacts crosswalk
|
||||||
|
- CHES family files
|
||||||
|
- POPPA from Harvard Dataverse
|
||||||
|
- Global Party Survey 2019 from Harvard Dataverse
|
||||||
|
- V-Party through the provider's download form
|
||||||
|
|
||||||
|
Two inputs require user-provided access/material:
|
||||||
|
|
||||||
|
- **Manifesto Project**: users must obtain the source data through their own Manifesto Project access.
|
||||||
|
- **Morgan historical expert data**: `morgan_positions_raw.csv` is not publicly downloadable; it can be provided on request and should be placed locally under `_local/raw/morgan/`.
|
||||||
|
|
||||||
|
The setup workflow never overwrites committed files in `data/`. See `data-setup/README.md` for exact commands and source details.
|
||||||
|
|
||||||
|
Run the full source-data setup workflow with:
|
||||||
|
|
||||||
|
```bash
|
||||||
|
bash data-setup/run_data_setup.sh
|
||||||
|
```
|
||||||
|
|
||||||
|
This downloads script-accessible source files, checks required local files, rebuilds model-ready inputs locally, and writes a comparison report. Manifesto Project requires your own provider access, and the Morgan OCR/transcription file can be provided on request.
|
||||||
|
|
||||||
|
## Running the pipeline
|
||||||
|
|
||||||
|
Run the full workflow with:
|
||||||
|
|
||||||
|
```bash
|
||||||
|
bash run_estimation.sh full
|
||||||
|
```
|
||||||
|
|
||||||
|
This checks that model-ready inputs are present, then executes the Julia/Stan workflow scripts:
|
||||||
|
|
||||||
|
```bash
|
||||||
|
bash scripts/01_prepare_data.sh # checks model-ready inputs; does not rebuild raw data
|
||||||
|
bash scripts/02_fit_model.sh
|
||||||
|
bash scripts/03_extract_estimates.sh
|
||||||
|
bash scripts/04_enrich_estimates.sh
|
||||||
|
bash scripts/05_validate_estimates.sh
|
||||||
|
```
|
||||||
|
|
||||||
|
The numbered scripts can also be run manually in that order.
|
||||||
|
|
||||||
|
The Bayesian model is computationally expensive. The production run used 4 cores on an AMD Ryzen 9 7945HX and took 60,372 seconds, approximately 16 hours 46 minutes.
|
||||||
|
|
||||||
|
If model output is already available, rebuild estimates without refitting Stan:
|
||||||
|
|
||||||
|
```bash
|
||||||
|
bash run_estimation.sh reuse
|
||||||
|
```
|
||||||
|
|
||||||
|
`reuse` verifies the model-ready inputs, then reruns post-estimation, enrichment, and validation while skipping the Stan fitting step.
|
||||||
|
|
||||||
|
To check the local setup without fitting the model, run:
|
||||||
|
|
||||||
|
```bash
|
||||||
|
bash run_estimation.sh dry-run
|
||||||
|
```
|
||||||
|
|
||||||
|
After estimation and validation have been run, regenerate the diagnostics report with:
|
||||||
|
|
||||||
|
```bash
|
||||||
|
Rscript diagnostics/generate_diagnostics.R
|
||||||
|
```
|
||||||
|
|
||||||
|
The report is written to `diagnostics/generated/diagnostics_report.pdf` and copied to `data/releases/party_2d_diagnostics_report_v0.pdf`.
|
||||||
|
|
||||||
|
## Data inputs
|
||||||
|
|
||||||
|
The model-ready inputs are included under `data/`:
|
||||||
|
|
||||||
|
- `text_data.csv`
|
||||||
|
- `expert.csv`
|
||||||
|
- `lr_data.csv`
|
||||||
|
- `union_mapping.csv`
|
||||||
|
- `party_families.csv`
|
||||||
|
|
||||||
|
Original raw source files are not redistributed. Rebuilding inputs from raw files is separate from the normal estimation workflow and never replaces committed `data/` inputs automatically.
|
||||||
|
|
||||||
|
## Output dimensions
|
||||||
|
|
||||||
|
The two position dimensions are scaled from 0 to 1:
|
||||||
|
|
||||||
|
- Economic left-right: economic left to economic right.
|
||||||
|
- Cultural cosmopolitan--traditionalist: cosmopolitan to traditionalist.
|
||||||
|
|
||||||
|
Column definitions are in `metadata/data_dictionary.csv`.
|
||||||
|
|
||||||
|
## Release files
|
||||||
|
|
||||||
|
The public release assets are:
|
||||||
|
|
||||||
|
- `party_2d_election_year_panel_v0.zip`
|
||||||
|
- `party_2d_annual_model_output_v0.csv.xz`
|
||||||
|
- `party_2d_diagnostics_report_v0.pdf`
|
||||||
|
- `SHA256SUMS`
|
||||||
|
|
||||||
|
`SHA256SUMS` hashes the release assets listed above.
|
||||||
|
|
||||||
|
## Public-content safeguard
|
||||||
|
|
||||||
|
This repository is intended only for publication-ready material. Before pushing,
|
||||||
|
run:
|
||||||
|
|
||||||
|
```bash
|
||||||
|
bash scripts/check_public_content.sh
|
||||||
|
```
|
||||||
|
|
||||||
|
To install the same audit as a local pre-push hook, run:
|
||||||
|
|
||||||
|
```bash
|
||||||
|
bash scripts/install_public_guard.sh
|
||||||
|
```
|
||||||
|
|
||||||
|
The audit checks the working public history and tracked files for non-public
|
||||||
|
material. If it fails, do not push until the issue is resolved.
|
||||||
|
|
||||||
|
## License
|
||||||
|
|
||||||
|
This release uses the Creative Commons Attribution 4.0 International License
|
||||||
|
(CC BY 4.0). The license applies to release materials created for this
|
||||||
|
repository; third-party source data remain governed by their own terms. See
|
||||||
|
`LICENSE`.
|
||||||
@@ -0,0 +1,16 @@
|
|||||||
|
#!/usr/bin/env Rscript
|
||||||
|
|
||||||
|
repo_root <- normalizePath(getwd(), mustWork = TRUE)
|
||||||
|
lib <- Sys.getenv("R_LIBS_USER", file.path(repo_root, "_local", "R", "library"))
|
||||||
|
dir.create(lib, recursive = TRUE, showWarnings = FALSE)
|
||||||
|
.libPaths(c(lib, .libPaths()))
|
||||||
|
|
||||||
|
required <- c("tidyverse", "countrycode", "haven", "foreign", "jsonlite")
|
||||||
|
missing <- required[!vapply(required, requireNamespace, quietly = TRUE, FUN.VALUE = logical(1))]
|
||||||
|
|
||||||
|
if (length(missing) > 0) {
|
||||||
|
message("Installing missing R packages into ", lib, ": ", paste(missing, collapse = ", "))
|
||||||
|
install.packages(missing, repos = "https://cloud.r-project.org", lib = lib)
|
||||||
|
} else {
|
||||||
|
message("R data-setup dependencies already available")
|
||||||
|
}
|
||||||
@@ -0,0 +1,203 @@
|
|||||||
|
#!/usr/bin/env Rscript
|
||||||
|
|
||||||
|
args <- commandArgs(trailingOnly = FALSE)
|
||||||
|
file_arg <- args[grepl("^--file=", args)][1]
|
||||||
|
if (!is.na(file_arg)) {
|
||||||
|
script_path <- sub("^--file=", "", file_arg)
|
||||||
|
repo_root <- normalizePath(file.path(dirname(script_path), "..", ".."), mustWork = FALSE)
|
||||||
|
} else {
|
||||||
|
repo_root <- normalizePath(getwd(), mustWork = TRUE)
|
||||||
|
}
|
||||||
|
if (!dir.exists(file.path(repo_root, "data-setup"))) repo_root <- normalizePath(getwd(), mustWork = TRUE)
|
||||||
|
|
||||||
|
raw_dir <- normalizePath(Sys.getenv("PARTY2D_RAW_DATA_DIR", file.path(repo_root, "_local", "raw")), mustWork = FALSE)
|
||||||
|
report_dir <- normalizePath(Sys.getenv("PARTY2D_REPORT_DIR", file.path(repo_root, "_local", "reports")), mustWork = FALSE)
|
||||||
|
dir.create(raw_dir, recursive = TRUE, showWarnings = FALSE)
|
||||||
|
dir.create(report_dir, recursive = TRUE, showWarnings = FALSE)
|
||||||
|
|
||||||
|
ua <- "party2d-data-setup/1.0 (+https://git.seimel.app/armin/party2d)"
|
||||||
|
|
||||||
|
download_file <- function(url, dest, overwrite = FALSE, headers = character()) {
|
||||||
|
dir.create(dirname(dest), recursive = TRUE, showWarnings = FALSE)
|
||||||
|
if (file.exists(dest) && file.info(dest)$size > 0 && !overwrite) {
|
||||||
|
message("OK existing: ", dest)
|
||||||
|
return(TRUE)
|
||||||
|
}
|
||||||
|
tmp <- paste0(dest, ".tmp")
|
||||||
|
if (file.exists(tmp)) unlink(tmp)
|
||||||
|
message("Downloading ", url, " -> ", dest)
|
||||||
|
ok <- tryCatch({
|
||||||
|
utils::download.file(
|
||||||
|
url,
|
||||||
|
tmp,
|
||||||
|
mode = "wb",
|
||||||
|
quiet = TRUE,
|
||||||
|
method = "libcurl",
|
||||||
|
headers = c("User-Agent" = ua, headers)
|
||||||
|
)
|
||||||
|
file.rename(tmp, dest)
|
||||||
|
}, error = function(e) {
|
||||||
|
message("FAILED ", url, ": ", conditionMessage(e))
|
||||||
|
FALSE
|
||||||
|
})
|
||||||
|
if (!ok && file.exists(tmp)) unlink(tmp)
|
||||||
|
isTRUE(ok)
|
||||||
|
}
|
||||||
|
|
||||||
|
download_dataverse <- function(doi, filename, dest, directory = NA_character_) {
|
||||||
|
if (!requireNamespace("jsonlite", quietly = TRUE)) {
|
||||||
|
message("FAILED Dataverse lookup: install R package jsonlite")
|
||||||
|
return(FALSE)
|
||||||
|
}
|
||||||
|
api <- paste0("https://dataverse.harvard.edu/api/datasets/:persistentId/?persistentId=", utils::URLencode(doi, reserved = TRUE))
|
||||||
|
fid <- tryCatch({
|
||||||
|
meta <- jsonlite::fromJSON(api, simplifyVector = FALSE)
|
||||||
|
files <- meta$data$latestVersion$files
|
||||||
|
matches <- Filter(function(x) {
|
||||||
|
same_file <- identical(x$dataFile$filename, filename)
|
||||||
|
same_dir <- is.na(directory) || identical(x$directoryLabel, directory)
|
||||||
|
same_file && same_dir
|
||||||
|
}, files)
|
||||||
|
if (length(matches) == 0) NA_integer_ else as.integer(matches[[1]]$dataFile$id)
|
||||||
|
}, error = function(e) {
|
||||||
|
message("FAILED Dataverse lookup ", doi, " ", filename, ": ", conditionMessage(e))
|
||||||
|
NA_integer_
|
||||||
|
})
|
||||||
|
if (is.na(fid)) {
|
||||||
|
message("FAILED Dataverse lookup ", doi, ": file not found: ", filename)
|
||||||
|
return(FALSE)
|
||||||
|
}
|
||||||
|
download_file(paste0("https://dataverse.harvard.edu/api/access/datafile/", fid), dest)
|
||||||
|
}
|
||||||
|
|
||||||
|
download_manifesto <- function(dest) {
|
||||||
|
key <- Sys.getenv("MANIFESTO_API_KEY", Sys.getenv("PARTY2D_MANIFESTO_API_KEY", ""))
|
||||||
|
if (!nzchar(key)) {
|
||||||
|
message("SKIP Manifesto: set MANIFESTO_API_KEY or PARTY2D_MANIFESTO_API_KEY")
|
||||||
|
return(FALSE)
|
||||||
|
}
|
||||||
|
url <- "https://manifesto-project.wzb.eu/api/v1/get_core?key=MPDS2025a&raw=true"
|
||||||
|
download_file(
|
||||||
|
url,
|
||||||
|
dest,
|
||||||
|
overwrite = TRUE,
|
||||||
|
headers = c("Referer" = "https://manifesto-project.wzb.eu/datasets", "API_KEY" = key)
|
||||||
|
)
|
||||||
|
}
|
||||||
|
|
||||||
|
download_vparty <- function(dest) {
|
||||||
|
email <- Sys.getenv("PARTY2D_VDEM_EMAIL", "")
|
||||||
|
if (!nzchar(email)) {
|
||||||
|
message("SKIP V-Party: set PARTY2D_VDEM_EMAIL to use V-Dem's required download form")
|
||||||
|
return(FALSE)
|
||||||
|
}
|
||||||
|
curl <- Sys.which("curl")
|
||||||
|
if (!nzchar(curl)) {
|
||||||
|
message("FAILED V-Party: curl is required for the provider form")
|
||||||
|
return(FALSE)
|
||||||
|
}
|
||||||
|
page <- "https://www.v-dem.net/data/v-party-dataset/country-party-date-v2/"
|
||||||
|
tmpdir <- tempfile("vparty")
|
||||||
|
dir.create(tmpdir)
|
||||||
|
on.exit(unlink(tmpdir, recursive = TRUE), add = TRUE)
|
||||||
|
cookie <- file.path(tmpdir, "cookies.txt")
|
||||||
|
html <- file.path(tmpdir, "page.html")
|
||||||
|
payload <- file.path(tmpdir, "payload.bin")
|
||||||
|
zip_path <- file.path(tmpdir, "vparty.zip")
|
||||||
|
status <- system2(curl, c("-L", "-A", ua, "-c", cookie, "-b", cookie, "-o", html, page), stdout = TRUE, stderr = TRUE)
|
||||||
|
if (!file.exists(html)) {
|
||||||
|
message("FAILED V-Party form load")
|
||||||
|
return(FALSE)
|
||||||
|
}
|
||||||
|
page_text <- paste(readLines(html, warn = FALSE), collapse = "\n")
|
||||||
|
csrf <- sub('.*name="csrfmiddlewaretoken"[^>]*value="([^"]*)".*', '\\1', page_text)
|
||||||
|
if (identical(csrf, page_text)) csrf <- ""
|
||||||
|
gender <- Sys.getenv("PARTY2D_VDEM_GENDER", "")
|
||||||
|
form <- c(
|
||||||
|
"csrfmiddlewaretoken", csrf,
|
||||||
|
"email", email,
|
||||||
|
"gender", gender,
|
||||||
|
"accept_terms", "on",
|
||||||
|
"dataset_file", "17",
|
||||||
|
"website", ""
|
||||||
|
)
|
||||||
|
args <- c(
|
||||||
|
"-L", "-A", ua, "-c", cookie, "-b", cookie,
|
||||||
|
"-H", paste0("Referer: ", page),
|
||||||
|
"-H", paste0("X-CSRFToken: ", csrf),
|
||||||
|
"-H", "X-Requested-With: XMLHttpRequest",
|
||||||
|
"-o", payload,
|
||||||
|
"--data-urlencode", paste0(form[1], "=", form[2]),
|
||||||
|
"--data-urlencode", paste0(form[3], "=", form[4]),
|
||||||
|
"--data-urlencode", paste0(form[5], "=", form[6]),
|
||||||
|
"--data-urlencode", paste0(form[7], "=", form[8]),
|
||||||
|
"--data-urlencode", paste0(form[9], "=", form[10]),
|
||||||
|
"--data-urlencode", paste0(form[11], "=", form[12]),
|
||||||
|
paste0(page, "#dataset-download")
|
||||||
|
)
|
||||||
|
system2(curl, args, stdout = TRUE, stderr = TRUE)
|
||||||
|
if (!file.exists(payload) || file.info(payload)$size == 0) {
|
||||||
|
message("FAILED V-Party form submit")
|
||||||
|
return(FALSE)
|
||||||
|
}
|
||||||
|
bytes <- readBin(payload, "raw", n = min(file.info(payload)$size, 4))
|
||||||
|
if (length(bytes) >= 2 && identical(as.integer(bytes[1:2]), c(0x50L, 0x4bL))) {
|
||||||
|
file.copy(payload, zip_path, overwrite = TRUE)
|
||||||
|
} else {
|
||||||
|
text <- paste(readLines(payload, warn = FALSE), collapse = "\n")
|
||||||
|
m <- regexpr('https?://[^" ]*CPD_V-Party_R_v2\\.zip|/[^" ]*CPD_V-Party_R_v2\\.zip', text)
|
||||||
|
if (m[1] < 0) {
|
||||||
|
message("FAILED V-Party form response did not include a recognizable ZIP download")
|
||||||
|
return(FALSE)
|
||||||
|
}
|
||||||
|
url <- regmatches(text, m)[1]
|
||||||
|
if (startsWith(url, "/")) url <- paste0("https://www.v-dem.net", url)
|
||||||
|
if (!download_file(url, zip_path, overwrite = TRUE, headers = c("Referer" = page))) return(FALSE)
|
||||||
|
}
|
||||||
|
listing <- utils::unzip(zip_path, list = TRUE)
|
||||||
|
member <- listing$Name[grepl("\\.(rds|rda|rdata)$", listing$Name, ignore.case = TRUE)][1]
|
||||||
|
if (is.na(member)) {
|
||||||
|
message("FAILED V-Party ZIP contains no R data file")
|
||||||
|
return(FALSE)
|
||||||
|
}
|
||||||
|
dir.create(dirname(dest), recursive = TRUE, showWarnings = FALSE)
|
||||||
|
utils::unzip(zip_path, files = member, exdir = tmpdir, overwrite = TRUE)
|
||||||
|
file.copy(file.path(tmpdir, member), dest, overwrite = TRUE)
|
||||||
|
message("Extracted V-Party R data: ", dest)
|
||||||
|
TRUE
|
||||||
|
}
|
||||||
|
|
||||||
|
status <- c()
|
||||||
|
status["PolDem"] <- download_file("https://poldem.eui.eu/downloads/cosa/poldem-election_all.csv", file.path(raw_dir, "poldem", "poldem-election_all.csv"))
|
||||||
|
status["PartyFacts external parties"] <- download_file("https://partyfacts.herokuapp.com/download/external-parties-csv/", file.path(raw_dir, "partyfacts", "partyfacts-external-parties.csv"))
|
||||||
|
status["Manifesto MPDS 2025a"] <- download_manifesto(file.path(raw_dir, "manifesto", "MPDataset_MPDS2025a.csv"))
|
||||||
|
|
||||||
|
ches_base <- "https://www.chesdata.eu/s/"
|
||||||
|
ches_files <- c(
|
||||||
|
"1999-2019_CHES_dataset_meansv3.csv" = "1999-2019_CHES_dataset_means(v3).csv",
|
||||||
|
"1999-2024_CHES_dataset_meansV2-3k4l.csv" = "1999-2024_CHES_dataset_meansV2-3k4l.csv",
|
||||||
|
"CHES_2024_final_v2.csv" = "CHES_2024_final_v2.csv",
|
||||||
|
"CHES_2024_ALL_Stacked_Expert.csv" = "CHES_2024_expert_level.csv",
|
||||||
|
"CHES_CA2023.csv" = "CHES_CA2023.csv",
|
||||||
|
"CHES_CA2023_expert-level.csv" = "CHES_CA2023_expert_level.csv",
|
||||||
|
"ches_la_2020_aggregate_level_v01.csv" = "ches_la_2020_aggregate_level_v01.csv",
|
||||||
|
"ches_la_2020_expert_level_v01.csv" = "CHES_LA2020_expert_level.csv",
|
||||||
|
"CHES_ISRAEL_means_2021_2022.csv" = "CHES_ISRAEL_means_2021_2022.csv",
|
||||||
|
"CHES_ISRAEL_expert_level_2021_2022.csv" = "CHES_IL_expert_level.csv"
|
||||||
|
)
|
||||||
|
for (remote in names(ches_files)) {
|
||||||
|
local <- ches_files[[remote]]
|
||||||
|
status[paste("CHES", local)] <- download_file(paste0(ches_base, utils::URLencode(remote, reserved = TRUE)), file.path(raw_dir, "ches", local))
|
||||||
|
}
|
||||||
|
|
||||||
|
status["POPPA integrated v2"] <- download_dataverse("doi:10.7910/DVN/RMQREQ", "poppa_integrated_v2.rds", file.path(raw_dir, "poppa", "poppa_integrated_v2.rds"), "final_data_v2")
|
||||||
|
status["GPS 2019 party"] <- download_dataverse("doi:10.7910/DVN/WMGTNS", "Global Party Survey by Party SPSS V2_1_Apr_2020-2.tab", file.path(raw_dir, "gps", "Global Party Survey by Party SPSS V2_1_Apr_2020-2.tab"))
|
||||||
|
status["V-Party"] <- download_vparty(file.path(raw_dir, "vparty", "V-Dem-CPD-Party-V2.rds"))
|
||||||
|
|
||||||
|
report <- file.path(report_dir, "download_sources_report.md")
|
||||||
|
lines <- c("# Download sources report", "", "| source | status |", "| --- | --- |")
|
||||||
|
for (name in names(status)) lines <- c(lines, paste0("| ", name, " | ", if (isTRUE(status[[name]])) "ok" else "missing/failed", " |"))
|
||||||
|
writeLines(lines, report)
|
||||||
|
message("Wrote download report: ", report)
|
||||||
|
|
||||||
|
quit(status = if (all(status)) 0 else 2)
|
||||||
@@ -0,0 +1,348 @@
|
|||||||
|
# ============================================================
|
||||||
|
# 02_build_model_inputs.R - Master Data Pipeline Orchestrator
|
||||||
|
# ============================================================
|
||||||
|
# Coordinates all data processing sub-scripts and produces
|
||||||
|
# final output files for the two-dimensional party-position model. By default this writes only
|
||||||
|
# to local-only directories under _local/ and never overwrites committed data/.
|
||||||
|
#
|
||||||
|
# Sub-scripts (run conditionally based on intermediate file existence):
|
||||||
|
# process_manifesto.R -> manifesto_data.csv
|
||||||
|
# process_poldem.R -> poldem_data.csv
|
||||||
|
# process_expert.R -> expert_raw.csv, lr_data_raw.csv
|
||||||
|
# process_morgan.R -> morgan_data.csv, morgan_lr.csv
|
||||||
|
#
|
||||||
|
# Final generated model inputs:
|
||||||
|
# text_data.csv - Combined manifesto + PolDem
|
||||||
|
# expert.csv - Expert survey data (CHES, V-Party, POPPA, GPS)
|
||||||
|
# lr_data.csv - General left-right anchoring data
|
||||||
|
# ============================================================
|
||||||
|
|
||||||
|
library(tidyverse)
|
||||||
|
library(countrycode)
|
||||||
|
|
||||||
|
cmd_args <- commandArgs(trailingOnly = FALSE)
|
||||||
|
file_arg <- grep("^--file=", cmd_args, value = TRUE)
|
||||||
|
if (length(file_arg) > 0) {
|
||||||
|
this_file <- normalizePath(sub("^--file=", "", file_arg[[1]]), mustWork = TRUE)
|
||||||
|
repo_root <- normalizePath(file.path(dirname(this_file), "..", ".."), mustWork = TRUE)
|
||||||
|
} else {
|
||||||
|
repo_root <- normalizePath(getwd(), mustWork = TRUE)
|
||||||
|
}
|
||||||
|
script_dir <- file.path(repo_root, "data-setup", "R")
|
||||||
|
build_dir <- normalizePath(
|
||||||
|
Sys.getenv("PARTY2D_BUILD_DIR", file.path(repo_root, "_local", "build")),
|
||||||
|
mustWork = FALSE
|
||||||
|
)
|
||||||
|
generated_input_dir <- normalizePath(
|
||||||
|
Sys.getenv("PARTY2D_GENERATED_INPUT_DIR", file.path(repo_root, "_local", "generated-inputs")),
|
||||||
|
mustWork = FALSE
|
||||||
|
)
|
||||||
|
dir.create(build_dir, recursive = TRUE, showWarnings = FALSE)
|
||||||
|
dir.create(generated_input_dir, recursive = TRUE, showWarnings = FALSE)
|
||||||
|
|
||||||
|
# The source-processing scripts use relative paths for intermediate files.
|
||||||
|
# Keep those intermediates in the ignored build directory, never committed data/.
|
||||||
|
setwd(build_dir)
|
||||||
|
|
||||||
|
# Static model support inputs are versioned in data/ and copied into the local
|
||||||
|
# generated-input set for comparison. They are not regenerated by raw-source setup.
|
||||||
|
for (support_file in c("union_mapping.csv", "party_families.csv")) {
|
||||||
|
src <- file.path(repo_root, "data", support_file)
|
||||||
|
if (!file.exists(src)) {
|
||||||
|
stop("Required committed support input not found: ", src)
|
||||||
|
}
|
||||||
|
file.copy(src, file.path(build_dir, support_file), overwrite = TRUE)
|
||||||
|
}
|
||||||
|
|
||||||
|
cat("============================================================\n")
|
||||||
|
cat("Data Management Pipeline\n")
|
||||||
|
cat("============================================================\n\n")
|
||||||
|
cat("Build directory: ", build_dir, "\n", sep = "")
|
||||||
|
cat("Generated input directory: ", generated_input_dir, "\n\n", sep = "")
|
||||||
|
|
||||||
|
# ============================================================
|
||||||
|
# Configuration: Set to TRUE to force re-run of sub-scripts
|
||||||
|
# ============================================================
|
||||||
|
|
||||||
|
FORCE_RERUN_MANIFESTO <- FALSE
|
||||||
|
FORCE_RERUN_POLDEM <- FALSE
|
||||||
|
FORCE_RERUN_EXPERT <- FALSE
|
||||||
|
FORCE_RERUN_MORGAN <- FALSE
|
||||||
|
|
||||||
|
# ============================================================
|
||||||
|
# Step 1: Manifesto Data
|
||||||
|
# ============================================================
|
||||||
|
|
||||||
|
cat("Step 1: Manifesto data\n")
|
||||||
|
if (!file.exists("manifesto_data.csv") || !file.exists("election_data.csv") || FORCE_RERUN_MANIFESTO) {
|
||||||
|
cat(" Running process_manifesto.R...\n")
|
||||||
|
source(file.path(script_dir, "process_manifesto.R"))
|
||||||
|
} else {
|
||||||
|
cat(" Loading cached manifesto_data.csv and election_data.csv...\n")
|
||||||
|
}
|
||||||
|
manifesto <- read_csv("manifesto_data.csv", show_col_types = FALSE)
|
||||||
|
election_data <- read_csv("election_data.csv", show_col_types = FALSE)
|
||||||
|
cat(sprintf(" Loaded manifesto: %d rows, %d parties\n", nrow(manifesto), n_distinct(manifesto$party)))
|
||||||
|
cat(sprintf(" Loaded election: %d rows, %d parties\n\n", nrow(election_data), n_distinct(election_data$party)))
|
||||||
|
|
||||||
|
# ============================================================
|
||||||
|
# Step 2: PolDem Media Data
|
||||||
|
# ============================================================
|
||||||
|
|
||||||
|
cat("Step 2: PolDem media data\n")
|
||||||
|
if (!file.exists("poldem_data.csv") || FORCE_RERUN_POLDEM) {
|
||||||
|
cat(" Running process_poldem.R...\n")
|
||||||
|
source(file.path(script_dir, "process_poldem.R"))
|
||||||
|
} else {
|
||||||
|
cat(" Loading cached poldem_data.csv...\n")
|
||||||
|
}
|
||||||
|
poldem_data <- read_csv("poldem_data.csv", show_col_types = FALSE)
|
||||||
|
cat(sprintf(" Loaded: %d rows, %d parties\n\n", nrow(poldem_data), n_distinct(poldem_data$party)))
|
||||||
|
|
||||||
|
# ============================================================
|
||||||
|
# Step 4: Expert Survey Data
|
||||||
|
# ============================================================
|
||||||
|
|
||||||
|
cat("Step 3: Expert survey data\n")
|
||||||
|
if (!file.exists("expert_raw.csv") || !file.exists("lr_data_raw.csv") || FORCE_RERUN_EXPERT) {
|
||||||
|
cat(" Running process_expert.R...\n")
|
||||||
|
source(file.path(script_dir, "process_expert.R"))
|
||||||
|
} else {
|
||||||
|
cat(" Loading cached expert_raw.csv and lr_data_raw.csv...\n")
|
||||||
|
}
|
||||||
|
expert_raw <- read_csv("expert_raw.csv", show_col_types = FALSE)
|
||||||
|
lr_data_raw <- read_csv("lr_data_raw.csv", show_col_types = FALSE)
|
||||||
|
cat(sprintf(" Expert: %d rows, LR: %d rows\n\n", nrow(expert_raw), nrow(lr_data_raw)))
|
||||||
|
|
||||||
|
# ============================================================
|
||||||
|
# Step 3b: Morgan (1976) Historical Expert Data
|
||||||
|
# ============================================================
|
||||||
|
|
||||||
|
cat("Step 3b: Morgan (1976) historical L-R data\n")
|
||||||
|
|
||||||
|
# First run to generate morgan_data.csv if needed
|
||||||
|
if (!file.exists("morgan_data.csv") || FORCE_RERUN_MORGAN) {
|
||||||
|
cat(" Running process_morgan.R (initial processing)...\n")
|
||||||
|
source(file.path(script_dir, "process_morgan.R"))
|
||||||
|
}
|
||||||
|
|
||||||
|
# morgan_lr.csv depends on text_data.csv, so we need to check if it needs regeneration
|
||||||
|
# It will be generated/regenerated below after text_data is created
|
||||||
|
|
||||||
|
# ============================================================
|
||||||
|
# Step 4: Combine Text Data Sources
|
||||||
|
# ============================================================
|
||||||
|
|
||||||
|
cat("Step 4: Combining text data sources\n")
|
||||||
|
text_data <- bind_rows(manifesto, poldem_data)
|
||||||
|
cat(sprintf(" Combined text_data: %d rows\n", nrow(text_data)))
|
||||||
|
|
||||||
|
# Save unfiltered text_data for reproducible mismatch diagnosis
|
||||||
|
write_csv(text_data, "text_data_unfiltered.csv")
|
||||||
|
cat(sprintf(" Saved unfiltered text_data: %d rows, %d parties\n", nrow(text_data), n_distinct(text_data$party)))
|
||||||
|
|
||||||
|
# ============================================================
|
||||||
|
# Step 4b: Party Renames (applied before filtering)
|
||||||
|
# ============================================================
|
||||||
|
# Renames must happen BEFORE the relevance filter so that party IDs
|
||||||
|
# match across text_data and expert_raw when computing expert coverage.
|
||||||
|
|
||||||
|
# Simple renames only (organizational continuity: same leadership/members)
|
||||||
|
simple_renames <- c(
|
||||||
|
`10` = 1816L, # DE: Greens -> Bündnis90/Grüne
|
||||||
|
`276` = 120L, # RO: FDSN/PDSR -> PSD (renamed 2001)
|
||||||
|
`8054` = 878L, # IT: PDS -> DS (renamed 1998)
|
||||||
|
`1696` = 813L, # IT: MSI -> AN (refounded 1995)
|
||||||
|
`553` = 1968L, # BE: Vlaams Blok -> Vlaams Belang (refounded 2004)
|
||||||
|
`8058` = 1626L # IT: Forza Italia (refounded 2013) -> Forza Italia (same party, Berlusconi)
|
||||||
|
)
|
||||||
|
|
||||||
|
apply_simple_renames <- function(df) {
|
||||||
|
for (old_id in names(simple_renames)) {
|
||||||
|
df <- df %>%
|
||||||
|
mutate(party = ifelse(party == as.integer(old_id), simple_renames[[old_id]], party))
|
||||||
|
}
|
||||||
|
df
|
||||||
|
}
|
||||||
|
|
||||||
|
cat("\nStep 4b: Party renames\n")
|
||||||
|
text_data <- apply_simple_renames(text_data)
|
||||||
|
cat(sprintf(" Applied %d renames to text_data\n", length(simple_renames)))
|
||||||
|
|
||||||
|
# ============================================================
|
||||||
|
# Step 4c: Relevance Filter
|
||||||
|
# ============================================================
|
||||||
|
# Design: R pipeline filters for RELEVANCE (is this party worth modeling?).
|
||||||
|
# Julia pipeline handles INTERPOLATION QUALITY (MAX_GAP=7 segment splitting, MIN_OBS=2).
|
||||||
|
# Expert survey coverage is a relevance signal: CHES only covers parties with >1% vote share.
|
||||||
|
|
||||||
|
cat("\nStep 4c: Relevance filter\n")
|
||||||
|
parties_before <- n_distinct(text_data$party)
|
||||||
|
|
||||||
|
# Compute expert coverage per party (with renames applied for consistent matching)
|
||||||
|
expert_year_counts <- bind_rows(
|
||||||
|
expert_raw %>% select(party, year),
|
||||||
|
lr_data_raw %>% select(party, year)
|
||||||
|
) %>% distinct() %>%
|
||||||
|
apply_simple_renames() %>%
|
||||||
|
distinct() %>%
|
||||||
|
count(party, name = "expert_years")
|
||||||
|
|
||||||
|
expert_party_ids <- unique(expert_year_counts$party)
|
||||||
|
cat(sprintf(" Parties with expert data: %d\n", length(expert_party_ids)))
|
||||||
|
|
||||||
|
# Three-tier relevance filter:
|
||||||
|
# Tier 1: 3+ text data years (always include, regardless of expert data)
|
||||||
|
# Tier 2: 2 text years + any expert data (major newer parties like M5S, ANO, LREM)
|
||||||
|
# Tier 3: 1 text year + 3+ expert survey years (parties with rich expert coverage)
|
||||||
|
text_data <- text_data %>%
|
||||||
|
group_by(country, party) %>%
|
||||||
|
mutate(n_years = n_distinct(year)) %>%
|
||||||
|
ungroup() %>%
|
||||||
|
left_join(expert_year_counts, by = "party") %>%
|
||||||
|
mutate(expert_years = replace_na(expert_years, 0L)) %>%
|
||||||
|
mutate(
|
||||||
|
tier = case_when(
|
||||||
|
n_years >= 3 ~ 1L,
|
||||||
|
n_years >= 2 & party %in% expert_party_ids ~ 2L,
|
||||||
|
n_years >= 1 & expert_years >= 3 ~ 3L,
|
||||||
|
TRUE ~ 0L
|
||||||
|
)
|
||||||
|
) %>%
|
||||||
|
filter(tier > 0) %>%
|
||||||
|
select(-n_years, -expert_years, -tier)
|
||||||
|
|
||||||
|
parties_after <- n_distinct(text_data$party)
|
||||||
|
cat(sprintf(" Parties before filter: %d\n", parties_before))
|
||||||
|
cat(sprintf(" Parties after filter: %d\n", parties_after))
|
||||||
|
cat(sprintf(" Parties removed: %d\n\n", parties_before - parties_after))
|
||||||
|
|
||||||
|
# ============================================================
|
||||||
|
# Step 5: Party Harmonization
|
||||||
|
# ============================================================
|
||||||
|
|
||||||
|
cat("Step 5: Party harmonization (union-aware)\n")
|
||||||
|
|
||||||
|
# Load union mapping to identify constituent parties
|
||||||
|
union_map <- read_csv("union_mapping.csv", show_col_types = FALSE)
|
||||||
|
|
||||||
|
# Build set of constituent parties whose union is in text_data
|
||||||
|
constituent_parties <- union_map %>%
|
||||||
|
filter(manifesto_pf_id %in% unique(text_data$party)) %>%
|
||||||
|
pull(expert_pf_id)
|
||||||
|
|
||||||
|
cat(sprintf(" Union mappings loaded: %d rows covering %d unions\n",
|
||||||
|
nrow(union_map), n_distinct(union_map$manifesto_pf_id)))
|
||||||
|
cat(sprintf(" Constituent parties with unions in text_data: %d\n",
|
||||||
|
length(unique(constituent_parties))))
|
||||||
|
|
||||||
|
# Deduplicate union manifesto rows: where multiple CMP codes map to the same
|
||||||
|
# union PF ID with identical content, keep only one set per (party, year, var)
|
||||||
|
text_data_before_dedup <- nrow(text_data)
|
||||||
|
text_data <- text_data %>%
|
||||||
|
distinct(country, party, year, var, .keep_all = TRUE)
|
||||||
|
cat(sprintf(" Text data: %d unique parties after harmonization\n", n_distinct(text_data$party)))
|
||||||
|
cat(sprintf(" Text data: deduplicated %d -> %d rows\n", text_data_before_dedup, nrow(text_data)))
|
||||||
|
|
||||||
|
# Filter expert data: keep parties in text_data OR constituent parties of unions in text_data
|
||||||
|
expert <- expert_raw %>%
|
||||||
|
apply_simple_renames() %>%
|
||||||
|
group_by(country, party, var, year) %>%
|
||||||
|
summarise(
|
||||||
|
val = mean(val, na.rm = TRUE),
|
||||||
|
val_int = first(val_int),
|
||||||
|
n_scale = first(n_scale),
|
||||||
|
n_experts = first(n_experts),
|
||||||
|
project = first(project),
|
||||||
|
type_low = first(type_low),
|
||||||
|
type_high = first(type_high),
|
||||||
|
.groups = "drop"
|
||||||
|
) %>%
|
||||||
|
filter(party %in% unique(text_data$party) | party %in% constituent_parties)
|
||||||
|
|
||||||
|
lr_data <- lr_data_raw %>%
|
||||||
|
apply_simple_renames() %>%
|
||||||
|
group_by(country, party, var, year) %>%
|
||||||
|
summarise(
|
||||||
|
val = mean(val, na.rm = TRUE),
|
||||||
|
val_int = first(val_int),
|
||||||
|
n_scale = first(n_scale),
|
||||||
|
n_experts = first(n_experts),
|
||||||
|
project = first(project),
|
||||||
|
.groups = "drop"
|
||||||
|
) %>%
|
||||||
|
filter(party %in% unique(text_data$party) | party %in% constituent_parties)
|
||||||
|
|
||||||
|
cat(sprintf(" Expert data: %d rows (filtered to text_data parties)\n", nrow(expert)))
|
||||||
|
cat(sprintf(" LR data (CHES/POPPA): %d rows (filtered to text_data parties)\n", nrow(lr_data)))
|
||||||
|
|
||||||
|
# ============================================================
|
||||||
|
# Step 5b: Integrate Morgan L-R Data
|
||||||
|
# ============================================================
|
||||||
|
|
||||||
|
cat("\nStep 5b: Morgan L-R data integration\n")
|
||||||
|
|
||||||
|
# Generate morgan_lr.csv (requires text_data.csv to exist)
|
||||||
|
# We need to regenerate it if text_data changed or if forced
|
||||||
|
if (!file.exists("morgan_lr.csv") || FORCE_RERUN_MORGAN) {
|
||||||
|
cat(" Generating morgan_lr.csv...\n")
|
||||||
|
# Write text_data first so morgan script can use it
|
||||||
|
write_csv(text_data, "text_data.csv")
|
||||||
|
source(file.path(script_dir, "process_morgan.R"))
|
||||||
|
}
|
||||||
|
|
||||||
|
# Load and integrate Morgan L-R data
|
||||||
|
if (file.exists("morgan_lr.csv")) {
|
||||||
|
morgan_lr <- read_csv("morgan_lr.csv", show_col_types = FALSE) %>%
|
||||||
|
apply_simple_renames() %>%
|
||||||
|
filter(party %in% unique(text_data$party) | party %in% constituent_parties)
|
||||||
|
|
||||||
|
cat(sprintf(" Morgan L-R: %d rows (filtered to text_data parties)\n", nrow(morgan_lr)))
|
||||||
|
cat(sprintf(" Morgan parties: %d\n", n_distinct(morgan_lr$party)))
|
||||||
|
cat(sprintf(" Morgan year range: %d-%d\n", min(morgan_lr$year), max(morgan_lr$year)))
|
||||||
|
|
||||||
|
# Combine with existing lr_data
|
||||||
|
lr_data_before <- nrow(lr_data)
|
||||||
|
lr_data <- bind_rows(lr_data, morgan_lr) %>%
|
||||||
|
arrange(country, party, year, var)
|
||||||
|
|
||||||
|
cat(sprintf(" Combined LR data: %d rows (+%d from Morgan)\n",
|
||||||
|
nrow(lr_data), nrow(lr_data) - lr_data_before))
|
||||||
|
} else {
|
||||||
|
cat(" Warning: morgan_lr.csv not found, skipping Morgan integration\n")
|
||||||
|
}
|
||||||
|
|
||||||
|
cat("\n")
|
||||||
|
|
||||||
|
# ============================================================
|
||||||
|
# Step 6: Write Final Outputs
|
||||||
|
# ============================================================
|
||||||
|
|
||||||
|
cat("Step 6: Writing final outputs\n")
|
||||||
|
|
||||||
|
write_csv(text_data, "text_data.csv")
|
||||||
|
write_csv(expert, "expert.csv")
|
||||||
|
write_csv(lr_data, "lr_data.csv")
|
||||||
|
|
||||||
|
for (final_file in c("text_data.csv", "expert.csv", "lr_data.csv", "union_mapping.csv", "party_families.csv")) {
|
||||||
|
file.copy(file.path(build_dir, final_file), file.path(generated_input_dir, final_file), overwrite = TRUE)
|
||||||
|
}
|
||||||
|
|
||||||
|
cat("\n============================================================\n")
|
||||||
|
cat("Pipeline Complete!\n")
|
||||||
|
cat("============================================================\n\n")
|
||||||
|
|
||||||
|
cat("Output files written:\n")
|
||||||
|
cat(sprintf(" local generated input dir: %s\n", generated_input_dir))
|
||||||
|
cat(sprintf(" text_data.csv: %d rows\n", nrow(text_data)))
|
||||||
|
cat(sprintf(" - Manifesto: %d rows\n", sum(grepl("_manifesto", text_data$var))))
|
||||||
|
cat(sprintf(" - PolDem: %d rows\n", sum(grepl("_poldem", text_data$var))))
|
||||||
|
cat(sprintf(" expert.csv: %d rows\n", nrow(expert)))
|
||||||
|
cat(sprintf(" lr_data.csv: %d rows\n", nrow(lr_data)))
|
||||||
|
cat(sprintf(" - CHES: %d rows\n", sum(lr_data$var == "lr_ches")))
|
||||||
|
cat(sprintf(" - POPPA: %d rows\n", sum(lr_data$var == "lr_poppa")))
|
||||||
|
cat(sprintf(" - Morgan: %d rows\n", sum(lr_data$var == "lr_morgan")))
|
||||||
|
|
||||||
|
cat("\nUnique parties in text_data:", n_distinct(text_data$party), "\n")
|
||||||
|
cat("Countries:", paste(sort(unique(text_data$country)), collapse = ", "), "\n")
|
||||||
|
cat("Year range:", min(text_data$year, na.rm = TRUE), "-", max(text_data$year, na.rm = TRUE), "\n")
|
||||||
@@ -0,0 +1,53 @@
|
|||||||
|
#!/usr/bin/env Rscript
|
||||||
|
|
||||||
|
args <- commandArgs(trailingOnly = FALSE)
|
||||||
|
file_arg <- args[grepl("^--file=", args)][1]
|
||||||
|
if (!is.na(file_arg)) {
|
||||||
|
script_path <- sub("^--file=", "", file_arg)
|
||||||
|
repo_root <- normalizePath(file.path(dirname(script_path), "..", ".."), mustWork = FALSE)
|
||||||
|
} else {
|
||||||
|
repo_root <- normalizePath(getwd(), mustWork = TRUE)
|
||||||
|
}
|
||||||
|
if (!dir.exists(file.path(repo_root, "data"))) repo_root <- normalizePath(getwd(), mustWork = TRUE)
|
||||||
|
|
||||||
|
generated_dir <- Sys.getenv("PARTY2D_GENERATED_INPUT_DIR", file.path(repo_root, "_local", "generated-inputs"))
|
||||||
|
report_dir <- Sys.getenv("PARTY2D_REPORT_DIR", file.path(repo_root, "_local", "reports"))
|
||||||
|
dir.create(report_dir, recursive = TRUE, showWarnings = FALSE)
|
||||||
|
|
||||||
|
files <- c("text_data.csv", "expert.csv", "lr_data.csv", "union_mapping.csv", "party_families.csv")
|
||||||
|
|
||||||
|
file_info <- lapply(files, function(file) {
|
||||||
|
committed <- file.path(repo_root, "data", file)
|
||||||
|
generated <- file.path(generated_dir, file)
|
||||||
|
committed_exists <- file.exists(committed)
|
||||||
|
generated_exists <- file.exists(generated)
|
||||||
|
committed_size <- if (committed_exists) file.info(committed)$size else NA_real_
|
||||||
|
generated_size <- if (generated_exists) file.info(generated)$size else NA_real_
|
||||||
|
identical_bytes <- committed_exists && generated_exists && isTRUE(tools::md5sum(committed) == tools::md5sum(generated))
|
||||||
|
data.frame(
|
||||||
|
file = file,
|
||||||
|
committed_exists = committed_exists,
|
||||||
|
generated_exists = generated_exists,
|
||||||
|
committed_size = committed_size,
|
||||||
|
generated_size = generated_size,
|
||||||
|
identical_bytes = identical_bytes,
|
||||||
|
stringsAsFactors = FALSE
|
||||||
|
)
|
||||||
|
})
|
||||||
|
|
||||||
|
summary <- do.call(rbind, file_info)
|
||||||
|
report <- file.path(report_dir, "input_comparison.md")
|
||||||
|
lines <- c(
|
||||||
|
"# Model input comparison",
|
||||||
|
"",
|
||||||
|
paste0("Generated: ", format(Sys.time(), "%Y-%m-%d %H:%M:%S %Z")),
|
||||||
|
"",
|
||||||
|
"| File | Committed exists | Generated exists | Committed bytes | Generated bytes | Identical bytes |",
|
||||||
|
"| --- | --- | --- | ---: | ---: | --- |",
|
||||||
|
apply(summary, 1, function(row) {
|
||||||
|
paste0("| ", row[["file"]], " | ", row[["committed_exists"]], " | ", row[["generated_exists"]], " | ", row[["committed_size"]], " | ", row[["generated_size"]], " | ", row[["identical_bytes"]], " |")
|
||||||
|
})
|
||||||
|
)
|
||||||
|
writeLines(lines, report)
|
||||||
|
print(summary, row.names = FALSE)
|
||||||
|
message("Comparison report written to ", report)
|
||||||
@@ -0,0 +1,620 @@
|
|||||||
|
# ============================================================
|
||||||
|
# process_expert.R - Expert Survey Data Processing
|
||||||
|
# ============================================================
|
||||||
|
# Processes expert survey data from multiple sources:
|
||||||
|
# - Chapel Hill Expert Survey (CHES)
|
||||||
|
# - V-Party Dataset
|
||||||
|
# - POPPA
|
||||||
|
# - GPS (Norris)
|
||||||
|
#
|
||||||
|
# Outputs: expert_raw.csv, lr_data_raw.csv
|
||||||
|
#
|
||||||
|
# V5 changes:
|
||||||
|
# - val_int (integer rounded to nearest scale point) and n_scale columns
|
||||||
|
# - n_experts column preserved (not dropped)
|
||||||
|
# - V-Party cultural expansion: 5 native items replace GPS ep_v6_lib_cons
|
||||||
|
# - V-Party economic expansion: v2pawelf added
|
||||||
|
# - Reverse-coding for V-Party cultural + welfare items
|
||||||
|
# ============================================================
|
||||||
|
|
||||||
|
library(tidyverse)
|
||||||
|
library(countrycode)
|
||||||
|
library(haven)
|
||||||
|
library(foreign)
|
||||||
|
|
||||||
|
# Set working directory (works both in RStudio and command line)
|
||||||
|
if (interactive() && requireNamespace("rstudioapi", quietly = TRUE)) {
|
||||||
|
try(setwd(dirname(rstudioapi::getActiveDocumentContext()$path)), silent = TRUE)
|
||||||
|
}
|
||||||
|
|
||||||
|
cat("Processing expert survey data...\n")
|
||||||
|
|
||||||
|
raw_data_dir <- Sys.getenv(
|
||||||
|
"PARTY2D_RAW_DATA_DIR",
|
||||||
|
unset = file.path("..", "..", "_local", "raw")
|
||||||
|
)
|
||||||
|
ches_dir <- file.path(raw_data_dir, "ches")
|
||||||
|
vparty_dir <- file.path(raw_data_dir, "vparty")
|
||||||
|
poppa_dir <- file.path(raw_data_dir, "poppa")
|
||||||
|
gps_dir <- file.path(raw_data_dir, "gps")
|
||||||
|
partyfacts_path <- file.path(raw_data_dir, "partyfacts", "partyfacts-external-parties.csv")
|
||||||
|
|
||||||
|
ches_country_iso2 <- function(country_id) {
|
||||||
|
lookup <- c(
|
||||||
|
`1` = "BE", `2` = "DK", `3` = "DE", `4` = "GR", `5` = "ES",
|
||||||
|
`6` = "FR", `7` = "IE", `8` = "IT", `10` = "NL", `11` = "GB",
|
||||||
|
`12` = "PT", `13` = "AT", `14` = "FI", `16` = "SE", `20` = "BG",
|
||||||
|
`21` = "CZ", `22` = "EE", `23` = "HU", `24` = "LV", `25` = "LT",
|
||||||
|
`26` = "PL", `27` = "RO", `28` = "SK", `29` = "SI", `31` = "HR",
|
||||||
|
`32` = "TR", `33` = "NO", `34` = "CH", `35` = "MT", `36` = "CY",
|
||||||
|
`37` = "IS", `38` = "CH", `40` = "CY"
|
||||||
|
)
|
||||||
|
unname(lookup[as.character(country_id)])
|
||||||
|
}
|
||||||
|
|
||||||
|
ches2024_country_iso2 <- function(country_id) {
|
||||||
|
lookup <- c(
|
||||||
|
`1` = "BE", `2` = "DK", `3` = "DE", `4` = "GR", `5` = "ES",
|
||||||
|
`6` = "FR", `7` = "IE", `8` = "IT", `10` = "NL", `11` = "GB",
|
||||||
|
`12` = "PT", `13` = "AT", `14` = "FI", `16` = "SE", `20` = "BG",
|
||||||
|
`21` = "CZ", `22` = "EE", `23` = "HU", `24` = "LV", `25` = "LT",
|
||||||
|
`26` = "PL", `27` = "RO", `28` = "SK", `29` = "SI", `31` = "HR",
|
||||||
|
`34` = "TR", `35` = "NO", `36` = "CH", `37` = "MT", `40` = "CY",
|
||||||
|
`45` = "IS"
|
||||||
|
)
|
||||||
|
unname(lookup[as.character(country_id)])
|
||||||
|
}
|
||||||
|
|
||||||
|
# ============================================================
|
||||||
|
# PartyFacts Linkage for CHES
|
||||||
|
# ============================================================
|
||||||
|
|
||||||
|
partyfacts_raw <- read_csv(partyfacts_path, show_col_types = FALSE)
|
||||||
|
ches_link <- partyfacts_raw %>%
|
||||||
|
filter(dataset_key == "ches") %>%
|
||||||
|
transmute(id = dataset_party_id,
|
||||||
|
country = countrycode(country, origin = 'iso3c', destination = "iso2c"),
|
||||||
|
party = partyfacts_id)
|
||||||
|
|
||||||
|
# ============================================================
|
||||||
|
# Expert Count Tables (from individual response files)
|
||||||
|
# ============================================================
|
||||||
|
|
||||||
|
cat(" Loading expert count tables from individual response files...\n")
|
||||||
|
|
||||||
|
# CHES 2024: dual lookup (party_id primary, country+name fallback for ID mismatches)
|
||||||
|
ches24_exp_raw <- read_csv(file.path(ches_dir, 'CHES_2024_expert_level.csv'), show_col_types = FALSE)
|
||||||
|
ches24_exp_by_id <- ches24_exp_raw %>%
|
||||||
|
group_by(party_id) %>%
|
||||||
|
summarise(n_experts_id = as.integer(n_distinct(id)), .groups = "drop")
|
||||||
|
ches24_exp_by_name <- ches24_exp_raw %>%
|
||||||
|
mutate(country_iso2 = countrycode(cname, origin = "country.name", destination = "iso2c")) %>%
|
||||||
|
group_by(country_iso2, party_name) %>%
|
||||||
|
summarise(n_experts_name = as.integer(n_distinct(id)), .groups = "drop")
|
||||||
|
|
||||||
|
ches_ca_expert_counts <- read_csv(file.path(ches_dir, 'CHES_CA2023_expert_level.csv'), show_col_types = FALSE) %>%
|
||||||
|
group_by(party_id) %>%
|
||||||
|
summarise(n_experts = as.integer(n_distinct(expert)), .groups = "drop")
|
||||||
|
|
||||||
|
ches_la_expert_counts <- read_csv(file.path(ches_dir, 'CHES_LA2020_expert_level.csv'), show_col_types = FALSE) %>%
|
||||||
|
group_by(party_id) %>%
|
||||||
|
summarise(n_experts = as.integer(n_distinct(expert_id)), .groups = "drop")
|
||||||
|
|
||||||
|
ches_il_expert_counts <- read_csv(file.path(ches_dir, 'CHES_IL_expert_level.csv'), show_col_types = FALSE) %>%
|
||||||
|
group_by(party_id, year) %>%
|
||||||
|
summarise(n_experts = as.integer(n_distinct(id)), .groups = "drop")
|
||||||
|
|
||||||
|
# ============================================================
|
||||||
|
# Chapel Hill Expert Survey (CHES) - 1999-2019
|
||||||
|
# ============================================================
|
||||||
|
|
||||||
|
cat(" Processing CHES 1999-2019...\n")
|
||||||
|
|
||||||
|
ches <- read_csv(file.path(ches_dir, '1999-2019_CHES_dataset_means(v3).csv'), show_col_types = FALSE) %>%
|
||||||
|
rename(country_id = country) %>%
|
||||||
|
transmute(country = ches_country_iso2(country_id),
|
||||||
|
vote = vote,
|
||||||
|
year = year,
|
||||||
|
id = as.character(party_id),
|
||||||
|
project = 'CHES',
|
||||||
|
n_experts = as.integer(expert),
|
||||||
|
lrecon_ches = lrecon/10,
|
||||||
|
galtan_ches = galtan/10) %>%
|
||||||
|
pivot_longer(cols = lrecon_ches:galtan_ches, names_to = 'var', values_to = 'val') %>%
|
||||||
|
mutate(n_scale = 10L) %>%
|
||||||
|
left_join(ches_link, by = c("id", "country")) %>%
|
||||||
|
filter(!is.na(val), !is.na(party), !is.na(country)) %>%
|
||||||
|
select(-id) %>%
|
||||||
|
mutate(type_low = ifelse(var == "lrecon_ches", "pro_welfare", "cosmopolitan"),
|
||||||
|
type_high = ifelse(var == "lrecon_ches", "pro_market", "traditional"))
|
||||||
|
|
||||||
|
# ============================================================
|
||||||
|
# CHES 2024 Update
|
||||||
|
# ============================================================
|
||||||
|
|
||||||
|
cat(" Processing CHES 2024...\n")
|
||||||
|
|
||||||
|
# Country code lookup for CHES 2024 format
|
||||||
|
country_lookup <- c(
|
||||||
|
"be" = "BE", "dk" = "DK", "ge" = "DE", "gr" = "GR", "esp" = "ES",
|
||||||
|
"fr" = "FR", "irl" = "IE", "it" = "IT", "nl" = "NL", "uk" = "GB",
|
||||||
|
"por" = "PT", "aus" = "AT", "fin" = "FI", "sv" = "SE", "bul" = "BG",
|
||||||
|
"cz" = "CZ", "est" = "EE", "hun" = "HU", "lat" = "LV", "lith" = "LT",
|
||||||
|
"pol" = "PL", "rom" = "RO", "slo" = "SK", "sle" = "SI", "cro" = "HR",
|
||||||
|
"tur" = "TR", "nor" = "NO", "swi" = "CH", "mal" = "MT", "cyp" = "CY",
|
||||||
|
"ice" = "IS"
|
||||||
|
)
|
||||||
|
|
||||||
|
convert_country_codes <- function(codes) {
|
||||||
|
numeric_result <- ches2024_country_iso2(codes)
|
||||||
|
result <- country_lookup[codes]
|
||||||
|
result[is.na(result)] <- numeric_result[is.na(result)]
|
||||||
|
result[is.na(result)] <- codes[is.na(result)]
|
||||||
|
return(unname(result))
|
||||||
|
}
|
||||||
|
|
||||||
|
ches24 <- read_csv(file.path(ches_dir, 'CHES_2024_final_v2.csv'), show_col_types = FALSE) %>%
|
||||||
|
mutate(country_iso2 = convert_country_codes(country)) %>%
|
||||||
|
left_join(ches24_exp_by_id, by = "party_id") %>%
|
||||||
|
left_join(ches24_exp_by_name, by = c("country_iso2", "party" = "party_name")) %>%
|
||||||
|
transmute(country = country_iso2,
|
||||||
|
vote = vote,
|
||||||
|
year = 2024,
|
||||||
|
id = as.character(party_id),
|
||||||
|
project = 'CHES',
|
||||||
|
n_experts = coalesce(n_experts_id, n_experts_name),
|
||||||
|
lrecon_ches = lrecon/10,
|
||||||
|
galtan_ches = galtan/10) %>%
|
||||||
|
pivot_longer(cols = lrecon_ches:galtan_ches, names_to = 'var', values_to = 'val') %>%
|
||||||
|
mutate(n_scale = 10L) %>%
|
||||||
|
left_join(ches_link, by = c("id", "country")) %>%
|
||||||
|
filter(!is.na(val), !is.na(party), !is.na(country)) %>%
|
||||||
|
select(-id) %>%
|
||||||
|
mutate(type_low = ifelse(var == "lrecon_ches", "pro_welfare", "cosmopolitan"),
|
||||||
|
type_high = ifelse(var == "lrecon_ches", "pro_market", "traditional"))
|
||||||
|
|
||||||
|
ches <- bind_rows(ches, ches24)
|
||||||
|
|
||||||
|
# ============================================================
|
||||||
|
# CHES Canada 2023
|
||||||
|
# ============================================================
|
||||||
|
|
||||||
|
cat(" Processing CHES Canada 2023...\n")
|
||||||
|
|
||||||
|
ches_ca <- read_csv(file.path(ches_dir, 'CHES_CA2023.csv'), show_col_types = FALSE) %>%
|
||||||
|
filter(!is.na(partyfacts_id)) %>%
|
||||||
|
left_join(ches_ca_expert_counts, by = "party_id") %>%
|
||||||
|
transmute(country = "CA",
|
||||||
|
year = 2023,
|
||||||
|
party = partyfacts_id,
|
||||||
|
project = 'CHES',
|
||||||
|
n_experts = n_experts,
|
||||||
|
lrecon_ches = lrecon/10,
|
||||||
|
galtan_ches = galtan/10) %>%
|
||||||
|
pivot_longer(cols = lrecon_ches:galtan_ches, names_to = 'var', values_to = 'val') %>%
|
||||||
|
mutate(n_scale = 10L) %>%
|
||||||
|
filter(!is.na(val), !is.na(party)) %>%
|
||||||
|
mutate(type_low = ifelse(var == "lrecon_ches", "pro_welfare", "cosmopolitan"),
|
||||||
|
type_high = ifelse(var == "lrecon_ches", "pro_market", "traditional"))
|
||||||
|
|
||||||
|
ches <- bind_rows(ches, ches_ca)
|
||||||
|
cat(sprintf(" CHES Canada: %d observations\n", nrow(ches_ca)))
|
||||||
|
|
||||||
|
# ============================================================
|
||||||
|
# CHES Latin America 2020
|
||||||
|
# ============================================================
|
||||||
|
|
||||||
|
cat(" Processing CHES Latin America 2020...\n")
|
||||||
|
|
||||||
|
ches_la_link <- partyfacts_raw %>%
|
||||||
|
filter(dataset_key == "ches") %>%
|
||||||
|
transmute(id = as.character(dataset_party_id),
|
||||||
|
country = countrycode(country, origin = "iso3c", destination = "iso2c"),
|
||||||
|
party = partyfacts_id)
|
||||||
|
|
||||||
|
ches_la <- read_csv(file.path(ches_dir, 'ches_la_2020_aggregate_level_v01.csv'), show_col_types = FALSE) %>%
|
||||||
|
left_join(ches_la_expert_counts, by = "party_id") %>%
|
||||||
|
transmute(country = toupper(country_abb),
|
||||||
|
year = 2020,
|
||||||
|
id = as.character(party_id),
|
||||||
|
project = 'CHES',
|
||||||
|
n_experts = n_experts,
|
||||||
|
lrecon_ches = lrecon/10,
|
||||||
|
galtan_ches = galtan/10) %>%
|
||||||
|
pivot_longer(cols = lrecon_ches:galtan_ches, names_to = 'var', values_to = 'val') %>%
|
||||||
|
mutate(n_scale = 10L) %>%
|
||||||
|
left_join(ches_la_link, by = c("id", "country")) %>%
|
||||||
|
filter(!is.na(val), !is.na(party), !is.na(country)) %>%
|
||||||
|
select(-id) %>%
|
||||||
|
mutate(type_low = as.character(ifelse(var == "lrecon_ches", "pro_welfare", "cosmopolitan")),
|
||||||
|
type_high = as.character(ifelse(var == "lrecon_ches", "pro_market", "traditional")))
|
||||||
|
|
||||||
|
ches <- bind_rows(ches, ches_la)
|
||||||
|
cat(sprintf(" CHES Latin America: %d observations\n", nrow(ches_la)))
|
||||||
|
|
||||||
|
# ============================================================
|
||||||
|
# CHES Israel 2021-2022
|
||||||
|
# ============================================================
|
||||||
|
|
||||||
|
cat(" Processing CHES Israel 2021-2022...\n")
|
||||||
|
|
||||||
|
ches_il_link <- partyfacts_raw %>%
|
||||||
|
filter(dataset_key == "ches") %>%
|
||||||
|
transmute(id = as.character(dataset_party_id),
|
||||||
|
country = countrycode(country, origin = "iso3c", destination = "iso2c"),
|
||||||
|
party = partyfacts_id)
|
||||||
|
|
||||||
|
ches_il <- read_csv(file.path(ches_dir, 'CHES_ISRAEL_means_2021_2022.csv'), show_col_types = FALSE) %>%
|
||||||
|
left_join(ches_il_expert_counts, by = c("party_id", "year")) %>%
|
||||||
|
transmute(country = "IL",
|
||||||
|
year = year,
|
||||||
|
id = as.character(party_id),
|
||||||
|
project = 'CHES',
|
||||||
|
n_experts = n_experts,
|
||||||
|
lrecon_ches = lrecon/10,
|
||||||
|
galtan_ches = galtan/10) %>%
|
||||||
|
pivot_longer(cols = lrecon_ches:galtan_ches, names_to = 'var', values_to = 'val') %>%
|
||||||
|
mutate(n_scale = 10L) %>%
|
||||||
|
left_join(ches_il_link, by = c("id", "country")) %>%
|
||||||
|
filter(!is.na(val), !is.na(party), !is.na(country)) %>%
|
||||||
|
select(-id) %>%
|
||||||
|
mutate(type_low = ifelse(var == "lrecon_ches", "pro_welfare", "cosmopolitan"),
|
||||||
|
type_high = ifelse(var == "lrecon_ches", "pro_market", "traditional"))
|
||||||
|
|
||||||
|
ches <- bind_rows(ches, ches_il)
|
||||||
|
cat(sprintf(" CHES Israel: %d observations\n", nrow(ches_il)))
|
||||||
|
|
||||||
|
cat(sprintf(" CHES total: %d observations\n", nrow(ches)))
|
||||||
|
|
||||||
|
# ============================================================
|
||||||
|
# CHES General Left-Right (for anchoring)
|
||||||
|
# ============================================================
|
||||||
|
|
||||||
|
cat(" Processing CHES LR anchoring data...\n")
|
||||||
|
|
||||||
|
ches_lr <- read_csv(file.path(ches_dir, '1999-2019_CHES_dataset_means(v3).csv'), show_col_types = FALSE) %>%
|
||||||
|
rename(country_id = country) %>%
|
||||||
|
transmute(country = ches_country_iso2(country_id),
|
||||||
|
vote = vote,
|
||||||
|
year = year,
|
||||||
|
id = as.character(party_id),
|
||||||
|
project = 'CHES',
|
||||||
|
n_experts = as.integer(expert),
|
||||||
|
val = lrgen/10,
|
||||||
|
var = 'lr_ches',
|
||||||
|
n_scale = 10L) %>%
|
||||||
|
left_join(ches_link, by = c("id", "country")) %>%
|
||||||
|
filter(!is.na(val), !is.na(party), !is.na(country)) %>%
|
||||||
|
select(-id)
|
||||||
|
|
||||||
|
ches24_lr <- read_csv(file.path(ches_dir, 'CHES_2024_final_v2.csv'), show_col_types = FALSE) %>%
|
||||||
|
mutate(country_iso2 = convert_country_codes(country)) %>%
|
||||||
|
left_join(ches24_exp_by_id, by = "party_id") %>%
|
||||||
|
left_join(ches24_exp_by_name, by = c("country_iso2", "party" = "party_name")) %>%
|
||||||
|
transmute(country = country_iso2,
|
||||||
|
vote = vote,
|
||||||
|
year = 2024,
|
||||||
|
id = as.character(party_id),
|
||||||
|
project = 'CHES',
|
||||||
|
n_experts = coalesce(n_experts_id, n_experts_name),
|
||||||
|
val = lrgen/10,
|
||||||
|
var = 'lr_ches',
|
||||||
|
n_scale = 10L) %>%
|
||||||
|
left_join(ches_link, by = c("id", "country")) %>%
|
||||||
|
filter(!is.na(val), !is.na(party), !is.na(country)) %>%
|
||||||
|
select(-id)
|
||||||
|
|
||||||
|
# CHES Canada LR
|
||||||
|
ches_ca_lr <- read_csv(file.path(ches_dir, 'CHES_CA2023.csv'), show_col_types = FALSE) %>%
|
||||||
|
filter(!is.na(partyfacts_id)) %>%
|
||||||
|
left_join(ches_ca_expert_counts, by = "party_id") %>%
|
||||||
|
transmute(country = "CA",
|
||||||
|
year = 2023,
|
||||||
|
party = partyfacts_id,
|
||||||
|
project = 'CHES',
|
||||||
|
n_experts = n_experts,
|
||||||
|
val = lrgen/10,
|
||||||
|
var = 'lr_ches',
|
||||||
|
n_scale = 10L) %>%
|
||||||
|
filter(!is.na(val), !is.na(party))
|
||||||
|
|
||||||
|
# CHES Latin America LR
|
||||||
|
ches_la_lr <- read_csv(file.path(ches_dir, 'ches_la_2020_aggregate_level_v01.csv'), show_col_types = FALSE) %>%
|
||||||
|
left_join(ches_la_expert_counts, by = "party_id") %>%
|
||||||
|
transmute(country = toupper(country_abb),
|
||||||
|
year = 2020,
|
||||||
|
id = as.character(party_id),
|
||||||
|
project = 'CHES',
|
||||||
|
n_experts = n_experts,
|
||||||
|
val = lrgen/10,
|
||||||
|
var = 'lr_ches',
|
||||||
|
n_scale = 10L) %>%
|
||||||
|
left_join(ches_la_link, by = c("id", "country")) %>%
|
||||||
|
filter(!is.na(val), !is.na(party), !is.na(country)) %>%
|
||||||
|
select(-id)
|
||||||
|
|
||||||
|
# CHES Israel LR
|
||||||
|
ches_il_lr <- read_csv(file.path(ches_dir, 'CHES_ISRAEL_means_2021_2022.csv'), show_col_types = FALSE) %>%
|
||||||
|
left_join(ches_il_expert_counts, by = c("party_id", "year")) %>%
|
||||||
|
transmute(country = "IL",
|
||||||
|
year = year,
|
||||||
|
id = as.character(party_id),
|
||||||
|
project = 'CHES',
|
||||||
|
n_experts = n_experts,
|
||||||
|
val = lrgen/10,
|
||||||
|
var = 'lr_ches',
|
||||||
|
n_scale = 10L) %>%
|
||||||
|
left_join(ches_il_link, by = c("id", "country")) %>%
|
||||||
|
filter(!is.na(val), !is.na(party), !is.na(country)) %>%
|
||||||
|
select(-id)
|
||||||
|
|
||||||
|
ches_lr <- bind_rows(ches_lr, ches24_lr, ches_ca_lr, ches_la_lr, ches_il_lr)
|
||||||
|
|
||||||
|
# ============================================================
|
||||||
|
# V-Party Dataset (V5: expanded to 7 variables)
|
||||||
|
# ============================================================
|
||||||
|
|
||||||
|
cat(" Processing V-Party...\n")
|
||||||
|
|
||||||
|
vparty_raw <- readRDS(file.path(vparty_dir, 'V-Dem-CPD-Party-V2.rds'))
|
||||||
|
|
||||||
|
# Economic 1: v2pariglef_osp (0-6 scale, higher = more right, NO reverse)
|
||||||
|
vparty_econ1 <- vparty_raw %>%
|
||||||
|
transmute(
|
||||||
|
country = countrycode(country_name, origin = "country.name", destination = "iso2c"),
|
||||||
|
year = year,
|
||||||
|
party = pf_party_id,
|
||||||
|
project = "V-Party",
|
||||||
|
n_experts = as.integer(v2pariglef_nr),
|
||||||
|
val = v2pariglef_osp / 6,
|
||||||
|
val_int = as.integer(round(v2pariglef_osp)),
|
||||||
|
n_scale = 6L,
|
||||||
|
var = "lrecon_vparty",
|
||||||
|
type_low = "pro_welfare",
|
||||||
|
type_high = "pro_market"
|
||||||
|
) %>%
|
||||||
|
na.omit()
|
||||||
|
|
||||||
|
# Economic 2 (NEW): v2pawelf_osp (0-5 scale, higher = more welfare = LEFT, REVERSE)
|
||||||
|
vparty_econ2 <- vparty_raw %>%
|
||||||
|
transmute(
|
||||||
|
country = countrycode(country_name, origin = "country.name", destination = "iso2c"),
|
||||||
|
year = year,
|
||||||
|
party = pf_party_id,
|
||||||
|
project = "V-Party",
|
||||||
|
n_experts = as.integer(v2pawelf_nr),
|
||||||
|
val = 1 - v2pawelf_osp / 5,
|
||||||
|
val_int = 5L - as.integer(round(v2pawelf_osp)),
|
||||||
|
n_scale = 5L,
|
||||||
|
var = "welf_vparty",
|
||||||
|
type_low = "pro_welfare",
|
||||||
|
type_high = "pro_market"
|
||||||
|
) %>%
|
||||||
|
na.omit()
|
||||||
|
|
||||||
|
# Cultural 1 (NEW): v2paimmig_osp (0-4 scale, higher = more pro-immigration = GAL, REVERSE)
|
||||||
|
vparty_cult1 <- vparty_raw %>%
|
||||||
|
transmute(
|
||||||
|
country = countrycode(country_name, origin = "country.name", destination = "iso2c"),
|
||||||
|
year = year,
|
||||||
|
party = pf_party_id,
|
||||||
|
project = "V-Party",
|
||||||
|
n_experts = as.integer(v2paimmig_nr),
|
||||||
|
val = 1 - v2paimmig_osp / 4,
|
||||||
|
val_int = 4L - as.integer(round(v2paimmig_osp)),
|
||||||
|
n_scale = 4L,
|
||||||
|
var = "immig_vparty",
|
||||||
|
type_low = "cosmopolitan",
|
||||||
|
type_high = "traditional"
|
||||||
|
) %>%
|
||||||
|
na.omit()
|
||||||
|
|
||||||
|
# Cultural 2 (NEW): v2palgbt_osp (0-4 scale, higher = more pro-LGBT = GAL, REVERSE)
|
||||||
|
vparty_cult2 <- vparty_raw %>%
|
||||||
|
transmute(
|
||||||
|
country = countrycode(country_name, origin = "country.name", destination = "iso2c"),
|
||||||
|
year = year,
|
||||||
|
party = pf_party_id,
|
||||||
|
project = "V-Party",
|
||||||
|
n_experts = as.integer(v2palgbt_nr),
|
||||||
|
val = 1 - v2palgbt_osp / 4,
|
||||||
|
val_int = 4L - as.integer(round(v2palgbt_osp)),
|
||||||
|
n_scale = 4L,
|
||||||
|
var = "lgbt_vparty",
|
||||||
|
type_low = "cosmopolitan",
|
||||||
|
type_high = "traditional"
|
||||||
|
) %>%
|
||||||
|
na.omit()
|
||||||
|
|
||||||
|
# Cultural 3 (NEW): v2paculsup_osp (0-4 scale, higher = less cultural superiority = GAL, REVERSE)
|
||||||
|
vparty_cult3 <- vparty_raw %>%
|
||||||
|
transmute(
|
||||||
|
country = countrycode(country_name, origin = "country.name", destination = "iso2c"),
|
||||||
|
year = year,
|
||||||
|
party = pf_party_id,
|
||||||
|
project = "V-Party",
|
||||||
|
n_experts = as.integer(v2paculsup_nr),
|
||||||
|
val = 1 - v2paculsup_osp / 4,
|
||||||
|
val_int = 4L - as.integer(round(v2paculsup_osp)),
|
||||||
|
n_scale = 4L,
|
||||||
|
var = "culsup_vparty",
|
||||||
|
type_low = "cosmopolitan",
|
||||||
|
type_high = "traditional"
|
||||||
|
) %>%
|
||||||
|
na.omit()
|
||||||
|
|
||||||
|
# Cultural 4 (NEW): v2parelig_osp (0-4 scale, higher = less religious = GAL, REVERSE)
|
||||||
|
vparty_cult4 <- vparty_raw %>%
|
||||||
|
transmute(
|
||||||
|
country = countrycode(country_name, origin = "country.name", destination = "iso2c"),
|
||||||
|
year = year,
|
||||||
|
party = pf_party_id,
|
||||||
|
project = "V-Party",
|
||||||
|
n_experts = as.integer(v2parelig_nr),
|
||||||
|
val = 1 - v2parelig_osp / 4,
|
||||||
|
val_int = 4L - as.integer(round(v2parelig_osp)),
|
||||||
|
n_scale = 4L,
|
||||||
|
var = "relig_vparty",
|
||||||
|
type_low = "cosmopolitan",
|
||||||
|
type_high = "traditional"
|
||||||
|
) %>%
|
||||||
|
na.omit()
|
||||||
|
|
||||||
|
# Cultural 5 (NEW): v2pagender_osp (0-4 scale, higher = more pro-gender equality = GAL, REVERSE)
|
||||||
|
vparty_cult5 <- vparty_raw %>%
|
||||||
|
transmute(
|
||||||
|
country = countrycode(country_name, origin = "country.name", destination = "iso2c"),
|
||||||
|
year = year,
|
||||||
|
party = pf_party_id,
|
||||||
|
project = "V-Party",
|
||||||
|
n_experts = as.integer(v2pagender_nr),
|
||||||
|
val = 1 - v2pagender_osp / 4,
|
||||||
|
val_int = 4L - as.integer(round(v2pagender_osp)),
|
||||||
|
n_scale = 4L,
|
||||||
|
var = "gender_vparty",
|
||||||
|
type_low = "cosmopolitan",
|
||||||
|
type_high = "traditional"
|
||||||
|
) %>%
|
||||||
|
na.omit()
|
||||||
|
|
||||||
|
vparty <- bind_rows(vparty_econ1, vparty_econ2,
|
||||||
|
vparty_cult1, vparty_cult2, vparty_cult3,
|
||||||
|
vparty_cult4, vparty_cult5)
|
||||||
|
|
||||||
|
cat(sprintf(" V-Party: %d observations (7 variables)\n", nrow(vparty)))
|
||||||
|
cat(sprintf(" lrecon: %d, welf: %d\n", nrow(vparty_econ1), nrow(vparty_econ2)))
|
||||||
|
cat(sprintf(" immig: %d, lgbt: %d, culsup: %d, relig: %d, gender: %d\n",
|
||||||
|
nrow(vparty_cult1), nrow(vparty_cult2), nrow(vparty_cult3),
|
||||||
|
nrow(vparty_cult4), nrow(vparty_cult5)))
|
||||||
|
|
||||||
|
# ============================================================
|
||||||
|
# POPPA Dataset
|
||||||
|
# ============================================================
|
||||||
|
|
||||||
|
cat(" Processing POPPA...\n")
|
||||||
|
|
||||||
|
poppa <- readRDS(file.path(poppa_dir, 'poppa_integrated_v2.rds')) %>%
|
||||||
|
transmute(country = countrycode(country_short, origin = "iso3c", destination = "iso2c"),
|
||||||
|
party = partyfacts_id,
|
||||||
|
val = lrecon/10,
|
||||||
|
var = "lrecon_poppa",
|
||||||
|
type_low = "pro_welfare",
|
||||||
|
type_high = "pro_market",
|
||||||
|
n_experts = as.integer(n_experts),
|
||||||
|
n_scale = 10L,
|
||||||
|
year = as.numeric(sub(".*-\\s*(\\d+)", "\\1", wave)),
|
||||||
|
project = "POPPA") %>%
|
||||||
|
na.omit()
|
||||||
|
|
||||||
|
cat(sprintf(" POPPA: %d observations\n", nrow(poppa)))
|
||||||
|
|
||||||
|
# POPPA General LR
|
||||||
|
poppa_lr <- readRDS(file.path(poppa_dir, 'poppa_integrated_v2.rds')) %>%
|
||||||
|
transmute(country = countrycode(country_short, origin = "iso3c", destination = "iso2c"),
|
||||||
|
party = partyfacts_id,
|
||||||
|
val = lroverall/10,
|
||||||
|
var = "lr_poppa",
|
||||||
|
n_experts = as.integer(n_experts),
|
||||||
|
n_scale = 10L,
|
||||||
|
year = as.numeric(sub(".*-\\s*(\\d+)", "\\1", wave)),
|
||||||
|
project = "POPPA") %>%
|
||||||
|
na.omit()
|
||||||
|
|
||||||
|
# ============================================================
|
||||||
|
# GPS (Norris) Survey
|
||||||
|
# ============================================================
|
||||||
|
|
||||||
|
cat(" Processing GPS...\n")
|
||||||
|
|
||||||
|
gps <- read.delim(file.path(gps_dir, "Global Party Survey by Party SPSS V2_1_Apr_2020-2.tab")) %>%
|
||||||
|
transmute(n_experts = as.integer(Experts),
|
||||||
|
lrecon_gps = as.numeric(V4_Scale)/10,
|
||||||
|
libcon_gps = as.numeric(V6_Scale)/10,
|
||||||
|
party = ID_PartyFacts,
|
||||||
|
country = countrycode(ifelse(ISO == "MAC", "MKD", ISO), origin = "iso3c", destination = "iso2c"),
|
||||||
|
year = 2019,
|
||||||
|
n_scale = 10L,
|
||||||
|
project = "GPS") %>%
|
||||||
|
pivot_longer(cols = lrecon_gps:libcon_gps, names_to = 'var', values_to = 'val') %>%
|
||||||
|
mutate(type_low = ifelse(var == "lrecon_gps", "pro_welfare", "cosmopolitan"),
|
||||||
|
type_high = ifelse(var == "lrecon_gps", "pro_market", "traditional")) %>%
|
||||||
|
na.omit()
|
||||||
|
|
||||||
|
cat(sprintf(" GPS: %d observations\n", nrow(gps)))
|
||||||
|
|
||||||
|
# ============================================================
|
||||||
|
# Combine Expert Data
|
||||||
|
# ============================================================
|
||||||
|
|
||||||
|
cat(" Combining expert surveys...\n")
|
||||||
|
|
||||||
|
expert_raw <- select(ches, -vote) %>%
|
||||||
|
bind_rows(vparty) %>%
|
||||||
|
bind_rows(gps) %>%
|
||||||
|
bind_rows(poppa) %>%
|
||||||
|
unique() %>%
|
||||||
|
arrange(country, party, year, var) %>%
|
||||||
|
filter(!is.na(val), !is.na(party), !is.na(country), !is.na(var))
|
||||||
|
|
||||||
|
# Compute val_int for datasets that don't have it pre-computed
|
||||||
|
# V-Party already has val_int; CHES/GPS/POPPA need it computed from val * n_scale
|
||||||
|
expert_raw <- expert_raw %>%
|
||||||
|
mutate(
|
||||||
|
val_int = ifelse(is.na(val_int), as.integer(round(val * n_scale)), val_int),
|
||||||
|
val_int = pmin(pmax(val_int, 0L), n_scale)
|
||||||
|
)
|
||||||
|
|
||||||
|
# Boundary adjustments for continuous val (avoid exact 0 or 1 for Stan prior means)
|
||||||
|
expert_raw <- expert_raw %>%
|
||||||
|
mutate(
|
||||||
|
val = case_when(
|
||||||
|
val == 0 ~ val + 1e-4,
|
||||||
|
val == 1 ~ val - 1e-4,
|
||||||
|
TRUE ~ val
|
||||||
|
))
|
||||||
|
|
||||||
|
# ============================================================
|
||||||
|
# Combine LR Data
|
||||||
|
# ============================================================
|
||||||
|
|
||||||
|
lr_data_raw <- ches_lr %>%
|
||||||
|
bind_rows(poppa_lr) %>%
|
||||||
|
select(-any_of("vote"))
|
||||||
|
|
||||||
|
# Boundary adjustments for continuous val (avoid exact 0 or 1)
|
||||||
|
lr_data_raw <- lr_data_raw %>%
|
||||||
|
mutate(
|
||||||
|
val = case_when(
|
||||||
|
val == 0 ~ val + 1e-4,
|
||||||
|
val == 1 ~ val - 1e-4,
|
||||||
|
TRUE ~ val
|
||||||
|
))
|
||||||
|
|
||||||
|
# Compute val_int for LR data
|
||||||
|
lr_data_raw <- lr_data_raw %>%
|
||||||
|
mutate(
|
||||||
|
val_int = as.integer(round(val * n_scale)),
|
||||||
|
val_int = pmin(pmax(val_int, 0L), n_scale)
|
||||||
|
)
|
||||||
|
|
||||||
|
# ============================================================
|
||||||
|
# Write Outputs
|
||||||
|
# ============================================================
|
||||||
|
|
||||||
|
write_csv(expert_raw, "expert_raw.csv")
|
||||||
|
write_csv(lr_data_raw, "lr_data_raw.csv")
|
||||||
|
|
||||||
|
cat(sprintf("\nOutputs written:\n"))
|
||||||
|
cat(sprintf(" expert_raw.csv: %d rows\n", nrow(expert_raw)))
|
||||||
|
cat(sprintf(" lr_data_raw.csv: %d rows\n", nrow(lr_data_raw)))
|
||||||
|
|
||||||
|
cat("\n Expert data by source:\n")
|
||||||
|
expert_raw %>%
|
||||||
|
group_by(project) %>%
|
||||||
|
summarise(n = n(), .groups = "drop") %>%
|
||||||
|
print()
|
||||||
|
|
||||||
|
cat("\n New columns check:\n")
|
||||||
|
cat(sprintf(" val_int range: %d - %d\n", min(expert_raw$val_int), max(expert_raw$val_int)))
|
||||||
|
cat(sprintf(" n_scale values: %s\n", paste(sort(unique(expert_raw$n_scale)), collapse = ", ")))
|
||||||
|
cat(sprintf(" n_experts non-NA: %d / %d\n", sum(!is.na(expert_raw$n_experts)), nrow(expert_raw)))
|
||||||
@@ -0,0 +1,179 @@
|
|||||||
|
# ============================================================
|
||||||
|
# process_manifesto.R - Manifesto Project Data Processing
|
||||||
|
# ============================================================
|
||||||
|
# Processes Manifesto Project data for the two-dimensional party-position model
|
||||||
|
# Input: $PARTY2D_RAW_DATA_DIR/manifesto/MPDataset_MPDS2025a.csv
|
||||||
|
# Output: manifesto_data.csv
|
||||||
|
# ============================================================
|
||||||
|
|
||||||
|
library(tidyverse)
|
||||||
|
library(countrycode)
|
||||||
|
library(purrr)
|
||||||
|
|
||||||
|
# Set working directory (works both in RStudio and command line)
|
||||||
|
if (interactive() && requireNamespace("rstudioapi", quietly = TRUE)) {
|
||||||
|
try(setwd(dirname(rstudioapi::getActiveDocumentContext()$path)), silent = TRUE)
|
||||||
|
}
|
||||||
|
|
||||||
|
cat("Processing Manifesto Project data...\n")
|
||||||
|
|
||||||
|
raw_data_dir <- Sys.getenv(
|
||||||
|
"PARTY2D_RAW_DATA_DIR",
|
||||||
|
unset = file.path("..", "..", "_local", "raw")
|
||||||
|
)
|
||||||
|
manifesto_raw_path <- file.path(raw_data_dir, "manifesto", "MPDataset_MPDS2025a.csv")
|
||||||
|
partyfacts_path <- file.path(raw_data_dir, "partyfacts", "partyfacts-external-parties.csv")
|
||||||
|
|
||||||
|
# ============================================================
|
||||||
|
# PartyFacts Linkage
|
||||||
|
# ============================================================
|
||||||
|
|
||||||
|
partyfacts_raw <- read_csv(partyfacts_path, show_col_types = FALSE)
|
||||||
|
manifesto_link <- partyfacts_raw %>%
|
||||||
|
filter(dataset_key == "manifesto") %>%
|
||||||
|
transmute(id = dataset_party_id,
|
||||||
|
country = countrycode(country, origin = 'iso3c', destination = "iso2c"),
|
||||||
|
party = partyfacts_id,
|
||||||
|
party = ifelse(party == 622, 604, party))
|
||||||
|
|
||||||
|
# ============================================================
|
||||||
|
# Load Manifesto Data
|
||||||
|
# ============================================================
|
||||||
|
|
||||||
|
manifesto_data <- read_csv(manifesto_raw_path, show_col_types = FALSE)
|
||||||
|
|
||||||
|
# ============================================================
|
||||||
|
# CMP Code Mapping to 4 Dimensions
|
||||||
|
# ============================================================
|
||||||
|
|
||||||
|
vars <- tribble(
|
||||||
|
~type, ~subtype, ~per_var, ~stance, ~label,
|
||||||
|
# pro_market
|
||||||
|
"pro_market", "Market Regulation", "per401", "Positive", "Free Market Economy",
|
||||||
|
"pro_market", "Economic Liberalization","per402", "Positive", "Incentives: Positive",
|
||||||
|
"pro_market", "Market Regulation", "per407", "Positive", "Protectionism: Negative",
|
||||||
|
"pro_market", "Economic Liberalization","per414", "Positive", "Economic Orthodoxy",
|
||||||
|
"pro_market", "Economic Liberalization","per505", "Positive", "Welfare State Limitation",
|
||||||
|
"pro_market", "Economic Liberalization","per507", "Positive", "Education Limitation",
|
||||||
|
"pro_market", "Economic Liberalization","per702", "Positive", "Labour Groups: Negative",
|
||||||
|
"pro_market", "Market Regulation", "per406", "Negative", "Protectionism: Positive",
|
||||||
|
"pro_market", "Market Regulation", "per412", "Negative", "Controlled Economy",
|
||||||
|
"pro_market", "Economic Liberalization","per504", "Negative", "Welfare State Expansion",
|
||||||
|
# pro_welfare
|
||||||
|
"pro_welfare", "Economic Intervention", "per403", "Positive", "Market Regulation",
|
||||||
|
"pro_welfare", "Economic Intervention", "per404", "Positive", "Economic Planning",
|
||||||
|
"pro_welfare", "Economic Intervention", "per412", "Positive", "Controlled Economy",
|
||||||
|
"pro_welfare", "Economic Intervention", "per413", "Positive", "Nationalisation",
|
||||||
|
"pro_welfare", "Social Services", "per504", "Positive", "Welfare State Expansion",
|
||||||
|
"pro_welfare", "Social Services", "per506", "Positive", "Education Expansion",
|
||||||
|
"pro_welfare", "Economic Intervention", "per701", "Positive", "Labour Groups: Positive",
|
||||||
|
"pro_welfare", "Economic Intervention", "per401", "Negative", "Free Market Economy",
|
||||||
|
"pro_welfare", "Social Services", "per505", "Negative", "Welfare State Limitation",
|
||||||
|
# cosmopolitan
|
||||||
|
"cosmopolitan", "Internationalism", "per107", "Positive", "Internationalism: Positive",
|
||||||
|
"cosmopolitan", "Internationalism", "per108", "Positive", "European Community/Union: Positive",
|
||||||
|
"cosmopolitan", "Multiculturalism", "per607", "Positive", "Multiculturalism: Positive",
|
||||||
|
"cosmopolitan", "Multiculturalism", "per201", "Positive", "Freedom and Human Rights",
|
||||||
|
"cosmopolitan", "Multiculturalism", "per604", "Positive", "traditional Morality: Negative",
|
||||||
|
"cosmopolitan", "Internationalism", "per109", "Negative", "Internationalism: Negative",
|
||||||
|
"cosmopolitan", "Multiculturalism", "per601", "Negative", "National Way of Life: Positive",
|
||||||
|
# traditional
|
||||||
|
"traditional", "National Identity", "per109", "Positive", "Internationalism: Negative",
|
||||||
|
"traditional", "Conservative Morality", "per110", "Positive", "European Community/Union: Negative",
|
||||||
|
"traditional", "National Identity", "per601", "Positive", "National Way of Life: Positive",
|
||||||
|
"traditional", "Conservative Morality", "per603", "Positive", "traditional Morality: Positive",
|
||||||
|
"traditional", "Conservative Morality", "per608", "Positive", "Multiculturalism: Negative",
|
||||||
|
"traditional", "Conservative Morality", "per605", "Positive", "Law and Order: Positive",
|
||||||
|
"traditional", "National Identity", "per107", "Negative", "Internationalism: Positive",
|
||||||
|
"traditional", "Conservative Morality", "per607", "Negative", "Multiculturalism: Positive"
|
||||||
|
)
|
||||||
|
|
||||||
|
# ============================================================
|
||||||
|
# Process Manifesto Data
|
||||||
|
# ============================================================
|
||||||
|
|
||||||
|
manifesto <- vars %>%
|
||||||
|
pmap_dfr(~ manifesto_data %>%
|
||||||
|
transmute(country = countrycode(countryname, origin = 'country.name', destination = 'iso2c'),
|
||||||
|
year = as.numeric(format(as.Date(edate, format = "%d/%m/%Y"), "%Y")),
|
||||||
|
id = as.character(party),
|
||||||
|
count = round(.data[[..3]]),
|
||||||
|
var = ..3,
|
||||||
|
label = ..5,
|
||||||
|
type = ..1,
|
||||||
|
subtype = ..2,
|
||||||
|
stance = ..4,
|
||||||
|
project = 'Manifesto Project') %>%
|
||||||
|
left_join(manifesto_link, by = c("id", "country")) %>%
|
||||||
|
select(-id)) %>%
|
||||||
|
group_by(party, country, year, subtype) %>%
|
||||||
|
summarise(
|
||||||
|
positive = sum(count[stance == "Positive"], na.rm = TRUE),
|
||||||
|
sample = sum(count, na.rm = TRUE),
|
||||||
|
type = first(type),
|
||||||
|
project = first(project),
|
||||||
|
.groups = "drop"
|
||||||
|
) %>%
|
||||||
|
na.omit() %>%
|
||||||
|
rename(var = subtype) %>%
|
||||||
|
# Convert to bipolar bridge structure (type_high/type_low)
|
||||||
|
mutate(
|
||||||
|
type_high = case_when(
|
||||||
|
type == "pro_welfare" ~ "pro_welfare",
|
||||||
|
type == "pro_market" ~ "pro_market",
|
||||||
|
type == "cosmopolitan" ~ "cosmopolitan",
|
||||||
|
type == "traditional" ~ "traditional"
|
||||||
|
),
|
||||||
|
type_low = case_when(
|
||||||
|
type %in% c("pro_welfare", "pro_market") ~ ifelse(type == "pro_welfare", "pro_market", "pro_welfare"),
|
||||||
|
type %in% c("cosmopolitan", "traditional") ~ ifelse(type == "cosmopolitan", "traditional", "cosmopolitan")
|
||||||
|
)
|
||||||
|
) %>%
|
||||||
|
select(-type) %>%
|
||||||
|
# Add _manifesto suffix to variable names
|
||||||
|
mutate(var = paste0(tolower(gsub(" ", "_", var)), "_manifesto"))
|
||||||
|
|
||||||
|
# ============================================================
|
||||||
|
# NOTE: Temporal continuity filter moved to the data setup orchestrator.
|
||||||
|
# This allows exempting parties that appear in parliamentary data
|
||||||
|
# (parties in parliament are by definition not fringe parties)
|
||||||
|
# ============================================================
|
||||||
|
|
||||||
|
cat("Skipping temporal filter (applied in 02_build_model_inputs.R after combining with other text data)\n")
|
||||||
|
cat(sprintf(" Parties: %d\n", n_distinct(manifesto$party)))
|
||||||
|
|
||||||
|
# ============================================================
|
||||||
|
# Write Output
|
||||||
|
# ============================================================
|
||||||
|
|
||||||
|
write_csv(manifesto, "manifesto_data.csv")
|
||||||
|
cat(sprintf("Output: manifesto_data.csv (%d rows, %d parties)\n",
|
||||||
|
nrow(manifesto), n_distinct(manifesto$party)))
|
||||||
|
|
||||||
|
# ============================================================
|
||||||
|
# Election Data Extraction (vote shares)
|
||||||
|
# ============================================================
|
||||||
|
|
||||||
|
cat("\nExtracting election data (pervote)...\n")
|
||||||
|
|
||||||
|
election_data <- manifesto_data %>%
|
||||||
|
transmute(
|
||||||
|
country = countrycode(countryname, origin = 'country.name', destination = 'iso2c'),
|
||||||
|
year = as.numeric(format(as.Date(edate, format = "%d/%m/%Y"), "%Y")),
|
||||||
|
id = as.character(party),
|
||||||
|
pervote = pervote
|
||||||
|
) %>%
|
||||||
|
left_join(manifesto_link, by = c("id", "country")) %>%
|
||||||
|
select(-id) %>%
|
||||||
|
filter(!is.na(party), !is.na(pervote)) %>%
|
||||||
|
# Keep one row per (party, country, year) — take max pervote if duplicates
|
||||||
|
group_by(party, country, year) %>%
|
||||||
|
summarise(pervote = max(pervote, na.rm = TRUE), .groups = "drop") %>%
|
||||||
|
arrange(country, party, year)
|
||||||
|
|
||||||
|
write_csv(election_data, "election_data.csv")
|
||||||
|
cat(sprintf("Output: election_data.csv (%d rows, %d parties)\n",
|
||||||
|
nrow(election_data), n_distinct(election_data$party)))
|
||||||
|
|
||||||
|
# Export manifesto_link for use by other scripts
|
||||||
|
# (poldem also needs it for CMP linkage)
|
||||||
@@ -0,0 +1,399 @@
|
|||||||
|
# process_morgan.R
|
||||||
|
# Process Morgan (1976) expert party position data
|
||||||
|
#
|
||||||
|
# Source: Morgan, Michael-John (1976). "The Modelling of Governmental
|
||||||
|
# Coalition Formation: A Policy-Based Approach with Interval Measurement."
|
||||||
|
# PhD dissertation, University of Michigan.
|
||||||
|
#
|
||||||
|
# Data extracted from Appendix B.3 (Tables B.3.1-B.3.12) via OCR.
|
||||||
|
# Position scores are 25%-truncated means (midmeans) from expert surveys.
|
||||||
|
# Scale: 0-100 (left-right)
|
||||||
|
|
||||||
|
library(tidyverse)
|
||||||
|
|
||||||
|
cat("Processing Morgan (1976) expert party position data...\n")
|
||||||
|
|
||||||
|
raw_data_dir <- Sys.getenv(
|
||||||
|
"PARTY2D_RAW_DATA_DIR",
|
||||||
|
unset = file.path("..", "..", "_local", "raw")
|
||||||
|
)
|
||||||
|
morgan_raw_path <- file.path(raw_data_dir, "morgan", "morgan_positions_raw.csv")
|
||||||
|
partyfacts_path <- file.path(raw_data_dir, "partyfacts", "partyfacts-external-parties.csv")
|
||||||
|
|
||||||
|
# Load raw extracted data
|
||||||
|
morgan_raw <- read_csv(morgan_raw_path, show_col_types = FALSE)
|
||||||
|
|
||||||
|
cat(sprintf("Loaded %d party-period observations from %d countries\n",
|
||||||
|
nrow(morgan_raw), n_distinct(morgan_raw$country)))
|
||||||
|
|
||||||
|
# Load PartyFacts linkage data
|
||||||
|
partyfacts <- read_csv(partyfacts_path, show_col_types = FALSE)
|
||||||
|
|
||||||
|
# Filter to Morgan dataset entries
|
||||||
|
morgan_pf <- partyfacts %>%
|
||||||
|
filter(dataset_key == "morgan") %>%
|
||||||
|
select(country, name_short, name_english, year_first, year_last,
|
||||||
|
external_id, partyfacts_id) %>%
|
||||||
|
rename(party_abbrev_pf = name_short)
|
||||||
|
|
||||||
|
cat(sprintf("Found %d Morgan parties in PartyFacts\n", nrow(morgan_pf)))
|
||||||
|
|
||||||
|
# Map extracted abbreviations to PartyFacts abbreviations
|
||||||
|
# Some adjustments needed due to OCR/transcription differences
|
||||||
|
abbrev_map <- tribble(
|
||||||
|
~country, ~party_abbrev, ~party_abbrev_pf,
|
||||||
|
# Denmark
|
||||||
|
"DNK", "SOCd", "SOCD",
|
||||||
|
"DNK", "SOCL", "SOCL",
|
||||||
|
"DNK", "COMM", "COMM",
|
||||||
|
"DNK", "RAD", "RAD",
|
||||||
|
"DNK", "LIB", "LIB",
|
||||||
|
"DNK", "CONS", "CONS",
|
||||||
|
"DNK", "LS", "LS",
|
||||||
|
"DNK", "LC", "LC",
|
||||||
|
"DNK", "JUST", "JUST",
|
||||||
|
# Finland
|
||||||
|
"FIN", "SKDL", "SKDL",
|
||||||
|
"FIN", "SOCd", "SOCD",
|
||||||
|
"FIN", "PROG", "PROG",
|
||||||
|
"FIN", "AGR", "AGR",
|
||||||
|
"FIN", "SWPP", "SWPP",
|
||||||
|
"FIN", "CONS", "CONS",
|
||||||
|
"FIN", "NPF", "NPF",
|
||||||
|
"FIN", "PDEM", "PDEM",
|
||||||
|
"FIN", "SDWS", "SDWS",
|
||||||
|
"FIN", "CENT", "CENT",
|
||||||
|
"FIN", "FRP", "FRP",
|
||||||
|
"FIN", "LIB", "LIB",
|
||||||
|
# Iceland
|
||||||
|
"ISL", "COMM", "COMM",
|
||||||
|
"ISL", "SOCd", "SOCD",
|
||||||
|
"ISL", "PROG", "PROG",
|
||||||
|
"ISL", "LIB", "LIB",
|
||||||
|
"ISL", "INDP", "INDP",
|
||||||
|
"ISL", "CONS", "CONS",
|
||||||
|
"ISL", "LLIB", "LLIB",
|
||||||
|
# Norway
|
||||||
|
"NOR", "LAB", "LAB",
|
||||||
|
"NOR", "LIB", "LIB",
|
||||||
|
"NOR", "AGR", "AGR",
|
||||||
|
"NOR", "CONS", "CONS",
|
||||||
|
"NOR", "COMM", "COMM",
|
||||||
|
"NOR", "SOCL", "SOCL",
|
||||||
|
"NOR", "CHPP", "CHPP",
|
||||||
|
"NOR", "CENT", "CENT",
|
||||||
|
# Sweden
|
||||||
|
"SWE", "COMM", "COMM",
|
||||||
|
"SWE", "SOCd", "SOCD",
|
||||||
|
"SWE", "AGR", "AGR",
|
||||||
|
"SWE", "LIB", "LIB",
|
||||||
|
"SWE", "CONS", "CONS",
|
||||||
|
"SWE", "CENT", "CENT",
|
||||||
|
# Netherlands
|
||||||
|
"NLD", "CPN", "CPN",
|
||||||
|
"NLD", "SOCd", "SOCD",
|
||||||
|
"NLD", "RAD", "RAD",
|
||||||
|
"NLD", "KVP", "KVP",
|
||||||
|
"NLD", "CHU", "CHU",
|
||||||
|
"NLD", "LIB", "LIB",
|
||||||
|
"NLD", "ARP", "ARP",
|
||||||
|
"NLD", "SGP", "SGP",
|
||||||
|
"NLD", "NSB", "NSB",
|
||||||
|
"NLD", "PVDA", "PVDA",
|
||||||
|
"NLD", "VVD", "VVD",
|
||||||
|
"NLD", "PSP", "PSP",
|
||||||
|
"NLD", "PPR", "PPR",
|
||||||
|
"NLD", "D66", "D66",
|
||||||
|
"NLD", "DS70", "DS70",
|
||||||
|
"NLD", "GPV", "GPV",
|
||||||
|
"NLD", "BP", "BP",
|
||||||
|
# Belgium
|
||||||
|
"BEL", "COMM", "COMM",
|
||||||
|
"BEL", "POB", "POB",
|
||||||
|
"BEL", "CATH", "CATH",
|
||||||
|
"BEL", "LIB", "LIB",
|
||||||
|
"BEL", "FNAT", "FNAT",
|
||||||
|
"BEL", "REX", "REX",
|
||||||
|
"BEL", "PSB", "PSB",
|
||||||
|
"BEL", "RW", "RW",
|
||||||
|
"BEL", "PSC", "PSC",
|
||||||
|
"BEL", "FDF", "FDF",
|
||||||
|
"BEL", "VOLK", "VOLK",
|
||||||
|
"BEL", "PLP", "PLP",
|
||||||
|
# France (Fourth Republic)
|
||||||
|
"FRA", "PCF", "PCF",
|
||||||
|
"FRA", "SFIO", "SFIO",
|
||||||
|
"FRA", "MRP", "MRP",
|
||||||
|
"FRA", "RDA", "RDA",
|
||||||
|
"FRA", "UDSR", "UDSR",
|
||||||
|
"FRA", "RAD", "RAD",
|
||||||
|
"FRA", "RS", "RS",
|
||||||
|
"FRA", "RPF", "RPF",
|
||||||
|
"FRA", "AR", "AR",
|
||||||
|
"FRA", "ARS", "ARS",
|
||||||
|
"FRA", "RI", "RI",
|
||||||
|
"FRA", "CNIP", "CNIP",
|
||||||
|
"FRA", "PUS", "PUS",
|
||||||
|
"FRA", "PAYS", "PAYS",
|
||||||
|
"FRA", "AP", "AP",
|
||||||
|
"FRA", "PRL", "PRL",
|
||||||
|
"FRA", "POUJ", "POUJ",
|
||||||
|
# Weimar Germany
|
||||||
|
"DEU", "KPD", "KPD",
|
||||||
|
"DEU", "SDAP", "SDAP",
|
||||||
|
"DEU", "DDP", "DDP",
|
||||||
|
"DEU", "DZP", "DZP",
|
||||||
|
"DEU", "BVP", "BVP",
|
||||||
|
"DEU", "DVP", "DVP",
|
||||||
|
"DEU", "RDMW", "RDMW",
|
||||||
|
"DEU", "LVP", "LVP",
|
||||||
|
"DEU", "DNVP", "DNVP",
|
||||||
|
"DEU", "NAZI", "NAZI",
|
||||||
|
# Italy
|
||||||
|
"ITA", "PCI", "PCI",
|
||||||
|
"ITA", "PSIU", "PSIU",
|
||||||
|
"ITA", "PSI", "PSI",
|
||||||
|
"ITA", "PSDI", "PSDI",
|
||||||
|
"ITA", "PRI", "PRI",
|
||||||
|
"ITA", "DC", "DC",
|
||||||
|
"ITA", "PLI", "PLI",
|
||||||
|
"ITA", "MON", "MON",
|
||||||
|
"ITA", "MSI", "MSI",
|
||||||
|
# Luxembourg
|
||||||
|
"LUX", "COMM", "COMM",
|
||||||
|
"LUX", "SOCd", "SOCD",
|
||||||
|
"LUX", "CSOC", "CSOC",
|
||||||
|
"LUX", "GRPD", "GRPD",
|
||||||
|
# Israel
|
||||||
|
"ISR", "RAKA", "RAKA",
|
||||||
|
"ISR", "MAKI", "MAKI",
|
||||||
|
"ISR", "MAPM", "MAPM",
|
||||||
|
"ISR", "MADT", "MADT",
|
||||||
|
"ISR", "ADUT", "ADUT",
|
||||||
|
"ISR", "MAAR", "MAAR",
|
||||||
|
"ISR", "LAB", "LAB",
|
||||||
|
"ISR", "MAPI", "MAPI",
|
||||||
|
"ISR", "PAUG", "PAUG",
|
||||||
|
"ISR", "RAFI", "RAFI",
|
||||||
|
"ISR", "PROG", "PROG",
|
||||||
|
"ISR", "ILIB", "ILIB",
|
||||||
|
"ISR", "NRP", "NRP",
|
||||||
|
"ISR", "URF", "URF",
|
||||||
|
"ISR", "LIB", "LIB",
|
||||||
|
"ISR", "NATL", "NATL",
|
||||||
|
"ISR", "TORA", "TORA",
|
||||||
|
"ISR", "LIKD", "LIKD",
|
||||||
|
"ISR", "ZION", "ZION",
|
||||||
|
"ISR", "GHAL", "GHAL",
|
||||||
|
"ISR", "AGDT", "AGDT",
|
||||||
|
"ISR", "HRUT", "HRUT"
|
||||||
|
)
|
||||||
|
|
||||||
|
# Some parties in raw data that don't have exact matches - need special handling
|
||||||
|
# (e.g., parties that only exist in one period in PartyFacts but appear in both)
|
||||||
|
# We'll join using the period-based matching
|
||||||
|
|
||||||
|
# Expand periods to years for matching
|
||||||
|
morgan_expanded <- morgan_raw %>%
|
||||||
|
mutate(
|
||||||
|
year_start = as.integer(str_extract(period, "^\\d{4}")),
|
||||||
|
year_end = as.integer(str_extract(period, "\\d{4}$"))
|
||||||
|
)
|
||||||
|
|
||||||
|
# Join with abbreviation map
|
||||||
|
morgan_mapped <- morgan_expanded %>%
|
||||||
|
left_join(abbrev_map, by = c("country", "party_abbrev"))
|
||||||
|
|
||||||
|
# Check for unmatched abbreviations
|
||||||
|
unmatched_abbrev <- morgan_mapped %>%
|
||||||
|
filter(is.na(party_abbrev_pf)) %>%
|
||||||
|
distinct(country, party_abbrev)
|
||||||
|
|
||||||
|
if (nrow(unmatched_abbrev) > 0) {
|
||||||
|
cat("\nWarning: Unmatched abbreviations:\n")
|
||||||
|
print(unmatched_abbrev)
|
||||||
|
}
|
||||||
|
|
||||||
|
# Join with PartyFacts
|
||||||
|
morgan_joined <- morgan_mapped %>%
|
||||||
|
left_join(morgan_pf, by = c("country", "party_abbrev_pf")) %>%
|
||||||
|
# For parties with overlapping periods, use period overlap
|
||||||
|
mutate(
|
||||||
|
period_overlap = pmax(0,
|
||||||
|
pmin(year_end, year_last) - pmax(year_start, year_first) + 1)
|
||||||
|
) %>%
|
||||||
|
# Keep best match per party-period (max overlap)
|
||||||
|
group_by(country, party_abbrev, period) %>%
|
||||||
|
slice_max(period_overlap, n = 1, with_ties = FALSE) %>%
|
||||||
|
ungroup()
|
||||||
|
|
||||||
|
# Check for unmatched parties
|
||||||
|
unmatched <- morgan_joined %>%
|
||||||
|
filter(is.na(partyfacts_id)) %>%
|
||||||
|
distinct(country, party_abbrev, party_name, period)
|
||||||
|
|
||||||
|
if (nrow(unmatched) > 0) {
|
||||||
|
cat(sprintf("\n%d party-periods without PartyFacts match:\n", nrow(unmatched)))
|
||||||
|
print(unmatched)
|
||||||
|
}
|
||||||
|
|
||||||
|
# Dedup: when multiple abbreviations map to the same PF ID, keep only one
|
||||||
|
matched <- morgan_joined %>%
|
||||||
|
filter(!is.na(partyfacts_id)) %>%
|
||||||
|
group_by(country, partyfacts_id, period) %>%
|
||||||
|
slice(1) %>%
|
||||||
|
ungroup()
|
||||||
|
|
||||||
|
cat(sprintf("\nMatched %d of %d party-period observations (%.1f%%)\n",
|
||||||
|
nrow(matched), nrow(morgan_raw),
|
||||||
|
100 * nrow(matched) / nrow(morgan_raw)))
|
||||||
|
|
||||||
|
# Normalize position to [0,1] scale
|
||||||
|
# Original scale: 0-100
|
||||||
|
# Apply boundary adjustments like other expert data
|
||||||
|
eps <- 0.005
|
||||||
|
morgan_processed <- matched %>%
|
||||||
|
mutate(
|
||||||
|
# Normalize to [0,1]
|
||||||
|
lr_morgan = position / 100,
|
||||||
|
# Apply boundary adjustments
|
||||||
|
lr_morgan = case_when(
|
||||||
|
lr_morgan <= 0 ~ eps,
|
||||||
|
lr_morgan >= 1 ~ 1 - eps,
|
||||||
|
TRUE ~ lr_morgan
|
||||||
|
),
|
||||||
|
# Calculate standard error (sd / sqrt(n))
|
||||||
|
lr_morgan_se = (sd / 100) / sqrt(n_surveys),
|
||||||
|
# Set minimum SE for extreme parties (sd=0)
|
||||||
|
lr_morgan_se = pmax(lr_morgan_se, 0.01)
|
||||||
|
) %>%
|
||||||
|
select(
|
||||||
|
country,
|
||||||
|
partyfacts_id,
|
||||||
|
period,
|
||||||
|
year_start,
|
||||||
|
year_end,
|
||||||
|
party_abbrev,
|
||||||
|
party_name,
|
||||||
|
lr_morgan,
|
||||||
|
lr_morgan_se,
|
||||||
|
n_surveys
|
||||||
|
) %>%
|
||||||
|
arrange(country, year_start, lr_morgan)
|
||||||
|
|
||||||
|
# Summary statistics
|
||||||
|
cat("\nSummary of processed Morgan data:\n")
|
||||||
|
cat(sprintf(" Countries: %d\n", n_distinct(morgan_processed$country)))
|
||||||
|
cat(sprintf(" Parties: %d\n", n_distinct(morgan_processed$partyfacts_id)))
|
||||||
|
cat(sprintf(" Observations: %d\n", nrow(morgan_processed)))
|
||||||
|
|
||||||
|
# Distribution of positions
|
||||||
|
cat("\nPosition distribution:\n")
|
||||||
|
print(summary(morgan_processed$lr_morgan))
|
||||||
|
|
||||||
|
# Write output
|
||||||
|
write_csv(morgan_processed, "morgan_data.csv")
|
||||||
|
cat(sprintf("\nWrote morgan_data.csv with %d rows\n", nrow(morgan_processed)))
|
||||||
|
|
||||||
|
# Also provide a summary by country and period
|
||||||
|
summary_by_country <- morgan_processed %>%
|
||||||
|
group_by(country, period) %>%
|
||||||
|
summarise(
|
||||||
|
n_parties = n(),
|
||||||
|
mean_pos = mean(lr_morgan),
|
||||||
|
sd_pos = sd(lr_morgan),
|
||||||
|
.groups = "drop"
|
||||||
|
)
|
||||||
|
|
||||||
|
cat("\nSummary by country and period:\n")
|
||||||
|
print(summary_by_country, n = 50)
|
||||||
|
|
||||||
|
# ============================================================
|
||||||
|
# Generate lr_data-compatible output for pipeline integration
|
||||||
|
# ============================================================
|
||||||
|
|
||||||
|
cat("\n============================================================\n")
|
||||||
|
cat("Generating lr_data-compatible output (postwar only)\n")
|
||||||
|
cat("============================================================\n")
|
||||||
|
|
||||||
|
# Load text_data to get party-years with manifesto/PolDem coverage
|
||||||
|
if (!file.exists("text_data.csv")) {
|
||||||
|
cat("text_data.csv not present yet; skipping morgan_lr.csv generation on this pass.\n")
|
||||||
|
} else {
|
||||||
|
text_data <- read_csv("text_data.csv", show_col_types = FALSE)
|
||||||
|
|
||||||
|
# Convert Morgan ISO3 country codes to ISO2 (matching text_data format)
|
||||||
|
iso3_to_iso2 <- c(
|
||||||
|
"DNK" = "DK", "FIN" = "FI", "ISL" = "IS", "NOR" = "NO", "SWE" = "SE",
|
||||||
|
"NLD" = "NL", "BEL" = "BE", "DEU" = "DE", "FRA" = "FR", "ITA" = "IT",
|
||||||
|
"LUX" = "LU", "ISR" = "IL"
|
||||||
|
)
|
||||||
|
|
||||||
|
# Filter to postwar periods only (1945+)
|
||||||
|
morgan_postwar <- morgan_processed %>%
|
||||||
|
filter(year_end >= 1945) %>%
|
||||||
|
mutate(country_iso2 = iso3_to_iso2[country])
|
||||||
|
|
||||||
|
cat(sprintf("Postwar Morgan observations: %d party-periods\n", nrow(morgan_postwar)))
|
||||||
|
cat(sprintf("Countries: %s\n", paste(unique(morgan_postwar$country_iso2), collapse = ", ")))
|
||||||
|
|
||||||
|
# Get unique party-years from text_data
|
||||||
|
text_party_years <- text_data %>%
|
||||||
|
select(party, country, year) %>%
|
||||||
|
distinct()
|
||||||
|
|
||||||
|
cat(sprintf("Unique party-years in text_data: %d\n", nrow(text_party_years)))
|
||||||
|
|
||||||
|
# For each Morgan party-period, expand to all years where that party has text data
|
||||||
|
# within the Morgan period range (1945-1973 for postwar)
|
||||||
|
morgan_lr <- morgan_postwar %>%
|
||||||
|
# Join with text_data party-years
|
||||||
|
# Many-to-many is expected: one Morgan party-period maps to multiple years
|
||||||
|
inner_join(
|
||||||
|
text_party_years,
|
||||||
|
by = c("partyfacts_id" = "party", "country_iso2" = "country"),
|
||||||
|
relationship = "many-to-many"
|
||||||
|
) %>%
|
||||||
|
# Keep only years within the Morgan period
|
||||||
|
filter(year >= year_start & year <= year_end) %>%
|
||||||
|
# Format for lr_data.csv compatibility
|
||||||
|
transmute(
|
||||||
|
country = country_iso2,
|
||||||
|
party = partyfacts_id,
|
||||||
|
var = "lr_morgan",
|
||||||
|
year = year,
|
||||||
|
val = lr_morgan,
|
||||||
|
project = "Morgan",
|
||||||
|
# Morgan's continuous 0-100 scale is discretized to 10 points (matching CHES
|
||||||
|
# resolution) with the actual number of experts. The reconstructed sum
|
||||||
|
# round(mean × K × 10) is analogous to how CHES means are handled.
|
||||||
|
# See docs/k_scaling_validation.md Section 4.
|
||||||
|
n_scale = 10L,
|
||||||
|
val_int = as.integer(round(lr_morgan * 10)),
|
||||||
|
n_experts = as.integer(n_surveys)
|
||||||
|
) %>%
|
||||||
|
distinct() %>%
|
||||||
|
arrange(country, party, year)
|
||||||
|
|
||||||
|
cat(sprintf("\nGenerated %d lr_morgan observations\n", nrow(morgan_lr)))
|
||||||
|
cat(sprintf(" Unique parties: %d\n", n_distinct(morgan_lr$party)))
|
||||||
|
cat(sprintf(" Year range: %d-%d\n", min(morgan_lr$year), max(morgan_lr$year)))
|
||||||
|
|
||||||
|
# Summary by country
|
||||||
|
morgan_lr_summary <- morgan_lr %>%
|
||||||
|
group_by(country) %>%
|
||||||
|
summarise(
|
||||||
|
n_parties = n_distinct(party),
|
||||||
|
n_obs = n(),
|
||||||
|
year_min = min(year),
|
||||||
|
year_max = max(year),
|
||||||
|
.groups = "drop"
|
||||||
|
)
|
||||||
|
|
||||||
|
cat("\nMorgan L-R data by country:\n")
|
||||||
|
print(morgan_lr_summary, n = 20)
|
||||||
|
|
||||||
|
# Write morgan_lr.csv
|
||||||
|
write_csv(morgan_lr, "morgan_lr.csv")
|
||||||
|
cat(sprintf("\nWrote morgan_lr.csv with %d rows\n", nrow(morgan_lr)))
|
||||||
|
}
|
||||||
@@ -0,0 +1,162 @@
|
|||||||
|
# ============================================================
|
||||||
|
# process_poldem.R - PolDem Media Data Processing
|
||||||
|
# ============================================================
|
||||||
|
# Processes PolDem (Political Deliberation in the Media) data
|
||||||
|
# for the two-dimensional party-position model
|
||||||
|
#
|
||||||
|
# Input: $PARTY2D_RAW_DATA_DIR/poldem/poldem-election_all.csv (sentence-level)
|
||||||
|
# Output: poldem_data.csv (party-year-var aggregates)
|
||||||
|
# ============================================================
|
||||||
|
|
||||||
|
library(tidyverse)
|
||||||
|
library(countrycode)
|
||||||
|
|
||||||
|
# Set working directory (works both in RStudio and command line)
|
||||||
|
if (interactive() && requireNamespace("rstudioapi", quietly = TRUE)) {
|
||||||
|
try(setwd(dirname(rstudioapi::getActiveDocumentContext()$path)), silent = TRUE)
|
||||||
|
}
|
||||||
|
|
||||||
|
cat("Processing PolDem media data...\n")
|
||||||
|
|
||||||
|
raw_data_dir <- Sys.getenv(
|
||||||
|
"PARTY2D_RAW_DATA_DIR",
|
||||||
|
unset = file.path("..", "..", "_local", "raw")
|
||||||
|
)
|
||||||
|
poldem_raw_path <- file.path(raw_data_dir, "poldem", "poldem-election_all.csv")
|
||||||
|
partyfacts_path <- file.path(raw_data_dir, "partyfacts", "partyfacts-external-parties.csv")
|
||||||
|
|
||||||
|
# ============================================================
|
||||||
|
# PartyFacts Linkage (via CMP party IDs)
|
||||||
|
# ============================================================
|
||||||
|
|
||||||
|
partyfacts_raw <- read_csv(partyfacts_path, show_col_types = FALSE)
|
||||||
|
manifesto_link <- partyfacts_raw %>%
|
||||||
|
filter(dataset_key == "manifesto") %>%
|
||||||
|
transmute(cmp = as.numeric(dataset_party_id), # Convert to numeric for join
|
||||||
|
country_pf = countrycode(country, origin = 'iso3c', destination = "iso2c"),
|
||||||
|
party = partyfacts_id,
|
||||||
|
party = ifelse(party == 622, 604, party))
|
||||||
|
|
||||||
|
# ============================================================
|
||||||
|
# Issue Category Mapping to 4 Dimensions
|
||||||
|
# ============================================================
|
||||||
|
# For positive direction: type_high is the active trait
|
||||||
|
# For negative direction: we flip (same data, just contributes to the opposite trait)
|
||||||
|
|
||||||
|
poldem_mapping <- tribble(
|
||||||
|
~issue_cat, ~dimension, ~type_high, ~type_low,
|
||||||
|
# Economic dimension
|
||||||
|
"ecolib", "economic", "pro_market", "pro_welfare", # Economic liberalization
|
||||||
|
"welfare", "economic", "pro_welfare", "pro_market", # Welfare state
|
||||||
|
# Final exclusion: the PolDem economic-reform category is intentionally
|
||||||
|
# omitted because item-response diagnostics showed that it did not load
|
||||||
|
# substantively onto the economic latent trait.
|
||||||
|
# Cultural dimension
|
||||||
|
"immig", "cultural", "cosmopolitan", "traditional", # Immigration (pro = cosmopolitan)
|
||||||
|
"cultlib", "cultural", "cosmopolitan", "traditional", # Cultural liberalism
|
||||||
|
"nationalism", "cultural", "traditional", "cosmopolitan", # Nationalism (pro = traditional)
|
||||||
|
"europe", "cultural", "cosmopolitan", "traditional", # EU integration (pro = cosmopolitan)
|
||||||
|
"euro", "cultural", "cosmopolitan", "traditional", # Euro currency (pro = cosmopolitan)
|
||||||
|
"defense", "cultural", "traditional", "cosmopolitan", # Defense (pro = traditional)
|
||||||
|
"security", "cultural", "traditional", "cosmopolitan" # Security/law-order (pro = traditional)
|
||||||
|
)
|
||||||
|
|
||||||
|
cat(sprintf(" Using %d issue categories\n", nrow(poldem_mapping)))
|
||||||
|
|
||||||
|
# ============================================================
|
||||||
|
# Load and Clean PolDem Data
|
||||||
|
# ============================================================
|
||||||
|
|
||||||
|
poldem_raw <- read_csv(poldem_raw_path, show_col_types = FALSE)
|
||||||
|
cat(sprintf(" Raw PolDem data: %d rows\n", nrow(poldem_raw)))
|
||||||
|
|
||||||
|
poldem <- poldem_raw %>%
|
||||||
|
# Fix country codes
|
||||||
|
mutate(country = case_when(
|
||||||
|
iso2code == "AU" ~ "AT", # Austria (PolDem uses AU instead of AT)
|
||||||
|
iso2code == "UK" ~ "GB", # United Kingdom
|
||||||
|
TRUE ~ iso2code
|
||||||
|
)) %>%
|
||||||
|
# Extract year from article date (format: YYYY-MM-DD)
|
||||||
|
mutate(year = suppressWarnings(as.numeric(substr(date_art, 1, 4)))) %>%
|
||||||
|
# Filter to valid issue categories only
|
||||||
|
filter(issue_cat %in% poldem_mapping$issue_cat) %>%
|
||||||
|
# Convert direction to numeric and filter out neutral/NA
|
||||||
|
mutate(direction = as.numeric(direction)) %>%
|
||||||
|
filter(!is.na(direction) & direction != 0) %>%
|
||||||
|
# Remove rows with invalid years
|
||||||
|
filter(!is.na(year))
|
||||||
|
|
||||||
|
cat(sprintf(" After filtering: %d rows (valid issues, non-neutral)\n", nrow(poldem)))
|
||||||
|
|
||||||
|
# ============================================================
|
||||||
|
# Link to PartyFacts via CMP codes
|
||||||
|
# ============================================================
|
||||||
|
|
||||||
|
poldem <- poldem %>%
|
||||||
|
mutate(cmp = as.numeric(cmp)) %>%
|
||||||
|
left_join(manifesto_link, by = "cmp") %>%
|
||||||
|
filter(!is.na(party))
|
||||||
|
|
||||||
|
# Report linkage
|
||||||
|
n_linked <- nrow(poldem)
|
||||||
|
n_unlinked <- nrow(poldem_raw %>%
|
||||||
|
filter(issue_cat %in% poldem_mapping$issue_cat) %>%
|
||||||
|
mutate(direction = as.numeric(direction)) %>%
|
||||||
|
filter(!is.na(direction) & direction != 0)) - n_linked
|
||||||
|
|
||||||
|
cat(sprintf(" Linked to PartyFacts: %d rows\n", n_linked))
|
||||||
|
if (n_unlinked > 0) {
|
||||||
|
cat(sprintf(" Warning: %d rows could not be linked (missing CMP mapping)\n", n_unlinked))
|
||||||
|
}
|
||||||
|
|
||||||
|
# ============================================================
|
||||||
|
# Aggregate to Party-Year-Issue Level
|
||||||
|
# Using round(sum()) for weak direction values (0.5, -0.5)
|
||||||
|
# ============================================================
|
||||||
|
|
||||||
|
poldem_agg <- poldem %>%
|
||||||
|
left_join(poldem_mapping, by = "issue_cat") %>%
|
||||||
|
group_by(party, country, year, issue_cat, type_high, type_low) %>%
|
||||||
|
summarise(
|
||||||
|
# Sum positive directions (0.5 and 1), then round
|
||||||
|
positive = round(sum(direction[direction > 0])),
|
||||||
|
# Sum absolute directions for sample (all non-neutral), then round
|
||||||
|
sample = round(sum(abs(direction))),
|
||||||
|
n_obs = n(),
|
||||||
|
.groups = "drop"
|
||||||
|
) %>%
|
||||||
|
# Minimum 3 observations per group
|
||||||
|
filter(n_obs >= 3) %>%
|
||||||
|
select(-n_obs)
|
||||||
|
|
||||||
|
cat(sprintf(" After aggregation: %d party-year-issue observations\n", nrow(poldem_agg)))
|
||||||
|
|
||||||
|
# ============================================================
|
||||||
|
# Format Output (matching manifesto structure)
|
||||||
|
# ============================================================
|
||||||
|
|
||||||
|
poldem_data <- poldem_agg %>%
|
||||||
|
mutate(
|
||||||
|
var = paste0(issue_cat, "_poldem"),
|
||||||
|
project = "PolDem"
|
||||||
|
) %>%
|
||||||
|
select(party, country, year, var, positive, sample, type_high, type_low, project)
|
||||||
|
|
||||||
|
# ============================================================
|
||||||
|
# Write Output
|
||||||
|
# ============================================================
|
||||||
|
|
||||||
|
write_csv(poldem_data, "poldem_data.csv")
|
||||||
|
|
||||||
|
cat(sprintf("\nOutput: poldem_data.csv\n"))
|
||||||
|
cat(sprintf(" Total rows: %d\n", nrow(poldem_data)))
|
||||||
|
cat(sprintf(" Unique parties: %d\n", n_distinct(poldem_data$party)))
|
||||||
|
cat(sprintf(" Countries: %s\n", paste(sort(unique(poldem_data$country)), collapse = ", ")))
|
||||||
|
cat(sprintf(" Year range: %d-%d\n", min(poldem_data$year, na.rm = TRUE), max(poldem_data$year, na.rm = TRUE)))
|
||||||
|
cat("\n Rows by issue category:\n")
|
||||||
|
poldem_data %>%
|
||||||
|
group_by(var) %>%
|
||||||
|
summarise(n = n(), .groups = "drop") %>%
|
||||||
|
arrange(desc(n)) %>%
|
||||||
|
print()
|
||||||
@@ -0,0 +1,86 @@
|
|||||||
|
# Data setup
|
||||||
|
|
||||||
|
This directory exists because the public repository cannot redistribute the original raw/source files. It downloads script-accessible sources, checks user-provided source files, rebuilds model-ready inputs, and compares them with the committed files in `../data/`.
|
||||||
|
|
||||||
|
The main estimation workflow does not run these scripts. Once the five model-ready CSVs exist in `data/`, fitting and post-estimation are Julia/Stan-only.
|
||||||
|
|
||||||
|
The setup workflow never overwrites committed files in `data/`.
|
||||||
|
|
||||||
|
Put raw source files in `_local/raw/` or set `PARTY2D_RAW_DATA_DIR` to another local directory. See `source_manifest.csv` and the source notes below for source-specific access requirements.
|
||||||
|
|
||||||
|
## What downloads automatically?
|
||||||
|
|
||||||
|
`data-setup/R/01_download_sources.R` downloads source files that are script-accessible under the providers' terms and reports sources that require credentials or manual local files.
|
||||||
|
|
||||||
|
| Source | Automatic? | Requirement |
|
||||||
|
| --- | --- | --- |
|
||||||
|
| PolDem | Yes | none |
|
||||||
|
| PartyFacts crosswalk | Yes | none |
|
||||||
|
| CHES family files | Yes | none where provider links are live |
|
||||||
|
| POPPA | Yes | none; downloaded from Harvard Dataverse |
|
||||||
|
| Global Party Survey 2019 | Yes | none; downloaded from Harvard Dataverse |
|
||||||
|
| V-Party | Yes, with provider form details | set `PARTY2D_VDEM_EMAIL`; optionally set `PARTY2D_VDEM_GENDER` |
|
||||||
|
| Manifesto Project | Yes, with credentials | set your own `MANIFESTO_API_KEY` or `PARTY2D_MANIFESTO_API_KEY` |
|
||||||
|
| Morgan historical expert data | No | place `morgan_positions_raw.csv` locally; available on request |
|
||||||
|
|
||||||
|
Do not commit downloaded or user-provided source files.
|
||||||
|
|
||||||
|
Recommended local layout:
|
||||||
|
|
||||||
|
```text
|
||||||
|
_local/raw/
|
||||||
|
manifesto/MPDataset_MPDS2025a.csv
|
||||||
|
poldem/poldem-election_all.csv
|
||||||
|
ches/...
|
||||||
|
vparty/...
|
||||||
|
poppa/...
|
||||||
|
gps/...
|
||||||
|
morgan/...
|
||||||
|
partyfacts/partyfacts-external-parties.csv
|
||||||
|
```
|
||||||
|
|
||||||
|
Local output layout:
|
||||||
|
|
||||||
|
```text
|
||||||
|
_local/build/ # intermediate processing files
|
||||||
|
_local/generated-inputs/ # regenerated final model-input CSVs
|
||||||
|
_local/reports/ # comparison reports
|
||||||
|
```
|
||||||
|
|
||||||
|
## Commands
|
||||||
|
|
||||||
|
Run the full source-data setup workflow with:
|
||||||
|
|
||||||
|
```bash
|
||||||
|
bash data-setup/run_data_setup.sh
|
||||||
|
```
|
||||||
|
|
||||||
|
The command downloads script-accessible sources, checks required local files,
|
||||||
|
rebuilds model-ready inputs under `_local/generated-inputs/`, and writes a
|
||||||
|
comparison report under `_local/reports/`. It never replaces committed files in
|
||||||
|
`data/`.
|
||||||
|
|
||||||
|
V-Party:
|
||||||
|
|
||||||
|
```bash
|
||||||
|
export PARTY2D_VDEM_EMAIL='you@example.org'
|
||||||
|
export PARTY2D_VDEM_GENDER='' # blank means prefer not to say
|
||||||
|
```
|
||||||
|
|
||||||
|
Manifesto Project:
|
||||||
|
|
||||||
|
```bash
|
||||||
|
export MANIFESTO_API_KEY='...'
|
||||||
|
# or
|
||||||
|
export PARTY2D_MANIFESTO_API_KEY='...'
|
||||||
|
```
|
||||||
|
|
||||||
|
Morgan is not a public provider download. The local OCR/transcription file can be provided on request and should be placed at:
|
||||||
|
|
||||||
|
```text
|
||||||
|
_local/raw/morgan/morgan_positions_raw.csv
|
||||||
|
```
|
||||||
|
|
||||||
|
The comparison writes `_local/reports/input_comparison.md`. Replacing committed inputs, if ever needed, is a separate manual decision and is not done by these scripts.
|
||||||
|
|
||||||
|
Known behavior: the current public/rebuilt sources run through the workflow successfully, but regenerated `text_data.csv`, `expert.csv`, and `lr_data.csv` are not byte-identical to the committed model-ready inputs because of source-version and linkage differences. The comparison report records those differences explicitly.
|
||||||
@@ -0,0 +1,58 @@
|
|||||||
|
#!/usr/bin/env bash
|
||||||
|
set -euo pipefail
|
||||||
|
|
||||||
|
repo_root="$(cd "$(dirname "${BASH_SOURCE[0]}")/.." && pwd -P)"
|
||||||
|
raw_data_dir="${PARTY2D_RAW_DATA_DIR:-$repo_root/_local/raw}"
|
||||||
|
report_dir="${PARTY2D_REPORT_DIR:-$repo_root/_local/reports}"
|
||||||
|
mkdir -p "$report_dir"
|
||||||
|
missing_report="$report_dir/raw_data_preflight_missing.txt"
|
||||||
|
: > "$missing_report"
|
||||||
|
|
||||||
|
required_files=(
|
||||||
|
"manifesto/MPDataset_MPDS2025a.csv"
|
||||||
|
"poldem/poldem-election_all.csv"
|
||||||
|
"partyfacts/partyfacts-external-parties.csv"
|
||||||
|
"ches/1999-2019_CHES_dataset_means(v3).csv"
|
||||||
|
"ches/CHES_2024_final_v2.csv"
|
||||||
|
"ches/CHES_2024_expert_level.csv"
|
||||||
|
"ches/CHES_CA2023.csv"
|
||||||
|
"ches/CHES_CA2023_expert_level.csv"
|
||||||
|
"ches/ches_la_2020_aggregate_level_v01.csv"
|
||||||
|
"ches/CHES_LA2020_expert_level.csv"
|
||||||
|
"ches/CHES_ISRAEL_means_2021_2022.csv"
|
||||||
|
"ches/CHES_IL_expert_level.csv"
|
||||||
|
"vparty/V-Dem-CPD-Party-V2.rds"
|
||||||
|
"poppa/poppa_integrated_v2.rds"
|
||||||
|
"gps/Global Party Survey by Party SPSS V2_1_Apr_2020-2.tab"
|
||||||
|
"morgan/morgan_positions_raw.csv"
|
||||||
|
)
|
||||||
|
|
||||||
|
echo "Raw data directory: $raw_data_dir"
|
||||||
|
echo
|
||||||
|
echo "Required raw inputs for regeneration:"
|
||||||
|
|
||||||
|
missing=0
|
||||||
|
for rel in "${required_files[@]}"; do
|
||||||
|
path="$raw_data_dir/$rel"
|
||||||
|
if [ -s "$path" ]; then
|
||||||
|
bytes="$(wc -c < "$path")"
|
||||||
|
read -r sha _ < <(sha256sum "$path")
|
||||||
|
echo " OK $rel ($bytes bytes, sha256=$sha)"
|
||||||
|
else
|
||||||
|
echo " MISSING $rel"
|
||||||
|
printf '%s\n' "$rel" >> "$missing_report"
|
||||||
|
missing=1
|
||||||
|
fi
|
||||||
|
done
|
||||||
|
|
||||||
|
if [ "$missing" -ne 0 ]; then
|
||||||
|
echo
|
||||||
|
echo "At least one required raw input is missing." >&2
|
||||||
|
echo "Missing-file report: $missing_report" >&2
|
||||||
|
echo "See data-setup/README.md and docs/RAW_DATA_SOURCES.md for instructions." >&2
|
||||||
|
exit 1
|
||||||
|
fi
|
||||||
|
|
||||||
|
echo
|
||||||
|
echo "Required raw data preflight passed."
|
||||||
|
rm -f "$missing_report"
|
||||||
@@ -0,0 +1,34 @@
|
|||||||
|
#!/usr/bin/env bash
|
||||||
|
set -euo pipefail
|
||||||
|
|
||||||
|
usage() {
|
||||||
|
cat >&2 <<'EOF'
|
||||||
|
Usage: bash data-setup/run_data_setup.sh
|
||||||
|
EOF
|
||||||
|
}
|
||||||
|
|
||||||
|
if [ "$#" -ne 0 ]; then
|
||||||
|
usage
|
||||||
|
exit 1
|
||||||
|
fi
|
||||||
|
|
||||||
|
repo_root="$(cd "$(dirname "${BASH_SOURCE[0]}")/.." && pwd -P)"
|
||||||
|
cd "$repo_root"
|
||||||
|
|
||||||
|
export PARTY2D_RAW_DATA_DIR="${PARTY2D_RAW_DATA_DIR:-$repo_root/_local/raw}"
|
||||||
|
export PARTY2D_BUILD_DIR="${PARTY2D_BUILD_DIR:-$repo_root/_local/build}"
|
||||||
|
export PARTY2D_GENERATED_INPUT_DIR="${PARTY2D_GENERATED_INPUT_DIR:-$repo_root/_local/generated-inputs}"
|
||||||
|
export PARTY2D_REPORT_DIR="${PARTY2D_REPORT_DIR:-$repo_root/_local/reports}"
|
||||||
|
export R_LIBS_USER="${R_LIBS_USER:-$repo_root/_local/R/library}"
|
||||||
|
mkdir -p "$R_LIBS_USER"
|
||||||
|
|
||||||
|
command -v bash >/dev/null
|
||||||
|
command -v Rscript >/dev/null
|
||||||
|
bash -n data-setup/check_raw_data.sh
|
||||||
|
Rscript data-setup/R/00_install_dependencies.R
|
||||||
|
Rscript data-setup/R/01_download_sources.R || true
|
||||||
|
bash data-setup/check_raw_data.sh
|
||||||
|
rm -rf "$PARTY2D_BUILD_DIR" "$PARTY2D_GENERATED_INPUT_DIR"
|
||||||
|
mkdir -p "$PARTY2D_BUILD_DIR" "$PARTY2D_GENERATED_INPUT_DIR"
|
||||||
|
Rscript data-setup/R/02_build_model_inputs.R
|
||||||
|
Rscript data-setup/R/03_compare_generated_inputs.R
|
||||||
@@ -0,0 +1,17 @@
|
|||||||
|
source,scope,local_path,access,automatic_download,notes
|
||||||
|
Manifesto Project,manifesto text/coding,manifesto/MPDataset_MPDS2025a.csv,API key/login required,yes with MANIFESTO_API_KEY,Use the MPDS 2025a CSV export; raw file is not redistributed.
|
||||||
|
PolDem Election Campaigns,media campaign issue statements,poldem/poldem-election_all.csv,public download,yes,CSV URL https://poldem.eui.eu/downloads/cosa/poldem-election_all.csv; observed sha256 2cd8c9108b1b0b9c1b6594bb21acee709c70259cd02f450bc69fc09b505fc9fb.
|
||||||
|
CHES 1999-2019,expert party placements,ches/1999-2019_CHES_dataset_means(v3).csv,public archived download,yes,Downloaded from archived CHES URL at chesdata.eu.
|
||||||
|
CHES 2024,expert party placements,ches/CHES_2024_final_v2.csv,CHES terms,no,Requires matching expert-level file for expert counts.
|
||||||
|
CHES 2024 expert level,expert counts,ches/CHES_2024_expert_level.csv,CHES terms,no,Required for expert counts.
|
||||||
|
CHES Canada 2023 aggregate,expert party placements,ches/CHES_CA2023.csv,CHES terms,no,Required for Canada extension.
|
||||||
|
CHES Canada 2023,expert party placements,ches/CHES_CA2023_expert_level.csv,CHES terms,no,Used by expert-source processing where available.
|
||||||
|
CHES Latin America aggregate,expert party placements,ches/ches_la_2020_aggregate_level_v01.csv,CHES terms,no,Required for Latin America extension.
|
||||||
|
CHES Latin America 2020,expert party placements,ches/CHES_LA2020_expert_level.csv,CHES terms,no,Used by expert-source processing where available.
|
||||||
|
CHES Israel aggregate,expert party placements,ches/CHES_ISRAEL_means_2021_2022.csv,CHES terms,no,Required for Israel extension.
|
||||||
|
CHES Israel,expert party placements,ches/CHES_IL_expert_level.csv,CHES terms,no,Used by expert-source processing where available.
|
||||||
|
V-Party,expert-coded party variables,vparty/V-Dem-CPD-Party-V2.rds,V-Dem form terms,yes with PARTY2D_VDEM_EMAIL,Downloader submits provider form and extracts R data from ZIP.
|
||||||
|
POPPA,expert party placements,poppa/poppa_integrated_v2.rds,public Dataverse,yes,Downloaded from Harvard Dataverse DOI 10.7910/DVN/RMQREQ.
|
||||||
|
Global Party Survey 2019,expert party placements,gps/Global Party Survey by Party SPSS V2_1_Apr_2020-2.tab,public Dataverse,yes,Downloaded from Harvard Dataverse DOI 10.7910/DVN/WMGTNS.
|
||||||
|
Morgan historical expert data,historical left-right placements,morgan/morgan_positions_raw.csv,derived local transcription/no redistribution,no public URL,Local OCR/transcription source used for historical anchoring.
|
||||||
|
PartyFacts crosswalk,party ID harmonization,partyfacts/partyfacts-external-parties.csv,public download,yes,Support crosswalk required by source-processing scripts.
|
||||||
|
@@ -0,0 +1,13 @@
|
|||||||
|
# Data directory
|
||||||
|
|
||||||
|
This directory contains only the processed, model-ready inputs used by the Julia/Stan estimation pipeline:
|
||||||
|
|
||||||
|
- `text_data.csv`
|
||||||
|
- `expert.csv`
|
||||||
|
- `lr_data.csv`
|
||||||
|
- `union_mapping.csv`
|
||||||
|
- `party_families.csv`
|
||||||
|
|
||||||
|
Original raw source files and intermediate build products are not stored here. To regenerate the processed inputs, place raw files in a local directory and set `PARTY2D_RAW_DATA_DIR`; see `../data-setup/README.md`.
|
||||||
|
|
||||||
|
Generated outputs and temporary staging files are ignored by git.
|
||||||
+25079
File diff suppressed because it is too large
Load Diff
+2208
File diff suppressed because it is too large
Load Diff
File diff suppressed because it is too large
Load Diff
@@ -0,0 +1,3 @@
|
|||||||
|
6b9aeda20489000983e459e4eb19d2788ed035eaab5deb9a5e010338670b2ff1 party_2d_election_year_panel_v0.zip
|
||||||
|
a7ba39d1b35f6b21d98a92f779e64679646f0d245ea560a3f2ee551c97066280 party_2d_annual_model_output_v0.csv.xz
|
||||||
|
b99f7ef0e8a4c821183a2a0f957752303479e78e5afb685f173fd17f60039b3e party_2d_diagnostics_report_v0.pdf
|
||||||
Binary file not shown.
+38203
File diff suppressed because it is too large
Load Diff
@@ -0,0 +1,107 @@
|
|||||||
|
manifesto_pf_id,manifesto_name,expert_pf_id,expert_name,country,status
|
||||||
|
3889,PJ,6648,PF-PJ,AR,implemented
|
||||||
|
6161,FAP,1365,PS,AR,implemented
|
||||||
|
6161,FAP,6160,FR,AR,implemented
|
||||||
|
6161,FAP,6554,FPCyS,AR,implemented
|
||||||
|
486,LPA,1998,LP,AU,implemented
|
||||||
|
1743,NPA,338,NAT,AU,implemented
|
||||||
|
1760,HDZ BiH,3904,HDZ-HK~HNZ,BA,implemented
|
||||||
|
36,N-VA,756,CD+NVA,BE,implemented
|
||||||
|
604,CD/V,622,CD&V,BE,implemented
|
||||||
|
604,CD/V,756,CD+NVA,BE,implemented
|
||||||
|
1680,sp.a,1586,sp.a-SPIRIT,BE,implemented
|
||||||
|
374,NDSV,5848,KSII,BG,implemented
|
||||||
|
482,SDS,3908,G-VMRO; VMRO-BND,BG,implemented
|
||||||
|
1765,ONS,3908,G-VMRO; VMRO-BND,BG,implemented
|
||||||
|
5649,Patriotic Front - NFSB and VMRO,2057,NFSB,BG,implemented
|
||||||
|
5649,Patriotic Front - NFSB and VMRO,3908,G-VMRO; VMRO-BND,BG,implemented
|
||||||
|
360,FDP/PLR,1231,FDP/PLR,CH,implemented
|
||||||
|
6061,Alliance,928,RN,CL,implemented
|
||||||
|
6061,Alliance,1599,UDI,CL,implemented
|
||||||
|
1707,STAN,751,SNK-ED,CZ,implemented
|
||||||
|
2138,LB,1041,KSC,CZ,implemented
|
||||||
|
6202,KDU-ČSL-US-DEU,104,US-DEU,CZ,implemented
|
||||||
|
211,CDU/CSU,1375,CDU,DE,implemented
|
||||||
|
211,CDU/CSU,1731,CSU,DE,implemented
|
||||||
|
1816,B90/Grüne,10,Die Grünen,DE,implemented
|
||||||
|
3925,RED-ID,797,ID,EC,implemented
|
||||||
|
685,RP,491,ERP,EE,implemented
|
||||||
|
779,I/ERSP,908,RKI,EE,implemented
|
||||||
|
779,I/ERSP,1299,ERSP,EE,implemented
|
||||||
|
139,CiU,4795,CDC,ES,implemented
|
||||||
|
8271,Compromís–Podemos–EUPV,5623,CC,ES,implemented
|
||||||
|
213,MoDem,496,MoDem,FR,implemented
|
||||||
|
1108,EELV,5650,EELV,FR,implemented
|
||||||
|
1595,UMP,4628,Les Républicains,FR,implemented
|
||||||
|
1595,UMP,8168,LR,FR,implemented
|
||||||
|
1468,PASOK,7909,KINAL,GR,implemented
|
||||||
|
7347,EL,378,OP,GR,implemented
|
||||||
|
1475,SDP,8842,SDP-HSLS,HR,implemented
|
||||||
|
2522,Kukuriku,78,DC,HR,implemented
|
||||||
|
3648,ZL,8036,HKDU,HR,implemented
|
||||||
|
3918,DA-IDS-RDS,513,IDS,HR,implemented
|
||||||
|
242,PBP,8241,PBPS,IE,implemented
|
||||||
|
201,UdC,1758,UC,IT,implemented
|
||||||
|
1212,SEL,7031,SEL,IT,implemented
|
||||||
|
1737,Olive Tree,878,DS,IT,implemented
|
||||||
|
6241,House of Freedom,813,AN,IT,implemented
|
||||||
|
6241,House of Freedom,1519,CeD,IT,implemented
|
||||||
|
1967,SLFP,4020,CP / VLSSP,LK,implemented
|
||||||
|
1967,SLFP,5414,LSS,LK,implemented
|
||||||
|
1967,SLFP,6691,CP,LK,implemented
|
||||||
|
197,BSDK,168,LRS,LT,implemented
|
||||||
|
197,BSDK,1747,LMP-NDP,LT,implemented
|
||||||
|
377,LTS,1410,LLaS,LT,implemented
|
||||||
|
5779,SK,1407,LZP,LT,implemented
|
||||||
|
186,LSAP/POSL,898,SDP,LU,implemented
|
||||||
|
708,LNNK-LZP,1296,LZP,LV,implemented
|
||||||
|
1704,TB-LNNK,671,TB,LV,implemented
|
||||||
|
1704,TB-LNNK,1789,LNNK,LV,implemented
|
||||||
|
1704,TB-LNNK,7619,NATBLNNK,LV,implemented
|
||||||
|
7622,ACUM,7904,PAS,MD,implemented
|
||||||
|
3254,DSCG,3253,HGI,ME,implemented
|
||||||
|
1537,GL,1533,Groen,NL,implemented
|
||||||
|
716,Alliance,1119,NLP,NZ,implemented
|
||||||
|
4219,C90,5130,P2000,PE,implemented
|
||||||
|
1458,WAK,70,ZChN,PL,implemented
|
||||||
|
8268,UW,1566,D|W|U,PL,implemented
|
||||||
|
192,CDR,645,PAC,RO,implemented
|
||||||
|
1347,PSD-PUR,1443,PU|PC,RO,implemented
|
||||||
|
5941,USL,120,PSD,RO,implemented
|
||||||
|
5941,USL,481,PNL,RO,implemented
|
||||||
|
5941,USL,1541,UNPR,RO,implemented
|
||||||
|
6153,PSD-PC,1443,PU|PC,RO,implemented
|
||||||
|
8626,LDP/LSV/SDS,4769,LSV,RS,implemented
|
||||||
|
205,SV,200,SDSS,SK,implemented
|
||||||
|
226,SDK,200,SDSS,SK,implemented
|
||||||
|
1617,SDKÚ-DS,983,DS,SK,implemented
|
||||||
|
6629,DÚS,707,DUS,SK,implemented
|
||||||
|
1658,FA,3671,NE,UY,implemented
|
||||||
|
301,"SYRIZA, SYN; SYRIZA, Syriza, SYN/SYRIZA",1682,DIKKI,GR,implemented
|
||||||
|
676,"KDU, KDU-ČSL, KDU-CSL, KDU/CSL, KDUCSL, KDU–CSL, KDU–Č, KDU- ČSL, CSL",824,KDS,CZ,implemented
|
||||||
|
701,"ZZS, LZS",1702,LZS,LV,implemented
|
||||||
|
852,"V, Unity, UNITY, VIENOTIBA, JV, PS",1531,JL,LV,implemented
|
||||||
|
2190,"DSS, DSS/NS",2346,NS,RS,implemented
|
||||||
|
2530,"FpV, FPV, FPV-PJ, AFplV, FplV",623,PJ,AR,implemented
|
||||||
|
3906,NA,356,PT,BR,implemented
|
||||||
|
3906,NA,723,PSB,BR,implemented
|
||||||
|
3906,NA,1009,PDT,BR,implemented
|
||||||
|
3906,NA,4405,"PR, PR (2), PR / PL, PR/PL, PR PL, PL/PR",BR,implemented
|
||||||
|
3906,NA,458,PTB,BR,implemented
|
||||||
|
3906,NA,1823,PL,BR,implemented
|
||||||
|
4550,"C, Concertacion, CPD",6,PS,CL,implemented
|
||||||
|
4550,"C, Concertacion, CPD",54,PPD,CL,implemented
|
||||||
|
4550,"C, Concertacion, CPD",390,PDC,CL,implemented
|
||||||
|
4550,"C, Concertacion, CPD",437,PRSD,CL,implemented
|
||||||
|
8999,"ZMS, Aleksandar V..., PS-TN, Serbia is Wi..., ally",3177,SNS,RS,implemented
|
||||||
|
1117,PO,4630,.N,PL,implemented
|
||||||
|
4550,Concertacion,162,PC,CL,implemented
|
||||||
|
4550,Concertacion,209,PH,CL,implemented
|
||||||
|
5668,EH Bildu,1671,Amaiur,ES,implemented
|
||||||
|
6241,CdL,1767,CCD,IT,implemented
|
||||||
|
3979,Salvemos a México,1474,PRI,MX,implemented
|
||||||
|
3979,Salvemos a México,446,PVEM,MX,implemented
|
||||||
|
7912,Joint List,421,Hadash,IL,implemented
|
||||||
|
7912,Joint List,1663,Balad,IL,implemented
|
||||||
|
365,PdL,1626,FI,IT,implemented
|
||||||
|
365,PdL,813,AN,IT,implemented
|
||||||
|
@@ -0,0 +1,17 @@
|
|||||||
|
# Diagnostics
|
||||||
|
|
||||||
|
This folder contains the repository diagnostics report for the Scientific Data Data Descriptor. It is generated from the model-ready inputs and the completed model/post-estimation outputs, so it can only be rerun after the estimation workflow has produced party-position, convergence, and validation artifacts.
|
||||||
|
|
||||||
|
Regenerate from the repository root with:
|
||||||
|
|
||||||
|
```bash
|
||||||
|
Rscript diagnostics/generate_diagnostics.R
|
||||||
|
```
|
||||||
|
|
||||||
|
If model outputs are stored outside the repository root, point the script to them:
|
||||||
|
|
||||||
|
```bash
|
||||||
|
PARTY2D_OUTPUTS_DIR=/path/to/outputs Rscript diagnostics/generate_diagnostics.R
|
||||||
|
```
|
||||||
|
|
||||||
|
Generated files are written to `diagnostics/generated/`. The PDF report is also copied to `data/releases/` for the release bundle. PDF rendering uses R Markdown/Pandoc and requires a LaTeX engine such as `pdflatex`.
|
||||||
@@ -0,0 +1,609 @@
|
|||||||
|
#!/usr/bin/env Rscript
|
||||||
|
|
||||||
|
suppressPackageStartupMessages(library(tidyverse))
|
||||||
|
|
||||||
|
find_repo_root <- function() {
|
||||||
|
args <- commandArgs(trailingOnly = FALSE)
|
||||||
|
file_arg <- "--file="
|
||||||
|
script_arg <- args[startsWith(args, file_arg)][1]
|
||||||
|
if (!is.na(script_arg)) {
|
||||||
|
return(normalizePath(file.path(dirname(sub(file_arg, "", script_arg)), "..")))
|
||||||
|
}
|
||||||
|
if (file.exists("data/text_data.csv")) return(normalizePath(getwd()))
|
||||||
|
stop("Cannot find repository root. Run from the party2d repository root.")
|
||||||
|
}
|
||||||
|
|
||||||
|
repo_root <- find_repo_root()
|
||||||
|
setwd(repo_root)
|
||||||
|
|
||||||
|
release_version <- Sys.getenv("PARTY2D_RELEASE_VERSION", "v0")
|
||||||
|
outputs_dir <- Sys.getenv("PARTY2D_OUTPUTS_DIR", "outputs")
|
||||||
|
if (!grepl("^/", outputs_dir)) outputs_dir <- file.path(repo_root, outputs_dir)
|
||||||
|
if (!dir.exists(outputs_dir)) {
|
||||||
|
stop("Model output directory not found: ", outputs_dir, ". Run estimation/validation first or set PARTY2D_OUTPUTS_DIR.")
|
||||||
|
}
|
||||||
|
supplementary_inputs_dir <- Sys.getenv("PARTY2D_SUPPLEMENTARY_INPUTS_DIR", file.path(dirname(repo_root), "archive", "supplementary_inputs"))
|
||||||
|
if (!grepl("^/", supplementary_inputs_dir)) supplementary_inputs_dir <- file.path(repo_root, supplementary_inputs_dir)
|
||||||
|
|
||||||
|
generated_dir <- file.path(repo_root, "diagnostics", "generated")
|
||||||
|
release_dir <- file.path(repo_root, "data", "releases")
|
||||||
|
dir.create(generated_dir, recursive = TRUE, showWarnings = FALSE)
|
||||||
|
dir.create(release_dir, recursive = TRUE, showWarnings = FALSE)
|
||||||
|
|
||||||
|
required_inputs <- c(
|
||||||
|
"data/text_data.csv",
|
||||||
|
"data/expert.csv",
|
||||||
|
"data/lr_data.csv",
|
||||||
|
"data/union_mapping.csv",
|
||||||
|
"data/party_families.csv"
|
||||||
|
)
|
||||||
|
missing_inputs <- required_inputs[!file.exists(required_inputs)]
|
||||||
|
if (length(missing_inputs) > 0) {
|
||||||
|
stop("Missing required model input files: ", paste(missing_inputs, collapse = ", "))
|
||||||
|
}
|
||||||
|
|
||||||
|
latest_file <- function(path, pattern) {
|
||||||
|
if (!dir.exists(path)) return(NA_character_)
|
||||||
|
files <- list.files(path, pattern = pattern, full.names = TRUE)
|
||||||
|
if (length(files) == 0) return(NA_character_)
|
||||||
|
sort(files)[length(files)]
|
||||||
|
}
|
||||||
|
|
||||||
|
read_if_exists <- function(path) {
|
||||||
|
if (is.na(path) || !file.exists(path)) return(tibble())
|
||||||
|
readr::read_csv(path, show_col_types = FALSE)
|
||||||
|
}
|
||||||
|
|
||||||
|
supplementary_file <- function(...) {
|
||||||
|
path <- file.path(supplementary_inputs_dir, ...)
|
||||||
|
if (file.exists(path)) path else NA_character_
|
||||||
|
}
|
||||||
|
|
||||||
|
public_dimension <- function(x) {
|
||||||
|
dplyr::recode(as.character(x),
|
||||||
|
economic_lr = "economic left-right",
|
||||||
|
galtan = "cultural cosmopolitan--traditionalist",
|
||||||
|
Economic = "economic left-right",
|
||||||
|
Cultural = "cultural cosmopolitan--traditionalist",
|
||||||
|
`Economic Left-Right` = "economic left-right",
|
||||||
|
.default = as.character(x)
|
||||||
|
)
|
||||||
|
}
|
||||||
|
|
||||||
|
infer_dimension <- function(type_low, type_high) {
|
||||||
|
dplyr::case_when(
|
||||||
|
type_low %in% c("pro_market", "pro_welfare", "left", "right") |
|
||||||
|
type_high %in% c("pro_market", "pro_welfare", "left", "right") ~ "economic left-right",
|
||||||
|
type_low %in% c("cosmopolitan", "traditional") |
|
||||||
|
type_high %in% c("cosmopolitan", "traditional") ~ "cultural cosmopolitan--traditionalist",
|
||||||
|
TRUE ~ "general left-right"
|
||||||
|
)
|
||||||
|
}
|
||||||
|
|
||||||
|
is_reversed_for_reporting <- function(type_high) {
|
||||||
|
type_high %in% c("pro_welfare", "left", "cosmopolitan")
|
||||||
|
}
|
||||||
|
|
||||||
|
fmt_num <- function(x, digits = 3) {
|
||||||
|
ifelse(is.na(x), "NA", formatC(x, digits = digits, format = "f"))
|
||||||
|
}
|
||||||
|
|
||||||
|
display_path <- function(path) {
|
||||||
|
if (is.na(path) || !nzchar(path)) return("not available")
|
||||||
|
normalized <- normalizePath(path, mustWork = FALSE)
|
||||||
|
root_prefix <- paste0(normalizePath(repo_root, mustWork = FALSE), .Platform$file.sep)
|
||||||
|
if (startsWith(normalized, root_prefix)) return(sub(root_prefix, "", normalized, fixed = TRUE))
|
||||||
|
basename(path)
|
||||||
|
}
|
||||||
|
|
||||||
|
md_table <- function(df, n = Inf) {
|
||||||
|
if (nrow(df) == 0) return("_No rows available._\n")
|
||||||
|
df <- head(df, n)
|
||||||
|
df <- mutate(df, across(everything(), as.character))
|
||||||
|
header <- paste0("| ", paste(names(df), collapse = " | "), " |")
|
||||||
|
sep <- paste0("| ", paste(rep("---", ncol(df)), collapse = " | "), " |")
|
||||||
|
rows <- apply(df, 1, function(x) paste0("| ", paste(x, collapse = " | "), " |"))
|
||||||
|
paste(c(header, sep, rows), collapse = "\n")
|
||||||
|
}
|
||||||
|
|
||||||
|
text_data <- read_csv("data/text_data.csv", show_col_types = FALSE)
|
||||||
|
expert <- read_csv("data/expert.csv", show_col_types = FALSE)
|
||||||
|
lr_data <- read_csv("data/lr_data.csv", show_col_types = FALSE)
|
||||||
|
union_mapping <- read_csv("data/union_mapping.csv", show_col_types = FALSE)
|
||||||
|
party_families <- read_csv("data/party_families.csv", show_col_types = FALSE)
|
||||||
|
|
||||||
|
excluded_poldem <- text_data %>%
|
||||||
|
filter(project == "PolDem", str_detect(str_to_lower(var), "reform"))
|
||||||
|
if (nrow(excluded_poldem) > 0) {
|
||||||
|
stop("Excluded PolDem reform item is present in data/text_data.csv. Final-model diagnostics must be regenerated after removing it.")
|
||||||
|
}
|
||||||
|
|
||||||
|
annual_release <- file.path(release_dir, paste0("party_2d_annual_model_output_", release_version, ".csv.gz"))
|
||||||
|
panel_release <- file.path(release_dir, paste0("party_2d_election_year_panel_", release_version, ".csv.gz"))
|
||||||
|
model_positions_file <- latest_file(file.path(outputs_dir, "estimations", "latest"), "^party_positions_.*\\.csv$")
|
||||||
|
if (is.na(model_positions_file)) {
|
||||||
|
stop("No post-estimation party-position output found under ", outputs_dir, ". Run model estimation/post-estimation first, or set PARTY2D_OUTPUTS_DIR to an outputs directory.")
|
||||||
|
}
|
||||||
|
positions <- read_csv(model_positions_file, show_col_types = FALSE)
|
||||||
|
|
||||||
|
item_rows <- bind_rows(
|
||||||
|
text_data %>%
|
||||||
|
mutate(source_file = "text_data.csv") %>%
|
||||||
|
group_by(source_file, item = var, source = project, type_low, type_high) %>%
|
||||||
|
summarise(observations = n(), party_years = n_distinct(party, year), parties = n_distinct(party), countries = n_distinct(country), min_year = min(year), max_year = max(year), .groups = "drop"),
|
||||||
|
expert %>%
|
||||||
|
mutate(source_file = "expert.csv") %>%
|
||||||
|
group_by(source_file, item = var, source = project, type_low, type_high) %>%
|
||||||
|
summarise(observations = n(), party_years = n_distinct(party, year), parties = n_distinct(party), countries = n_distinct(country), min_year = min(year), max_year = max(year), .groups = "drop"),
|
||||||
|
lr_data %>%
|
||||||
|
mutate(source_file = "lr_data.csv", type_low = NA_character_, type_high = NA_character_) %>%
|
||||||
|
group_by(source_file, item = var, source = project, type_low, type_high) %>%
|
||||||
|
summarise(observations = n(), party_years = n_distinct(party, year), parties = n_distinct(party), countries = n_distinct(country), min_year = min(year), max_year = max(year), .groups = "drop")
|
||||||
|
) %>%
|
||||||
|
mutate(
|
||||||
|
dimension = infer_dimension(type_low, type_high),
|
||||||
|
higher_values_indicate = if_else(is.na(type_high), "source-coded left-right", type_high),
|
||||||
|
reversed_for_reporting = if_else(is_reversed_for_reporting(type_high), "yes", "no")
|
||||||
|
) %>%
|
||||||
|
select(source_file, item, source, dimension, type_low, type_high, higher_values_indicate, reversed_for_reporting, observations, party_years, parties, countries, min_year, max_year) %>%
|
||||||
|
arrange(source_file, source, dimension, item)
|
||||||
|
|
||||||
|
source_coverage <- bind_rows(
|
||||||
|
text_data %>% transmute(source_file = "text_data.csv", source = project, item = var, party, country, year),
|
||||||
|
expert %>% transmute(source_file = "expert.csv", source = project, item = var, party, country, year),
|
||||||
|
lr_data %>% transmute(source_file = "lr_data.csv", source = project, item = var, party, country, year)
|
||||||
|
) %>%
|
||||||
|
group_by(source_file, source) %>%
|
||||||
|
summarise(items = n_distinct(item), observations = n(), party_years = n_distinct(party, year), parties = n_distinct(party), countries = n_distinct(country), min_year = min(year), max_year = max(year), .groups = "drop") %>%
|
||||||
|
arrange(source_file, source)
|
||||||
|
|
||||||
|
party_year_source_coverage <- bind_rows(
|
||||||
|
text_data %>% distinct(party, country, year) %>% mutate(has_text = TRUE, has_expert = FALSE, has_general_lr = FALSE),
|
||||||
|
expert %>% distinct(party, country, year) %>% mutate(has_text = FALSE, has_expert = TRUE, has_general_lr = FALSE),
|
||||||
|
lr_data %>% distinct(party, country, year) %>% mutate(has_text = FALSE, has_expert = FALSE, has_general_lr = TRUE)
|
||||||
|
) %>%
|
||||||
|
group_by(party, country, year) %>%
|
||||||
|
summarise(has_text = any(has_text), has_expert = any(has_expert), has_general_lr = any(has_general_lr), n_source_types = has_text + has_expert + has_general_lr, .groups = "drop") %>%
|
||||||
|
arrange(year, country, party)
|
||||||
|
|
||||||
|
alliance_union_harmonization <- bind_rows(
|
||||||
|
tibble(metric = "constituent_mappings", category = "all", value = nrow(union_mapping)),
|
||||||
|
tibble(metric = "unique_union_or_alliance_ids", category = "all", value = n_distinct(union_mapping$manifesto_pf_id)),
|
||||||
|
tibble(metric = "unique_constituent_party_ids", category = "all", value = n_distinct(union_mapping$expert_pf_id)),
|
||||||
|
union_mapping %>% count(country, name = "value") %>% transmute(metric = "mappings_by_country", category = country, value),
|
||||||
|
union_mapping %>% count(status, name = "value") %>% transmute(metric = "mappings_by_status", category = status, value)
|
||||||
|
)
|
||||||
|
|
||||||
|
party_col <- if ("party_id" %in% names(positions)) "party_id" else "party"
|
||||||
|
party_family_coverage <- positions %>%
|
||||||
|
transmute(partyfacts_id = .data[[party_col]], country, year) %>%
|
||||||
|
inner_join(party_families, by = "partyfacts_id") %>%
|
||||||
|
group_by(family) %>%
|
||||||
|
summarise(parties = n_distinct(partyfacts_id), party_years = n(), countries = n_distinct(country), min_year = min(year), max_year = max(year), .groups = "drop") %>%
|
||||||
|
arrange(desc(party_years))
|
||||||
|
|
||||||
|
convergence_summary_file <- latest_file(file.path(outputs_dir, "diagnostics"), "^convergence_summary_.*\\.csv$")
|
||||||
|
convergence_detail_file <- latest_file(file.path(outputs_dir, "diagnostics"), "^convergence_diagnostics_.*\\.csv$")
|
||||||
|
if (is.na(convergence_summary_file) || is.na(convergence_detail_file)) {
|
||||||
|
stop("Convergence diagnostics not found under ", outputs_dir, ". Run the model diagnostics before generating the report.")
|
||||||
|
}
|
||||||
|
model_convergence_summary <- read_if_exists(convergence_summary_file) %>%
|
||||||
|
identity()
|
||||||
|
model_convergence_by_dimension <- read_if_exists(convergence_detail_file) %>%
|
||||||
|
group_by(dimension) %>%
|
||||||
|
summarise(parameters = n(), mean_rhat = mean(rhat, na.rm = TRUE), max_rhat = max(rhat, na.rm = TRUE), min_ess_bulk = min(ess_bulk, na.rm = TRUE), mean_ess_bulk = mean(ess_bulk, na.rm = TRUE), .groups = "drop") %>%
|
||||||
|
mutate(dimension = public_dimension(dimension)) %>%
|
||||||
|
arrange(dimension)
|
||||||
|
|
||||||
|
convergent_summary_file <- latest_file(file.path(outputs_dir, "validation", "latest"), "^convergent_summary_.*\\.csv$")
|
||||||
|
discriminant_summary_file <- latest_file(file.path(outputs_dir, "validation", "latest"), "^discriminant_summary_.*\\.csv$")
|
||||||
|
uncertainty_summary_file <- latest_file(file.path(outputs_dir, "validation", "latest"), "^uncertainty_cic_summary_.*\\.csv$")
|
||||||
|
external_validation_file <- latest_file(file.path(outputs_dir, "validation", "latest"), "^external_validation_.*\\.csv$")
|
||||||
|
construct_families_file <- latest_file(file.path(outputs_dir, "validation", "latest"), "^construct_families_.*\\.csv$")
|
||||||
|
construct_unstable_file <- latest_file(file.path(outputs_dir, "validation", "latest"), "^construct_unstable_.*\\.csv$")
|
||||||
|
if (any(is.na(c(convergent_summary_file, discriminant_summary_file, uncertainty_summary_file, external_validation_file, construct_families_file, construct_unstable_file)))) {
|
||||||
|
stop("Validation diagnostics not found under ", outputs_dir, ". Run validation before generating the report.")
|
||||||
|
}
|
||||||
|
|
||||||
|
convergent_summary <- read_if_exists(convergent_summary_file) %>%
|
||||||
|
mutate(diagnostic = "convergent validity", dimension = public_dimension(dimension))
|
||||||
|
discriminant_summary <- read_if_exists(discriminant_summary_file) %>%
|
||||||
|
mutate(diagnostic = "discriminant validity", model_dim = public_dimension(model_dim), expert_dim = public_dimension(expert_dim))
|
||||||
|
uncertainty_summary <- read_if_exists(uncertainty_summary_file) %>%
|
||||||
|
mutate(diagnostic = "posterior predictive coverage", dimension = public_dimension(dimension))
|
||||||
|
|
||||||
|
external_validation_correlations <- read_if_exists(external_validation_file) %>%
|
||||||
|
group_by(var, dimension) %>%
|
||||||
|
summarise(n = n(), pearson_r = cor(expert_val, model_val, use = "complete.obs"), mean_absolute_error = mean(abs_error, na.rm = TRUE), coverage_95 = mean(covered_95, na.rm = TRUE), .groups = "drop") %>%
|
||||||
|
mutate(dimension = public_dimension(dimension)) %>%
|
||||||
|
arrange(dimension, var)
|
||||||
|
construct_family_positions <- read_if_exists(construct_families_file) %>%
|
||||||
|
rename(mean_cultural = mean_galtan, sd_cultural = sd_galtan) %>%
|
||||||
|
arrange(mean_economic)
|
||||||
|
construct_temporal_stability <- read_if_exists(construct_unstable_file) %>%
|
||||||
|
mutate(dimension = public_dimension(dimension)) %>%
|
||||||
|
arrange(desc(annual_change))
|
||||||
|
source_composition_balance <- read_if_exists(supplementary_file("validation", "source_composition_balance.csv")) %>%
|
||||||
|
mutate(dimension = public_dimension(dimension))
|
||||||
|
robustness_sensitivity <- read_if_exists(supplementary_file("validation", "table10_sensitivity.csv")) %>%
|
||||||
|
mutate(
|
||||||
|
dimension = public_dimension(dimension),
|
||||||
|
across(everything(), ~ na_if(as.character(.x), "[INSERT VALUE]"))
|
||||||
|
) %>%
|
||||||
|
select(specification, ablated_source, dimension, matched_n, correlation_with_production,
|
||||||
|
mean_abs_difference, median_abs_difference, p95_abs_difference,
|
||||||
|
mean_interval_width_production, mean_interval_width_ablation)
|
||||||
|
|
||||||
|
posterior_uncertainty <- positions %>%
|
||||||
|
summarise(
|
||||||
|
rows = n(),
|
||||||
|
parties = n_distinct(.data[[party_col]]),
|
||||||
|
countries = n_distinct(country),
|
||||||
|
min_year = min(year),
|
||||||
|
max_year = max(year),
|
||||||
|
mean_economic_se = mean(economic_lr_se, na.rm = TRUE),
|
||||||
|
median_economic_se = median(economic_lr_se, na.rm = TRUE),
|
||||||
|
mean_cultural_se = mean(galtan_se, na.rm = TRUE),
|
||||||
|
median_cultural_se = median(galtan_se, na.rm = TRUE)
|
||||||
|
)
|
||||||
|
|
||||||
|
write_csv(item_rows, file.path(generated_dir, "item_coverage.csv"))
|
||||||
|
write_csv(source_coverage, file.path(generated_dir, "source_coverage.csv"))
|
||||||
|
write_csv(party_year_source_coverage, file.path(generated_dir, "party_year_source_coverage.csv"))
|
||||||
|
write_csv(item_rows, file.path(generated_dir, "item_coding_orientation.csv"))
|
||||||
|
write_csv(filter(item_rows, reversed_for_reporting == "yes"), file.path(generated_dir, "reversed_items.csv"))
|
||||||
|
write_csv(alliance_union_harmonization, file.path(generated_dir, "alliance_union_harmonization.csv"))
|
||||||
|
write_csv(party_family_coverage, file.path(generated_dir, "party_family_coverage.csv"))
|
||||||
|
write_csv(model_convergence_summary, file.path(generated_dir, "model_convergence_summary.csv"))
|
||||||
|
write_csv(model_convergence_by_dimension, file.path(generated_dir, "model_convergence_by_dimension.csv"))
|
||||||
|
write_csv(convergent_summary, file.path(generated_dir, "posterior_validation_convergent_summary.csv"))
|
||||||
|
write_csv(discriminant_summary, file.path(generated_dir, "posterior_validation_discriminant_summary.csv"))
|
||||||
|
write_csv(uncertainty_summary, file.path(generated_dir, "posterior_validation_uncertainty_summary.csv"))
|
||||||
|
write_csv(external_validation_correlations, file.path(generated_dir, "external_validation_correlations.csv"))
|
||||||
|
write_csv(construct_family_positions, file.path(generated_dir, "construct_family_positions.csv"))
|
||||||
|
write_csv(construct_temporal_stability, file.path(generated_dir, "construct_temporal_stability_flags.csv"))
|
||||||
|
write_csv(source_composition_balance, file.path(generated_dir, "source_composition_balance.csv"))
|
||||||
|
write_csv(robustness_sensitivity, file.path(generated_dir, "robustness_sensitivity.csv"))
|
||||||
|
write_csv(posterior_uncertainty, file.path(generated_dir, "posterior_uncertainty_summary.csv"))
|
||||||
|
|
||||||
|
item_counts <- item_rows %>% count(source_file, dimension, name = "items")
|
||||||
|
source_counts <- source_coverage %>% select(source_file, source, items, observations, party_years, parties, countries, min_year, max_year)
|
||||||
|
reversed_items <- filter(item_rows, reversed_for_reporting == "yes") %>% select(item, source, dimension, higher_values_indicate, observations, min_year, max_year)
|
||||||
|
conv_display <- model_convergence_summary %>% select(-any_of("source_file"))
|
||||||
|
conv_dim_display <- model_convergence_by_dimension %>% select(-any_of("source_file")) %>% mutate(across(where(is.numeric), ~ round(.x, 3)))
|
||||||
|
val_display <- bind_rows(
|
||||||
|
convergent_summary %>% transmute(diagnostic, dimension, n, pearson_r = round(r_pearson, 3), spearman_r = round(r_spearman, 3), mae = round(mae, 3), coverage = NA_real_),
|
||||||
|
uncertainty_summary %>% transmute(diagnostic, dimension, n, pearson_r = NA_real_, spearman_r = NA_real_, mae = NA_real_, coverage = round(cic, 3))
|
||||||
|
)
|
||||||
|
|
||||||
|
report_lines <- c(
|
||||||
|
"# Diagnostics report",
|
||||||
|
"",
|
||||||
|
paste0("Generated: ", format(Sys.time(), "%Y-%m-%d %H:%M:%S %Z")),
|
||||||
|
paste0("Release: ", release_version),
|
||||||
|
paste0("Model positions source: `", display_path(model_positions_file), "`"),
|
||||||
|
"",
|
||||||
|
"## Purpose",
|
||||||
|
"",
|
||||||
|
"The purpose of this report is to provide the appendix-style diagnostics that document how the released party-position estimates are constructed, checked, and validated. The main article reports the central validation evidence; this report keeps the larger technical tables with the release so readers can inspect item coverage, source coverage, coding orientation, harmonization, convergence, posterior uncertainty, and validation details in one reproducible place.",
|
||||||
|
"",
|
||||||
|
"## Overview",
|
||||||
|
"",
|
||||||
|
"This report follows the structure of the technical appendix material: data and item coverage, coding and scale orientation, party-union harmonization, construct checks, model convergence, and validation. It is generated from the model-ready inputs and completed model outputs; it is not part of the raw-data setup workflow.",
|
||||||
|
"",
|
||||||
|
"## Data and item coverage",
|
||||||
|
"",
|
||||||
|
"The model combines text-coded item counts, dimension-specific expert placements, and general left-right expert placements. Text items enter as positive/sample counts, expert items enter as aggregated ratings with scale and expert-count information, and general left-right ratings anchor the relationship between the two dimensions.",
|
||||||
|
"",
|
||||||
|
md_table(item_counts),
|
||||||
|
"",
|
||||||
|
"### Source coverage",
|
||||||
|
"",
|
||||||
|
md_table(source_counts),
|
||||||
|
"",
|
||||||
|
"## Data coding and item orientation",
|
||||||
|
"",
|
||||||
|
"All indicators are oriented toward the two reported dimensions: economic left-right and cultural cosmopolitan--traditionalist. For interpretability, generated diagnostics report whether higher observed values point toward the public high pole or are reversed for reporting. Original source variable names are preserved in the tables.",
|
||||||
|
"",
|
||||||
|
"### Reversed items",
|
||||||
|
"",
|
||||||
|
md_table(reversed_items),
|
||||||
|
"",
|
||||||
|
"## Party unions and electoral coalitions",
|
||||||
|
"",
|
||||||
|
"Alliance and union labels are handled through constituent mappings so the released party identifiers represent individual parties. Shared text evidence can inform constituent parties through the union mapping while expert data continue to constrain individual parties directly.",
|
||||||
|
"",
|
||||||
|
md_table(alliance_union_harmonization %>% head(30)),
|
||||||
|
"",
|
||||||
|
"## Party-family coverage",
|
||||||
|
"",
|
||||||
|
"Party-family classifications are used for construct-validity diagnostics and coverage summaries. The table below reports coverage in the completed model output by family code.",
|
||||||
|
"",
|
||||||
|
md_table(party_family_coverage),
|
||||||
|
"",
|
||||||
|
"### Construct-validity family means",
|
||||||
|
"",
|
||||||
|
"Substantive party-family means provide a construct-validity check: families should follow the expected ordering on the economic left-right and cultural cosmopolitan--traditionalist dimensions.",
|
||||||
|
"",
|
||||||
|
md_table(construct_family_positions %>% select(family_name, n_parties, n_obs, mean_economic, sd_economic, mean_cultural, sd_cultural) %>% mutate(across(where(is.numeric), ~ round(.x, 3)))),
|
||||||
|
"",
|
||||||
|
"### Temporal-stability flags",
|
||||||
|
"",
|
||||||
|
"The model permits movement through random walks, but unusually large one-year changes are flagged for inspection rather than treated as automatic errors.",
|
||||||
|
"",
|
||||||
|
md_table(construct_temporal_stability %>% select(party_id, country, dimension, year_from, year_to, val_from, val_to, annual_change) %>% mutate(across(where(is.numeric), ~ round(.x, 3))), n = 20),
|
||||||
|
"",
|
||||||
|
"## Model convergence diagnostics",
|
||||||
|
"",
|
||||||
|
if (nrow(model_convergence_summary) > 0) "Convergence is assessed using split R-hat and effective sample size over monitored parameters." else "Convergence summary files were not found in the configured outputs directory.",
|
||||||
|
"",
|
||||||
|
md_table(conv_display),
|
||||||
|
"",
|
||||||
|
"### Convergence by parameter group",
|
||||||
|
"",
|
||||||
|
md_table(conv_dim_display),
|
||||||
|
"",
|
||||||
|
"## Posterior uncertainty",
|
||||||
|
"",
|
||||||
|
"The completed party-position output reports posterior standard errors and interval endpoints for both dimensions. These summaries describe the typical uncertainty in the release file used by the report.",
|
||||||
|
"",
|
||||||
|
md_table(posterior_uncertainty %>% mutate(across(where(is.numeric), ~ round(.x, 3)))),
|
||||||
|
"",
|
||||||
|
"## Validation diagnostics",
|
||||||
|
"",
|
||||||
|
"The validation diagnostics combine convergent and discriminant comparisons with expert surveys, posterior predictive coverage, construct checks, and out-of-sample validation when the corresponding outputs are available.",
|
||||||
|
"",
|
||||||
|
md_table(val_display),
|
||||||
|
"",
|
||||||
|
"### Discriminant validity",
|
||||||
|
"",
|
||||||
|
md_table(discriminant_summary %>% select(-any_of("source_file")) %>% mutate(across(where(is.numeric), ~ round(.x, 3)))),
|
||||||
|
"",
|
||||||
|
"### External validation correlations",
|
||||||
|
"",
|
||||||
|
md_table(external_validation_correlations %>% select(-any_of("source_file")) %>% mutate(across(where(is.numeric), ~ round(.x, 3)))),
|
||||||
|
"",
|
||||||
|
"## Evidence-composition balance",
|
||||||
|
"",
|
||||||
|
"Evidence-composition balance checks whether estimates informed by different nearby source combinations are systematically shifted relative to rows with both text and expert evidence. The reported differences are adjusted differences on the unit scale relative to the overlapping text-and-expert reference category.",
|
||||||
|
"",
|
||||||
|
md_table(source_composition_balance),
|
||||||
|
"",
|
||||||
|
"## Robustness and sensitivity checks",
|
||||||
|
"",
|
||||||
|
"Sensitivity checks compare the released election-year estimates with source-ablation, segmentation-threshold, and item-screening variants where available. Correlations near one and small absolute differences indicate that the released estimates are stable to the corresponding design choice.",
|
||||||
|
"",
|
||||||
|
md_table(robustness_sensitivity),
|
||||||
|
"",
|
||||||
|
"## Generated tables",
|
||||||
|
"",
|
||||||
|
paste0("- `", list.files(generated_dir, pattern = "\\.csv$"), "`"),
|
||||||
|
""
|
||||||
|
)
|
||||||
|
|
||||||
|
pdf_source <- file.path(generated_dir, "diagnostics_report.Rmd")
|
||||||
|
pdf_file <- file.path(generated_dir, "diagnostics_report.pdf")
|
||||||
|
release_pdf <- file.path(release_dir, paste0("party_2d_diagnostics_report_", release_version, ".pdf"))
|
||||||
|
pdf_lines <- c(
|
||||||
|
"---",
|
||||||
|
"title: \"Diagnostics report\"",
|
||||||
|
paste0("date: \"", format(Sys.time(), "%Y-%m-%d"), "\""),
|
||||||
|
"output:",
|
||||||
|
" pdf_document:",
|
||||||
|
" toc: true",
|
||||||
|
" number_sections: true",
|
||||||
|
" latex_engine: pdflatex",
|
||||||
|
"geometry: margin=0.75in",
|
||||||
|
"fontsize: 10pt",
|
||||||
|
"header-includes:",
|
||||||
|
" - \\usepackage{booktabs}",
|
||||||
|
" - \\usepackage{longtable}",
|
||||||
|
" - \\usepackage{array}",
|
||||||
|
" - \\usepackage{pdflscape}",
|
||||||
|
" - \\setlength{\\tabcolsep}{4pt}",
|
||||||
|
" - \\renewcommand{\\arraystretch}{1.12}",
|
||||||
|
"---",
|
||||||
|
"",
|
||||||
|
"```{r setup, include=FALSE}",
|
||||||
|
"knitr::opts_chunk$set(echo = FALSE, message = FALSE, warning = FALSE)",
|
||||||
|
"print_table <- function(x, n = Inf, size = 'footnotesize') {",
|
||||||
|
" if (nrow(x) == 0) { cat('No rows available.\\n'); return(invisible(NULL)) }",
|
||||||
|
" x <- head(x, n)",
|
||||||
|
" x <- mutate(x, across(everything(), as.character))",
|
||||||
|
" x[is.na(x)] <- ''",
|
||||||
|
" names(x) <- gsub('_', ' ', names(x), fixed = TRUE)",
|
||||||
|
" cat(paste0(\"\\n\\\\begingroup\\\\\", size, \"\\n\"))",
|
||||||
|
" print(knitr::kable(x, format = 'latex', booktabs = TRUE, longtable = FALSE, digits = 3))",
|
||||||
|
" cat(\"\\n\\\\endgroup\\n\")",
|
||||||
|
"}",
|
||||||
|
"short_dim <- function(x) dplyr::recode(as.character(x), 'cultural cosmopolitan--traditionalist' = 'cultural', 'economic left-right' = 'economic', .default = as.character(x))",
|
||||||
|
"```",
|
||||||
|
"",
|
||||||
|
paste0("Generated: ", format(Sys.time(), "%Y-%m-%d %H:%M:%S %Z")),
|
||||||
|
"",
|
||||||
|
paste0("Release: ", release_version),
|
||||||
|
"",
|
||||||
|
paste0("Model positions source: `", display_path(model_positions_file), "`"),
|
||||||
|
"",
|
||||||
|
"# Purpose",
|
||||||
|
"",
|
||||||
|
"The purpose of this report is to provide the appendix-style diagnostics that document how the released party-position estimates are constructed, checked, and validated. The main article reports the central validation evidence; this report keeps the larger technical tables with the release so readers can inspect item coverage, source coverage, coding orientation, harmonization, convergence, posterior uncertainty, and validation details in one reproducible place.",
|
||||||
|
"",
|
||||||
|
"# Overview",
|
||||||
|
"",
|
||||||
|
"This report follows the structure of the technical appendix material: data and item coverage, coding and scale orientation, party-union harmonization, construct checks, model convergence, and validation. It is generated from the model-ready inputs and completed model outputs; it is not part of the raw-data setup workflow.",
|
||||||
|
"",
|
||||||
|
"# Data and item coverage",
|
||||||
|
"",
|
||||||
|
"The model combines text-coded item counts, dimension-specific expert placements, and general left-right expert placements. Text items enter as positive/sample counts, expert items enter as aggregated ratings with scale and expert-count information, and general left-right ratings anchor the relationship between the two dimensions.",
|
||||||
|
"",
|
||||||
|
"```{r item-counts, results='asis'}",
|
||||||
|
"print_table(item_counts %>% mutate(dimension = short_dim(dimension)))",
|
||||||
|
"```",
|
||||||
|
"",
|
||||||
|
"## Source coverage",
|
||||||
|
"",
|
||||||
|
"```{r source-coverage, results='asis'}",
|
||||||
|
"print_table(source_counts %>% transmute(file = recode(source_file, text_data.csv = 'text', expert.csv = 'expert', lr_data.csv = 'general LR'), source, items, obs = observations, party_years, parties, countries, years = paste0(min_year, '--', max_year)), size = 'scriptsize')",
|
||||||
|
"```",
|
||||||
|
"",
|
||||||
|
"Full item-level coverage and coding-orientation details are provided as generated CSV files listed at the end of this report.",
|
||||||
|
"",
|
||||||
|
"# Data coding and item orientation",
|
||||||
|
"",
|
||||||
|
"All indicators are oriented toward the two reported dimensions: economic left-right and cultural cosmopolitan--traditionalist. For interpretability, generated diagnostics report whether higher observed values point toward the public high pole or are reversed for reporting. Original source variable names are preserved in the tables.",
|
||||||
|
"",
|
||||||
|
"## Reversed items",
|
||||||
|
"",
|
||||||
|
"```{r reversed-items, results='asis'}",
|
||||||
|
"print_table(reversed_items %>% count(source, dimension = short_dim(dimension), higher_values_indicate, name = 'items'))",
|
||||||
|
"```",
|
||||||
|
"",
|
||||||
|
"# Party unions and electoral coalitions",
|
||||||
|
"",
|
||||||
|
"Alliance and union labels are handled through constituent mappings so the released party identifiers represent individual parties. Shared text evidence can inform constituent parties through the union mapping while expert data continue to constrain individual parties directly.",
|
||||||
|
"",
|
||||||
|
"```{r union-summary, results='asis'}",
|
||||||
|
"print_table(alliance_union_harmonization %>% transmute(metric, category, value), n = 40)",
|
||||||
|
"```",
|
||||||
|
"",
|
||||||
|
"# Party-family and construct coverage",
|
||||||
|
"",
|
||||||
|
"Party-family classifications are used for construct-validity diagnostics and coverage summaries. The table below reports coverage in the completed model output by family code.",
|
||||||
|
"",
|
||||||
|
"```{r family-coverage, results='asis'}",
|
||||||
|
"print_table(party_family_coverage %>% transmute(family, parties, party_years, countries, years = paste0(min_year, '--', max_year)))",
|
||||||
|
"```",
|
||||||
|
"",
|
||||||
|
"## Construct-validity family means",
|
||||||
|
"",
|
||||||
|
"Substantive party-family means provide a construct-validity check: families should follow the expected ordering on the economic left-right and cultural cosmopolitan--traditionalist dimensions.",
|
||||||
|
"",
|
||||||
|
"```{r construct-family, results='asis'}",
|
||||||
|
"print_table(construct_family_positions %>% transmute(family = family_name, parties = n_parties, obs = n_obs, econ_mean = round(mean_economic, 3), econ_sd = round(sd_economic, 3), cult_mean = round(mean_cultural, 3), cult_sd = round(sd_cultural, 3)), size = 'scriptsize')",
|
||||||
|
"```",
|
||||||
|
"",
|
||||||
|
"## Temporal-stability flags",
|
||||||
|
"",
|
||||||
|
"The model permits movement through random walks, but unusually large one-year changes are flagged for inspection rather than treated as automatic errors.",
|
||||||
|
"",
|
||||||
|
"```{r temporal-stability, results='asis'}",
|
||||||
|
"print_table(construct_temporal_stability %>% transmute(party = party_id, country, dim = short_dim(dimension), from = year_from, to = year_to, start = round(val_from, 3), end = round(val_to, 3), annual_change = round(annual_change, 3)), n = 12, size = 'scriptsize')",
|
||||||
|
"```",
|
||||||
|
"",
|
||||||
|
"# Model convergence diagnostics",
|
||||||
|
"",
|
||||||
|
"Convergence is assessed using split R-hat and effective sample size over monitored parameters.",
|
||||||
|
"",
|
||||||
|
"```{r convergence-summary, results='asis'}",
|
||||||
|
"print_table(conv_display)",
|
||||||
|
"```",
|
||||||
|
"",
|
||||||
|
"## Convergence by parameter group",
|
||||||
|
"",
|
||||||
|
"```{r convergence-dim, results='asis'}",
|
||||||
|
"print_table(conv_dim_display %>% mutate(dimension = short_dim(dimension)))",
|
||||||
|
"```",
|
||||||
|
"",
|
||||||
|
"# Posterior uncertainty",
|
||||||
|
"",
|
||||||
|
"The completed party-position output reports posterior standard errors and interval endpoints for both dimensions. These summaries describe the typical uncertainty in the release file used by the report.",
|
||||||
|
"",
|
||||||
|
"```{r posterior-uncertainty, results='asis'}",
|
||||||
|
"print_table(posterior_uncertainty %>% transmute(rows, parties, countries, years = paste0(min_year, '--', max_year), mean_econ_se = round(mean_economic_se, 3), median_econ_se = round(median_economic_se, 3), mean_cult_se = round(mean_cultural_se, 3), median_cult_se = round(median_cultural_se, 3)), size = 'scriptsize')",
|
||||||
|
"```",
|
||||||
|
"",
|
||||||
|
"# Validation diagnostics",
|
||||||
|
"",
|
||||||
|
"The validation diagnostics combine convergent and discriminant comparisons with expert surveys, posterior predictive coverage, construct checks, and out-of-sample validation when the corresponding outputs are available.",
|
||||||
|
"",
|
||||||
|
"```{r validation-summary, results='asis'}",
|
||||||
|
"print_table(val_display %>% mutate(dimension = short_dim(dimension)), size = 'scriptsize')",
|
||||||
|
"```",
|
||||||
|
"",
|
||||||
|
"## Discriminant validity",
|
||||||
|
"",
|
||||||
|
"```{r discriminant, results='asis'}",
|
||||||
|
"print_table(discriminant_summary %>% transmute(type, model = short_dim(model_dim), expert = expert_dim, n, pearson = round(r_pearson, 3), spearman = round(r_spearman, 3)))",
|
||||||
|
"```",
|
||||||
|
"",
|
||||||
|
"## External validation correlations",
|
||||||
|
"",
|
||||||
|
"```{r external-validation, results='asis'}",
|
||||||
|
"print_table(external_validation_correlations %>% transmute(item = var, dim = short_dim(dimension), n, r = round(pearson_r, 3), mae = round(mean_absolute_error, 3), coverage = round(coverage_95, 3)))",
|
||||||
|
"```",
|
||||||
|
"",
|
||||||
|
"# Evidence-composition balance",
|
||||||
|
"",
|
||||||
|
"Evidence-composition balance checks whether estimates informed by different nearby source combinations are systematically shifted relative to rows with both text and expert evidence. The reported differences are adjusted differences on the unit scale relative to the overlapping text-and-expert reference category.",
|
||||||
|
"",
|
||||||
|
"```{r source-balance, results='asis'}",
|
||||||
|
"print_table(source_composition_balance %>% transmute(dim = short_dim(dimension), evidence = recode(source_composition_class, both_direct_or_nearby = 'both', text_only_direct_or_nearby = 'text only', expert_only_direct_or_nearby = 'expert only', temporal_propagation = 'temporal'), ref = recode(reference_class, both_direct_or_nearby = 'both'), n, adj_diff = round(adjusted_difference, 3)))",
|
||||||
|
"```",
|
||||||
|
"",
|
||||||
|
"# Robustness and sensitivity checks",
|
||||||
|
"",
|
||||||
|
"Sensitivity checks compare the released election-year estimates with source-ablation, segmentation-threshold, and item-screening variants where available. Correlations near one and small absolute differences indicate that the released estimates are stable to the corresponding design choice.",
|
||||||
|
"",
|
||||||
|
"```{r robustness-sensitivity, results='asis'}",
|
||||||
|
"print_table(robustness_sensitivity %>% transmute(spec = specification, source = ablated_source, dim = short_dim(dimension), n = matched_n, r = round(as.numeric(correlation_with_production), 3), mean_abs = round(as.numeric(mean_abs_difference), 3), median_abs = round(as.numeric(median_abs_difference), 3), p95_abs = round(as.numeric(p95_abs_difference), 3)), size = 'scriptsize')",
|
||||||
|
"```",
|
||||||
|
"",
|
||||||
|
"# Generated tables",
|
||||||
|
"",
|
||||||
|
paste0("- `", list.files(generated_dir, pattern = "\\.csv$"), "`")
|
||||||
|
)
|
||||||
|
writeLines(pdf_lines, pdf_source)
|
||||||
|
if (!requireNamespace("rmarkdown", quietly = TRUE)) {
|
||||||
|
stop("The rmarkdown package is required to render the diagnostics PDF.")
|
||||||
|
}
|
||||||
|
rmarkdown::render(
|
||||||
|
input = pdf_source,
|
||||||
|
output_format = rmarkdown::pdf_document(toc = TRUE, number_sections = TRUE),
|
||||||
|
output_file = basename(pdf_file),
|
||||||
|
output_dir = generated_dir,
|
||||||
|
quiet = TRUE,
|
||||||
|
envir = environment()
|
||||||
|
)
|
||||||
|
invisible(file.copy(pdf_file, release_pdf, overwrite = TRUE))
|
||||||
|
unlink(c(
|
||||||
|
pdf_source,
|
||||||
|
file.path(generated_dir, "diagnostics_report.log"),
|
||||||
|
file.path(generated_dir, "diagnostics_report.aux"),
|
||||||
|
file.path(generated_dir, "diagnostics_report.out"),
|
||||||
|
file.path(generated_dir, "diagnostics_report.toc"),
|
||||||
|
file.path(generated_dir, "diagnostics_report.tex")
|
||||||
|
), force = TRUE)
|
||||||
|
|
||||||
|
generated_readme <- c(
|
||||||
|
"# Generated diagnostics",
|
||||||
|
"",
|
||||||
|
"These files are generated by:",
|
||||||
|
"",
|
||||||
|
"```bash",
|
||||||
|
"Rscript diagnostics/generate_diagnostics.R",
|
||||||
|
"```",
|
||||||
|
"",
|
||||||
|
"The command requires completed model/post-estimation outputs. If those outputs are outside the repository root, set `PARTY2D_OUTPUTS_DIR` before running the script.",
|
||||||
|
"",
|
||||||
|
"The report file is `diagnostics_report.pdf`; the same PDF is copied into `data/releases/` for the release files."
|
||||||
|
)
|
||||||
|
writeLines(generated_readme, file.path(generated_dir, "README.md"))
|
||||||
|
|
||||||
|
sha_file <- file.path(release_dir, "SHA256SUMS")
|
||||||
|
release_files_for_sha <- c(
|
||||||
|
paste0("party_2d_election_year_panel_", release_version, ".csv.gz"),
|
||||||
|
paste0("party_2d_annual_model_output_", release_version, ".csv.gz"),
|
||||||
|
basename(release_pdf)
|
||||||
|
)
|
||||||
|
existing_release_files <- release_files_for_sha[file.exists(file.path(release_dir, release_files_for_sha))]
|
||||||
|
sha_lines <- vapply(existing_release_files, function(f) {
|
||||||
|
old <- getwd()
|
||||||
|
on.exit(setwd(old), add = TRUE)
|
||||||
|
setwd(release_dir)
|
||||||
|
system2("sha256sum", f, stdout = TRUE)
|
||||||
|
}, character(1))
|
||||||
|
writeLines(sha_lines, sha_file)
|
||||||
|
|
||||||
|
message("Diagnostics written to diagnostics/generated")
|
||||||
|
message("Release diagnostics PDF written to ", release_pdf)
|
||||||
@@ -0,0 +1,11 @@
|
|||||||
|
# Generated diagnostics
|
||||||
|
|
||||||
|
These files are generated by:
|
||||||
|
|
||||||
|
```bash
|
||||||
|
Rscript diagnostics/generate_diagnostics.R
|
||||||
|
```
|
||||||
|
|
||||||
|
The command requires completed model/post-estimation outputs. If those outputs are outside the repository root, set `PARTY2D_OUTPUTS_DIR` before running the script.
|
||||||
|
|
||||||
|
The report file is `diagnostics_report.pdf`; the same PDF is copied into `data/releases/` for the release files.
|
||||||
@@ -0,0 +1,39 @@
|
|||||||
|
metric,category,value
|
||||||
|
constituent_mappings,all,106
|
||||||
|
unique_union_or_alliance_ids,all,76
|
||||||
|
unique_constituent_party_ids,all,100
|
||||||
|
mappings_by_country,AR,5
|
||||||
|
mappings_by_country,AU,2
|
||||||
|
mappings_by_country,BA,1
|
||||||
|
mappings_by_country,BE,4
|
||||||
|
mappings_by_country,BG,5
|
||||||
|
mappings_by_country,BR,6
|
||||||
|
mappings_by_country,CH,1
|
||||||
|
mappings_by_country,CL,8
|
||||||
|
mappings_by_country,CZ,4
|
||||||
|
mappings_by_country,DE,3
|
||||||
|
mappings_by_country,EC,1
|
||||||
|
mappings_by_country,EE,3
|
||||||
|
mappings_by_country,ES,3
|
||||||
|
mappings_by_country,FR,4
|
||||||
|
mappings_by_country,GR,3
|
||||||
|
mappings_by_country,HR,4
|
||||||
|
mappings_by_country,IE,1
|
||||||
|
mappings_by_country,IL,2
|
||||||
|
mappings_by_country,IT,8
|
||||||
|
mappings_by_country,LK,3
|
||||||
|
mappings_by_country,LT,4
|
||||||
|
mappings_by_country,LU,1
|
||||||
|
mappings_by_country,LV,6
|
||||||
|
mappings_by_country,MD,1
|
||||||
|
mappings_by_country,ME,1
|
||||||
|
mappings_by_country,MX,2
|
||||||
|
mappings_by_country,NL,1
|
||||||
|
mappings_by_country,NZ,1
|
||||||
|
mappings_by_country,PE,1
|
||||||
|
mappings_by_country,PL,3
|
||||||
|
mappings_by_country,RO,6
|
||||||
|
mappings_by_country,RS,3
|
||||||
|
mappings_by_country,SK,4
|
||||||
|
mappings_by_country,UY,1
|
||||||
|
mappings_by_status,implemented,106
|
||||||
|
@@ -0,0 +1,8 @@
|
|||||||
|
family,n_parties,n_obs,mean_economic,sd_economic,mean_cultural,sd_cultural,family_name
|
||||||
|
com,49,1241,0.12001647469327872,0.0901861773223108,0.3834027613298184,0.18041637351915937,Communist/Far Left
|
||||||
|
eco,30,723,0.26720980626115975,0.11467133878653364,0.2514849926574344,0.09223964864606182,Green/Ecological
|
||||||
|
soc,86,2892,0.3305567447692131,0.1240506242334158,0.3777170994931328,0.13783788204153472,Social Democratic
|
||||||
|
chr,41,1596,0.5961213268671609,0.11930537492981286,0.5293978691891922,0.13964161332639527,Christian Democratic
|
||||||
|
right,50,975,0.6222552715406671,0.18991001824381296,0.720426765976975,0.15147355617267264,Radical Right
|
||||||
|
con,82,2425,0.6519453485719768,0.17791542785462533,0.5375091531921928,0.14258881749137706,Conservative
|
||||||
|
lib,81,2063,0.6569946895012412,0.15224833699588255,0.37069062594934565,0.13158056834168683,Liberal
|
||||||
|
@@ -0,0 +1,60 @@
|
|||||||
|
party_id,country,dimension,year_from,year_to,val_from,val_to,change,annual_change
|
||||||
|
556,LT,cultural cosmopolitan--traditionalist,2019,2020,0.77107951125,0.5289456789999999,0.24213383225000007,0.24213383225000007
|
||||||
|
1663,IL,cultural cosmopolitan--traditionalist,2021,2022,0.1423029388125,0.35615504375,0.2138521049375,0.2138521049375
|
||||||
|
455,IL,cultural cosmopolitan--traditionalist,1997,1998,0.456850958125,0.6432782493750001,0.18642729125000007,0.18642729125000007
|
||||||
|
455,IL,cultural cosmopolitan--traditionalist,1996,1997,0.27663313025,0.456850958125,0.180217827875,0.180217827875
|
||||||
|
8393,LV,cultural cosmopolitan--traditionalist,2018,2019,0.4386382122500001,0.2627366591625,0.17590155308750005,0.17590155308750005
|
||||||
|
281,BE,cultural cosmopolitan--traditionalist,1977,1978,0.6706878695,0.843416257375,0.172728387875,0.172728387875
|
||||||
|
556,LT,cultural cosmopolitan--traditionalist,2001,2002,0.318926108375,0.49125159325,0.172325484875,0.172325484875
|
||||||
|
964,IS,economic left-right,2016,2017,0.691268056,0.520838928125,0.17042912787499995,0.17042912787499995
|
||||||
|
298,NL,cultural cosmopolitan--traditionalist,2018,2019,0.760919900625,0.592874692125,0.16804520850000004,0.16804520850000004
|
||||||
|
901,FI,economic left-right,1992,1993,0.486145510375,0.64721287575,0.161067365375,0.161067365375
|
||||||
|
467,SI,cultural cosmopolitan--traditionalist,2018,2019,0.563748886625,0.4041139849999999,0.15963490162500005,0.15963490162500005
|
||||||
|
1221,IT,economic left-right,2007,2008,0.559621162375,0.400493563,0.15912759937500004,0.15912759937500004
|
||||||
|
901,FI,economic left-right,1991,1992,0.327277968125,0.486145510375,0.15886754225,0.15886754225
|
||||||
|
455,IL,cultural cosmopolitan--traditionalist,1998,1999,0.6432782493750001,0.7970417803750001,0.15376353099999995,0.15376353099999995
|
||||||
|
1221,IT,economic left-right,2006,2007,0.7086087693750001,0.559621162375,0.14898760700000002,0.14898760700000002
|
||||||
|
2211,UA,economic left-right,2006,2007,0.416653024,0.5619799204999999,0.1453268964999999,0.1453268964999999
|
||||||
|
631,CH,economic left-right,2018,2019,0.613235049875,0.7569515025,0.143716452625,0.143716452625
|
||||||
|
298,NL,cultural cosmopolitan--traditionalist,2019,2020,0.592874692125,0.734536642,0.14166194987500005,0.14166194987500005
|
||||||
|
556,LT,cultural cosmopolitan--traditionalist,2000,2001,0.180505337875,0.318926108375,0.1384207705,0.1384207705
|
||||||
|
1417,IL,cultural cosmopolitan--traditionalist,1968,1969,0.49841048887499995,0.635275993875,0.13686550500000003,0.13686550500000003
|
||||||
|
81,ES,cultural cosmopolitan--traditionalist,1999,2000,0.44213168437499994,0.30819645050000005,0.1339352338749999,0.1339352338749999
|
||||||
|
1417,IL,cultural cosmopolitan--traditionalist,1967,1968,0.364764852,0.49841048887499995,0.13364563687499997,0.13364563687499997
|
||||||
|
901,FI,economic left-right,1993,1994,0.64721287575,0.78024709825,0.13303422250000008,0.13303422250000008
|
||||||
|
409,SE,economic left-right,2018,2019,0.4993397851250001,0.6246204093750001,0.12528062425000003,0.12528062425000003
|
||||||
|
48,GR,cultural cosmopolitan--traditionalist,1999,2000,0.471858772375,0.594323158,0.122464385625,0.122464385625
|
||||||
|
212,DK,cultural cosmopolitan--traditionalist,2014,2015,0.339423171125,0.459806116125,0.12038294500000002,0.12038294500000002
|
||||||
|
2415,IT,cultural cosmopolitan--traditionalist,2006,2007,0.6143681051250001,0.49546903375,0.11889907137500004,0.11889907137500004
|
||||||
|
1369,IT,cultural cosmopolitan--traditionalist,2013,2014,0.7406959332499999,0.6231419237500002,0.11755400949999972,0.11755400949999972
|
||||||
|
5852,IS,cultural cosmopolitan--traditionalist,2018,2019,0.336059923125,0.45280653625,0.116746613125,0.116746613125
|
||||||
|
1417,IL,cultural cosmopolitan--traditionalist,1966,1967,0.2487840847,0.364764852,0.11598076729999995,0.11598076729999995
|
||||||
|
828,NL,cultural cosmopolitan--traditionalist,2019,2020,0.399398152875,0.5144400794999999,0.11504192662499996,0.11504192662499996
|
||||||
|
2415,IT,cultural cosmopolitan--traditionalist,2007,2008,0.49546903375,0.3804985986625,0.1149704350875,0.1149704350875
|
||||||
|
828,NL,cultural cosmopolitan--traditionalist,2020,2021,0.5144400794999999,0.6280684987500001,0.11362841925000022,0.11362841925000022
|
||||||
|
573,DE,cultural cosmopolitan--traditionalist,2024,2025,0.34241723825000003,0.4558404575,0.11342321924999998,0.11342321924999998
|
||||||
|
1424,BE,economic left-right,1977,1978,0.7125746831249999,0.825810219125,0.11323553600000004,0.11323553600000004
|
||||||
|
1173,NO,cultural cosmopolitan--traditionalist,2018,2019,0.353365422125,0.46349993075,0.11013450862499996,0.11013450862499996
|
||||||
|
1651,GR,economic left-right,2013,2014,0.441388039625,0.551427035875,0.11003899624999997,0.11003899624999997
|
||||||
|
1660,GR,cultural cosmopolitan--traditionalist,2012,2013,0.70537148075,0.8146704603749999,0.1092989796249999,0.1092989796249999
|
||||||
|
433,FR,economic left-right,2018,2019,0.433590134625,0.542142367875,0.10855223325000002,0.10855223325000002
|
||||||
|
1002,GB,cultural cosmopolitan--traditionalist,2014,2015,0.3184758865,0.2101810030625,0.10829488343750002,0.10829488343750002
|
||||||
|
623,AR,economic left-right,1989,1990,0.4145806991249999,0.5228341057499999,0.10825340662499994,0.10825340662499994
|
||||||
|
1305,RO,economic left-right,2000,2001,0.553705481625,0.446037918375,0.10766756325,0.10766756325
|
||||||
|
298,NL,cultural cosmopolitan--traditionalist,2020,2021,0.734536642,0.84215486075,0.10761821875,0.10761821875
|
||||||
|
1359,PT,economic left-right,2004,2005,0.495169041875,0.602184326875,0.10701528500000002,0.10701528500000002
|
||||||
|
1651,GR,economic left-right,2012,2013,0.3355888995,0.441388039625,0.10579914012500002,0.10579914012500002
|
||||||
|
623,AR,economic left-right,1990,1991,0.5228341057499999,0.62815575375,0.1053216480000001,0.1053216480000001
|
||||||
|
1305,RO,economic left-right,2001,2002,0.446037918375,0.3412955855,0.104742332875,0.104742332875
|
||||||
|
455,IL,cultural cosmopolitan--traditionalist,1991,1992,0.5368312063749999,0.432172542875,0.10465866349999992,0.10465866349999992
|
||||||
|
2415,IT,cultural cosmopolitan--traditionalist,2008,2009,0.3804985986625,0.2762634036875,0.104235194975,0.104235194975
|
||||||
|
599,AT,economic left-right,2008,2009,0.4633838822500001,0.567595171125,0.10421128887499996,0.10421128887499996
|
||||||
|
669,CH,cultural cosmopolitan--traditionalist,2016,2017,0.514819538625,0.410640874875,0.10417866374999996,0.10417866374999996
|
||||||
|
5852,IS,cultural cosmopolitan--traditionalist,2017,2018,0.23270289265,0.336059923125,0.10335703047499996,0.10335703047499996
|
||||||
|
669,CH,cultural cosmopolitan--traditionalist,2015,2016,0.617986882875,0.514819538625,0.10316734425000008,0.10316734425000008
|
||||||
|
48,GR,cultural cosmopolitan--traditionalist,2010,2011,0.569647428,0.6723034049999999,0.10265597699999984,0.10265597699999984
|
||||||
|
1221,IT,economic left-right,2008,2009,0.400493563,0.50303734575,0.10254378275000003,0.10254378275000003
|
||||||
|
1651,GR,economic left-right,2014,2015,0.551427035875,0.6535508147500001,0.10212377887500013,0.10212377887500013
|
||||||
|
1221,IT,economic left-right,2009,2010,0.50303734575,0.604409204875,0.10137185912500002,0.10137185912500002
|
||||||
|
975,SI,economic left-right,1990,1991,0.579305296625,0.6803467895000002,0.1010414928750002,0.1010414928750002
|
||||||
|
338,AU,economic left-right,1992,1993,0.791994626125,0.6916490538750002,0.10034557224999983,0.10034557224999983
|
||||||
|
Binary file not shown.
@@ -0,0 +1,11 @@
|
|||||||
|
var,dimension,n,pearson_r,mean_absolute_error,coverage_95
|
||||||
|
culsup_vparty,cultural cosmopolitan--traditionalist,536,0.8121247297636631,0.12821713597308768,0.3843283582089552
|
||||||
|
galtan_ches,cultural cosmopolitan--traditionalist,222,0.9588005926603757,0.07875607868037135,0.5225225225225225
|
||||||
|
gender_vparty,cultural cosmopolitan--traditionalist,545,0.5626122704047269,0.16947520363543578,0.28990825688073396
|
||||||
|
immig_vparty,cultural cosmopolitan--traditionalist,537,0.7429394940511392,0.10311559715251396,0.4897579143389199
|
||||||
|
lgbt_vparty,cultural cosmopolitan--traditionalist,541,0.7941178030023652,0.09410224229993068,0.5508317929759704
|
||||||
|
relig_vparty,cultural cosmopolitan--traditionalist,548,0.6757229503671286,0.30309927660661495,0.04744525547445255
|
||||||
|
lrecon_ches,economic left-right,223,0.9739626905522167,0.05518814853885153,0.8116591928251121
|
||||||
|
lrecon_poppa,economic left-right,74,0.9799670973969279,0.0660246477855859,0.6621621621621622
|
||||||
|
lrecon_vparty,economic left-right,534,0.8664105550524236,0.08828332773956499,0.6741573033707865
|
||||||
|
welf_vparty,economic left-right,534,0.6821895613302613,0.17587920065205523,0.36329588014981273
|
||||||
|
@@ -0,0 +1,33 @@
|
|||||||
|
source_file,item,source,dimension,type_low,type_high,higher_values_indicate,reversed_for_reporting,observations,party_years,parties,countries,min_year,max_year
|
||||||
|
expert.csv,galtan_ches,CHES,cultural cosmopolitan--traditionalist,cosmopolitan,traditional,traditional,no,1319,1319,389,44,1999,2024
|
||||||
|
expert.csv,lrecon_ches,CHES,economic left-right,pro_welfare,pro_market,pro_market,no,1320,1320,390,44,1999,2024
|
||||||
|
expert.csv,libcon_gps,GPS,cultural cosmopolitan--traditionalist,cosmopolitan,traditional,traditional,no,269,269,269,61,2019,2019
|
||||||
|
expert.csv,lrecon_gps,GPS,economic left-right,pro_welfare,pro_market,pro_market,no,269,269,269,62,2019,2019
|
||||||
|
expert.csv,lrecon_poppa,POPPA,economic left-right,pro_welfare,pro_market,pro_market,no,413,413,225,31,2018,2023
|
||||||
|
expert.csv,culsup_vparty,V-Party,cultural cosmopolitan--traditionalist,cosmopolitan,traditional,traditional,no,3076,3076,589,65,1970,2019
|
||||||
|
expert.csv,gender_vparty,V-Party,cultural cosmopolitan--traditionalist,cosmopolitan,traditional,traditional,no,3043,3043,586,65,1970,2019
|
||||||
|
expert.csv,immig_vparty,V-Party,cultural cosmopolitan--traditionalist,cosmopolitan,traditional,traditional,no,3076,3076,589,65,1970,2019
|
||||||
|
expert.csv,lgbt_vparty,V-Party,cultural cosmopolitan--traditionalist,cosmopolitan,traditional,traditional,no,3076,3076,589,65,1970,2019
|
||||||
|
expert.csv,relig_vparty,V-Party,cultural cosmopolitan--traditionalist,cosmopolitan,traditional,traditional,no,3076,3076,589,65,1970,2019
|
||||||
|
expert.csv,lrecon_vparty,V-Party,economic left-right,pro_welfare,pro_market,pro_market,no,3075,3075,588,65,1970,2019
|
||||||
|
expert.csv,welf_vparty,V-Party,economic left-right,pro_welfare,pro_market,pro_market,no,3066,3066,585,65,1970,2019
|
||||||
|
lr_data.csv,lr_ches,CHES,general left-right,NA,NA,source-coded left-right,no,1320,1320,390,44,1999,2024
|
||||||
|
lr_data.csv,lr_morgan,Morgan,general left-right,NA,NA,source-coded left-right,no,471,471,72,11,1945,1973
|
||||||
|
lr_data.csv,lr_poppa,POPPA,general left-right,NA,NA,source-coded left-right,no,416,416,225,31,2018,2023
|
||||||
|
text_data.csv,conservative_morality_manifesto,Manifesto Project,cultural cosmopolitan--traditionalist,cosmopolitan,traditional,traditional,no,4501,4501,713,65,1920,2025
|
||||||
|
text_data.csv,internationalism_manifesto,Manifesto Project,cultural cosmopolitan--traditionalist,traditional,cosmopolitan,cosmopolitan,yes,4501,4501,713,65,1920,2025
|
||||||
|
text_data.csv,multiculturalism_manifesto,Manifesto Project,cultural cosmopolitan--traditionalist,traditional,cosmopolitan,cosmopolitan,yes,4501,4501,713,65,1920,2025
|
||||||
|
text_data.csv,national_identity_manifesto,Manifesto Project,cultural cosmopolitan--traditionalist,cosmopolitan,traditional,traditional,no,4501,4501,713,65,1920,2025
|
||||||
|
text_data.csv,economic_intervention_manifesto,Manifesto Project,economic left-right,pro_market,pro_welfare,pro_welfare,yes,4501,4501,713,65,1920,2025
|
||||||
|
text_data.csv,economic_liberalization_manifesto,Manifesto Project,economic left-right,pro_welfare,pro_market,pro_market,no,4501,4501,713,65,1920,2025
|
||||||
|
text_data.csv,market_regulation_manifesto,Manifesto Project,economic left-right,pro_welfare,pro_market,pro_market,no,4501,4501,713,65,1920,2025
|
||||||
|
text_data.csv,social_services_manifesto,Manifesto Project,economic left-right,pro_market,pro_welfare,pro_welfare,yes,4501,4501,713,65,1920,2025
|
||||||
|
text_data.csv,cultlib_poldem,PolDem,cultural cosmopolitan--traditionalist,traditional,cosmopolitan,cosmopolitan,yes,299,299,78,15,1972,2017
|
||||||
|
text_data.csv,defense_poldem,PolDem,cultural cosmopolitan--traditionalist,cosmopolitan,traditional,traditional,no,243,243,67,15,1972,2017
|
||||||
|
text_data.csv,euro_poldem,PolDem,cultural cosmopolitan--traditionalist,traditional,cosmopolitan,cosmopolitan,yes,93,93,44,13,1978,2017
|
||||||
|
text_data.csv,europe_poldem,PolDem,cultural cosmopolitan--traditionalist,traditional,cosmopolitan,cosmopolitan,yes,217,217,66,15,1972,2017
|
||||||
|
text_data.csv,immig_poldem,PolDem,cultural cosmopolitan--traditionalist,traditional,cosmopolitan,cosmopolitan,yes,236,236,67,14,1972,2017
|
||||||
|
text_data.csv,nationalism_poldem,PolDem,cultural cosmopolitan--traditionalist,cosmopolitan,traditional,traditional,no,131,131,60,15,1974,2017
|
||||||
|
text_data.csv,security_poldem,PolDem,cultural cosmopolitan--traditionalist,cosmopolitan,traditional,traditional,no,265,265,78,15,1972,2017
|
||||||
|
text_data.csv,ecolib_poldem,PolDem,economic left-right,pro_welfare,pro_market,pro_market,no,361,361,85,15,1972,2017
|
||||||
|
text_data.csv,welfare_poldem,PolDem,economic left-right,pro_market,pro_welfare,pro_welfare,yes,349,349,83,15,1972,2017
|
||||||
|
@@ -0,0 +1,33 @@
|
|||||||
|
source_file,item,source,dimension,type_low,type_high,higher_values_indicate,reversed_for_reporting,observations,party_years,parties,countries,min_year,max_year
|
||||||
|
expert.csv,galtan_ches,CHES,cultural cosmopolitan--traditionalist,cosmopolitan,traditional,traditional,no,1319,1319,389,44,1999,2024
|
||||||
|
expert.csv,lrecon_ches,CHES,economic left-right,pro_welfare,pro_market,pro_market,no,1320,1320,390,44,1999,2024
|
||||||
|
expert.csv,libcon_gps,GPS,cultural cosmopolitan--traditionalist,cosmopolitan,traditional,traditional,no,269,269,269,61,2019,2019
|
||||||
|
expert.csv,lrecon_gps,GPS,economic left-right,pro_welfare,pro_market,pro_market,no,269,269,269,62,2019,2019
|
||||||
|
expert.csv,lrecon_poppa,POPPA,economic left-right,pro_welfare,pro_market,pro_market,no,413,413,225,31,2018,2023
|
||||||
|
expert.csv,culsup_vparty,V-Party,cultural cosmopolitan--traditionalist,cosmopolitan,traditional,traditional,no,3076,3076,589,65,1970,2019
|
||||||
|
expert.csv,gender_vparty,V-Party,cultural cosmopolitan--traditionalist,cosmopolitan,traditional,traditional,no,3043,3043,586,65,1970,2019
|
||||||
|
expert.csv,immig_vparty,V-Party,cultural cosmopolitan--traditionalist,cosmopolitan,traditional,traditional,no,3076,3076,589,65,1970,2019
|
||||||
|
expert.csv,lgbt_vparty,V-Party,cultural cosmopolitan--traditionalist,cosmopolitan,traditional,traditional,no,3076,3076,589,65,1970,2019
|
||||||
|
expert.csv,relig_vparty,V-Party,cultural cosmopolitan--traditionalist,cosmopolitan,traditional,traditional,no,3076,3076,589,65,1970,2019
|
||||||
|
expert.csv,lrecon_vparty,V-Party,economic left-right,pro_welfare,pro_market,pro_market,no,3075,3075,588,65,1970,2019
|
||||||
|
expert.csv,welf_vparty,V-Party,economic left-right,pro_welfare,pro_market,pro_market,no,3066,3066,585,65,1970,2019
|
||||||
|
lr_data.csv,lr_ches,CHES,general left-right,NA,NA,source-coded left-right,no,1320,1320,390,44,1999,2024
|
||||||
|
lr_data.csv,lr_morgan,Morgan,general left-right,NA,NA,source-coded left-right,no,471,471,72,11,1945,1973
|
||||||
|
lr_data.csv,lr_poppa,POPPA,general left-right,NA,NA,source-coded left-right,no,416,416,225,31,2018,2023
|
||||||
|
text_data.csv,conservative_morality_manifesto,Manifesto Project,cultural cosmopolitan--traditionalist,cosmopolitan,traditional,traditional,no,4501,4501,713,65,1920,2025
|
||||||
|
text_data.csv,internationalism_manifesto,Manifesto Project,cultural cosmopolitan--traditionalist,traditional,cosmopolitan,cosmopolitan,yes,4501,4501,713,65,1920,2025
|
||||||
|
text_data.csv,multiculturalism_manifesto,Manifesto Project,cultural cosmopolitan--traditionalist,traditional,cosmopolitan,cosmopolitan,yes,4501,4501,713,65,1920,2025
|
||||||
|
text_data.csv,national_identity_manifesto,Manifesto Project,cultural cosmopolitan--traditionalist,cosmopolitan,traditional,traditional,no,4501,4501,713,65,1920,2025
|
||||||
|
text_data.csv,economic_intervention_manifesto,Manifesto Project,economic left-right,pro_market,pro_welfare,pro_welfare,yes,4501,4501,713,65,1920,2025
|
||||||
|
text_data.csv,economic_liberalization_manifesto,Manifesto Project,economic left-right,pro_welfare,pro_market,pro_market,no,4501,4501,713,65,1920,2025
|
||||||
|
text_data.csv,market_regulation_manifesto,Manifesto Project,economic left-right,pro_welfare,pro_market,pro_market,no,4501,4501,713,65,1920,2025
|
||||||
|
text_data.csv,social_services_manifesto,Manifesto Project,economic left-right,pro_market,pro_welfare,pro_welfare,yes,4501,4501,713,65,1920,2025
|
||||||
|
text_data.csv,cultlib_poldem,PolDem,cultural cosmopolitan--traditionalist,traditional,cosmopolitan,cosmopolitan,yes,299,299,78,15,1972,2017
|
||||||
|
text_data.csv,defense_poldem,PolDem,cultural cosmopolitan--traditionalist,cosmopolitan,traditional,traditional,no,243,243,67,15,1972,2017
|
||||||
|
text_data.csv,euro_poldem,PolDem,cultural cosmopolitan--traditionalist,traditional,cosmopolitan,cosmopolitan,yes,93,93,44,13,1978,2017
|
||||||
|
text_data.csv,europe_poldem,PolDem,cultural cosmopolitan--traditionalist,traditional,cosmopolitan,cosmopolitan,yes,217,217,66,15,1972,2017
|
||||||
|
text_data.csv,immig_poldem,PolDem,cultural cosmopolitan--traditionalist,traditional,cosmopolitan,cosmopolitan,yes,236,236,67,14,1972,2017
|
||||||
|
text_data.csv,nationalism_poldem,PolDem,cultural cosmopolitan--traditionalist,cosmopolitan,traditional,traditional,no,131,131,60,15,1974,2017
|
||||||
|
text_data.csv,security_poldem,PolDem,cultural cosmopolitan--traditionalist,cosmopolitan,traditional,traditional,no,265,265,78,15,1972,2017
|
||||||
|
text_data.csv,ecolib_poldem,PolDem,economic left-right,pro_welfare,pro_market,pro_market,no,361,361,85,15,1972,2017
|
||||||
|
text_data.csv,welfare_poldem,PolDem,economic left-right,pro_market,pro_welfare,pro_welfare,yes,349,349,83,15,1972,2017
|
||||||
|
@@ -0,0 +1,9 @@
|
|||||||
|
dimension,parameters,mean_rhat,max_rhat,min_ess_bulk,mean_ess_bulk
|
||||||
|
cultural cosmopolitan--traditionalist,17585,1.0006424726753271,1.0063876592524996,810.5088684922246,6889.24553993031
|
||||||
|
economic left-right,17585,1.0003816350246089,1.0041031832256586,670.7084347577414,5573.455663381927
|
||||||
|
lr_country_offset,65,1.0007983450724103,1.0042741902159762,1078.1975001602086,5434.867801353405
|
||||||
|
lr_decade_offset,8,1.0003240812347383,1.0011721739400503,2069.6763143867825,3372.015629437053
|
||||||
|
lr_sigma,3,1.0010806271267045,1.001817706171186,1956.6110527635497,2582.401306906664
|
||||||
|
lr_source_offset,3,1.0001329930739762,1.0002839044194227,3477.5194916377077,3814.373810300655
|
||||||
|
lr_weight,3,1.0023083674707072,1.0023587853631075,1540.8061256188755,1560.2173784478186
|
||||||
|
mean_sigma,6,1.00453801817315,1.0092946499280384,534.8105812273362,918.5017536375844
|
||||||
|
@@ -0,0 +1,7 @@
|
|||||||
|
metric,count,percentage
|
||||||
|
R-hat < 1.01,35258,100
|
||||||
|
R-hat 1.01-1.05,0,0
|
||||||
|
R-hat > 1.05,0,0
|
||||||
|
ESS > 1000,34853,98.85
|
||||||
|
ESS 400-1000,405,1.15
|
||||||
|
ESS < 400,0,0
|
||||||
|
@@ -0,0 +1,11 @@
|
|||||||
|
family,parties,party_years,countries,min_year,max_year
|
||||||
|
soc,86,2892,37,1944,2025
|
||||||
|
con,82,2425,34,1944,2024
|
||||||
|
lib,81,2063,35,1944,2025
|
||||||
|
chr,41,1596,24,1945,2025
|
||||||
|
com,49,1241,27,1944,2025
|
||||||
|
right,50,975,26,1946,2025
|
||||||
|
eco,30,723,24,1960,2024
|
||||||
|
agr,12,592,10,1944,2024
|
||||||
|
spec,25,520,14,1949,2024
|
||||||
|
other,2,40,2,1992,2024
|
||||||
|
File diff suppressed because it is too large
Load Diff
@@ -0,0 +1,2 @@
|
|||||||
|
rows,parties,countries,min_year,max_year,mean_economic_se,median_economic_se,mean_cultural_se,median_cultural_se
|
||||||
|
17585,708,65,1944,2025,0.06480914153714339,0.06297344495207625,0.05830289918910167,0.05511815999995913
|
||||||
|
@@ -0,0 +1,3 @@
|
|||||||
|
dimension,r_pearson,r_spearman,ci_lower,ci_upper,mae,rmse,n,diagnostic
|
||||||
|
economic left-right,0.9040661964265828,0.9011355686083358,0.8986678208050985,0.9091907336941026,0.07569891833571618,0.09876331318961644,4637,convergent validity
|
||||||
|
cultural cosmopolitan--traditionalist,0.9598645558100498,0.96729961069876,0.9555654780202564,0.96375540322017,0.07890038540202159,0.0987420458613568,1425,convergent validity
|
||||||
|
@@ -0,0 +1,5 @@
|
|||||||
|
model_dim,expert_dim,r_pearson,r_spearman,n,type,diagnostic
|
||||||
|
economic left-right,economic,0.9040661964265828,0.9011355686083358,4637,convergent,discriminant validity
|
||||||
|
cultural cosmopolitan--traditionalist,economic,0.42228411618900114,0.4352014815908644,4637,discriminant,discriminant validity
|
||||||
|
cultural cosmopolitan--traditionalist,cultural cosmopolitan--traditionalist,0.9598645558100498,0.96729961069876,1425,convergent,discriminant validity
|
||||||
|
economic left-right,cultural cosmopolitan--traditionalist,0.39267427502305646,0.3979530504947876,1425,discriminant,discriminant validity
|
||||||
|
@@ -0,0 +1,3 @@
|
|||||||
|
dimension,cic,cic_pct,ci_lower,ci_upper,n,covered,diagnostic
|
||||||
|
economic left-right,0.8987256874580818,89.9,0.8916705123992721,0.9053701433508112,7455,6700,posterior predictive coverage
|
||||||
|
cultural cosmopolitan--traditionalist,0.8477379496750113,84.8,0.8420030513784423,0.8533009527490912,15539,13173,posterior predictive coverage
|
||||||
|
@@ -0,0 +1,10 @@
|
|||||||
|
source_file,item,source,dimension,type_low,type_high,higher_values_indicate,reversed_for_reporting,observations,party_years,parties,countries,min_year,max_year
|
||||||
|
text_data.csv,internationalism_manifesto,Manifesto Project,cultural cosmopolitan--traditionalist,traditional,cosmopolitan,cosmopolitan,yes,4501,4501,713,65,1920,2025
|
||||||
|
text_data.csv,multiculturalism_manifesto,Manifesto Project,cultural cosmopolitan--traditionalist,traditional,cosmopolitan,cosmopolitan,yes,4501,4501,713,65,1920,2025
|
||||||
|
text_data.csv,economic_intervention_manifesto,Manifesto Project,economic left-right,pro_market,pro_welfare,pro_welfare,yes,4501,4501,713,65,1920,2025
|
||||||
|
text_data.csv,social_services_manifesto,Manifesto Project,economic left-right,pro_market,pro_welfare,pro_welfare,yes,4501,4501,713,65,1920,2025
|
||||||
|
text_data.csv,cultlib_poldem,PolDem,cultural cosmopolitan--traditionalist,traditional,cosmopolitan,cosmopolitan,yes,299,299,78,15,1972,2017
|
||||||
|
text_data.csv,euro_poldem,PolDem,cultural cosmopolitan--traditionalist,traditional,cosmopolitan,cosmopolitan,yes,93,93,44,13,1978,2017
|
||||||
|
text_data.csv,europe_poldem,PolDem,cultural cosmopolitan--traditionalist,traditional,cosmopolitan,cosmopolitan,yes,217,217,66,15,1972,2017
|
||||||
|
text_data.csv,immig_poldem,PolDem,cultural cosmopolitan--traditionalist,traditional,cosmopolitan,cosmopolitan,yes,236,236,67,14,1972,2017
|
||||||
|
text_data.csv,welfare_poldem,PolDem,economic left-right,pro_market,pro_welfare,pro_welfare,yes,349,349,83,15,1972,2017
|
||||||
|
@@ -0,0 +1,7 @@
|
|||||||
|
specification,ablated_source,dimension,matched_n,correlation_with_production,mean_abs_difference,median_abs_difference,p95_abs_difference,mean_interval_width_production,mean_interval_width_ablation
|
||||||
|
Source ablation,V Party,economic left-right,4248,0.954,0.05,0.033,0.159,NA,NA
|
||||||
|
Source ablation,V Party,cultural cosmopolitan--traditionalist,4248,0.898,0.08,0.062,0.228,0.202,0.323
|
||||||
|
Gap threshold 5 years,NA,economic left-right,4244,0.999,0.003,0.002,0.007,NA,NA
|
||||||
|
Gap threshold 5 years,NA,cultural cosmopolitan--traditionalist,4244,0.999,0.002,0.001,0.006,NA,NA
|
||||||
|
Gap threshold 10 years,NA,economic left-right,4265,0.999,0.006,0.006,0.011,NA,NA
|
||||||
|
Gap threshold 10 years,NA,cultural cosmopolitan--traditionalist,4265,1,0.003,0.002,0.006,NA,NA
|
||||||
|
@@ -0,0 +1,7 @@
|
|||||||
|
dimension,source_composition_class,reference_class,n,adjusted_difference
|
||||||
|
economic left-right,text_only_direct_or_nearby,both_direct_or_nearby,4916,0.014
|
||||||
|
economic left-right,expert_only_direct_or_nearby,both_direct_or_nearby,4916,-0.015
|
||||||
|
economic left-right,temporal_propagation,both_direct_or_nearby,4916,-0.046
|
||||||
|
cultural cosmopolitan--traditionalist,text_only_direct_or_nearby,both_direct_or_nearby,4916,0.017
|
||||||
|
cultural cosmopolitan--traditionalist,expert_only_direct_or_nearby,both_direct_or_nearby,4916,0.038
|
||||||
|
cultural cosmopolitan--traditionalist,temporal_propagation,both_direct_or_nearby,4916,-0.042
|
||||||
|
@@ -0,0 +1,10 @@
|
|||||||
|
source_file,source,items,observations,party_years,parties,countries,min_year,max_year
|
||||||
|
expert.csv,CHES,2,2639,1320,390,44,1999,2024
|
||||||
|
expert.csv,GPS,2,538,271,271,62,2019,2019
|
||||||
|
expert.csv,POPPA,1,413,413,225,31,2018,2023
|
||||||
|
expert.csv,V-Party,7,21488,3076,589,65,1970,2019
|
||||||
|
lr_data.csv,CHES,1,1320,1320,390,44,1999,2024
|
||||||
|
lr_data.csv,Morgan,1,471,471,72,11,1945,1973
|
||||||
|
lr_data.csv,POPPA,1,416,416,225,31,2018,2023
|
||||||
|
text_data.csv,Manifesto Project,8,36008,4501,713,65,1920,2025
|
||||||
|
text_data.csv,PolDem,9,2194,406,93,15,1972,2017
|
||||||
|
@@ -0,0 +1,66 @@
|
|||||||
|
"release","iso2","country","first_year","last_year","parties","election_year_rows","both_text_expert","text_only","expert_only","temporal_propagation"
|
||||||
|
"v0","AL","Albania",1991,2021,9,51,19,23,9,0
|
||||||
|
"v0","AM","Armenia",1995,2021,8,26,25,1,0,0
|
||||||
|
"v0","AR","Argentina",1983,2019,9,51,27,6,9,9
|
||||||
|
"v0","AT","Austria",1949,2019,7,84,64,20,0,0
|
||||||
|
"v0","AU","Australia",1946,2022,9,129,70,53,6,0
|
||||||
|
"v0","AZ","Azerbaijan",1995,2015,3,9,5,1,3,0
|
||||||
|
"v0","BA","Bosnia and Herzegovina",1990,2022,9,59,43,16,0,0
|
||||||
|
"v0","BE","Belgium",1946,2019,20,194,114,77,3,0
|
||||||
|
"v0","BG","Bulgaria",1990,2017,12,43,37,5,0,1
|
||||||
|
"v0","BO","Bolivia",1979,2014,2,13,2,1,10,0
|
||||||
|
"v0","BR","Brazil",1982,2022,8,69,3,57,6,3
|
||||||
|
"v0","BY","Belarus",1995,2008,1,4,1,0,3,0
|
||||||
|
"v0","CA","Canada",1945,2021,8,104,64,39,1,0
|
||||||
|
"v0","CH","Switzerland",1947,2019,16,155,69,86,0,0
|
||||||
|
"v0","CL","Chile",1989,2021,9,65,0,63,0,2
|
||||||
|
"v0","CO","Colombia",1970,2018,9,45,12,9,24,0
|
||||||
|
"v0","CR","Costa Rica",1970,2018,7,35,29,0,6,0
|
||||||
|
"v0","CY","Cyprus",1970,2016,7,44,22,4,18,0
|
||||||
|
"v0","CZ","Czechia",1990,2017,12,51,50,1,0,0
|
||||||
|
"v0","DE","Germany",1949,2025,11,115,56,59,0,0
|
||||||
|
"v0","DK","Denmark",1945,2019,17,236,131,105,0,0
|
||||||
|
"v0","DO","Dominican Republic",1978,2016,3,38,29,0,9,0
|
||||||
|
"v0","EC","Ecuador",1979,2017,9,66,36,4,20,6
|
||||||
|
"v0","EE","Estonia",1992,2019,12,50,41,6,3,0
|
||||||
|
"v0","ES","Spain",1977,2023,23,154,108,45,0,1
|
||||||
|
"v0","FI","Finland",1945,2019,11,158,88,70,0,0
|
||||||
|
"v0","FR","France",1946,2022,16,121,62,55,4,0
|
||||||
|
"v0","GB","United Kingdom",1945,2024,12,106,71,35,0,0
|
||||||
|
"v0","GE","Georgia",1992,2020,10,27,24,3,0,0
|
||||||
|
"v0","GR","Greece",1974,2023,13,76,69,4,2,1
|
||||||
|
"v0","HR","Croatia",1990,2020,13,60,41,15,3,1
|
||||||
|
"v0","HU","Hungary",1990,2022,12,50,42,7,1,0
|
||||||
|
"v0","IE","Ireland",1948,2020,12,100,63,37,0,0
|
||||||
|
"v0","IL","Israel",1949,2022,33,218,91,102,12,13
|
||||||
|
"v0","IS","Iceland",1946,2021,14,116,84,32,0,0
|
||||||
|
"v0","IT","Italy",1946,2018,27,132,55,69,0,8
|
||||||
|
"v0","JP","Japan",1960,2021,12,120,80,40,0,0
|
||||||
|
"v0","KR","South Korea",1992,2020,10,25,20,5,0,0
|
||||||
|
"v0","LK","Sri Lanka",1947,2015,4,32,8,17,7,0
|
||||||
|
"v0","LT","Lithuania",1992,2020,13,50,43,2,5,0
|
||||||
|
"v0","LU","Luxembourg",1945,2013,7,74,44,30,0,0
|
||||||
|
"v0","LV","Latvia",1993,2022,16,55,41,7,1,6
|
||||||
|
"v0","MD","Moldova",1994,2019,6,22,22,0,0,0
|
||||||
|
"v0","ME","Montenegro",1990,2023,13,50,32,18,0,0
|
||||||
|
"v0","MK","North Macedonia",1990,2016,11,57,33,23,1,0
|
||||||
|
"v0","MT","Malta",1971,2022,2,24,14,0,10,0
|
||||||
|
"v0","MX","Mexico",1946,2018,10,86,31,31,1,23
|
||||||
|
"v0","NL","Netherlands",1946,2021,23,188,123,65,0,0
|
||||||
|
"v0","NO","Norway",1945,2017,10,129,71,58,0,0
|
||||||
|
"v0","NZ","New Zealand",1946,2020,10,108,64,42,1,1
|
||||||
|
"v0","PA","Panama",1980,2019,6,26,4,8,14,0
|
||||||
|
"v0","PE","Peru",1978,2016,5,18,7,0,11,0
|
||||||
|
"v0","PL","Poland",1972,2019,12,53,37,10,6,0
|
||||||
|
"v0","PT","Portugal",1975,2022,15,121,73,47,1,0
|
||||||
|
"v0","RO","Romania",1990,2016,11,35,19,5,1,10
|
||||||
|
"v0","RS","Serbia",1990,2023,18,80,41,39,0,0
|
||||||
|
"v0","RU","Russia",1993,2011,10,31,28,3,0,0
|
||||||
|
"v0","SE","Sweden",1944,2022,8,144,103,41,0,0
|
||||||
|
"v0","SI","Slovenia",1990,2018,15,70,60,5,5,0
|
||||||
|
"v0","SK","Slovakia",1990,2016,14,54,47,7,0,0
|
||||||
|
"v0","TR","Türkiye",1950,2018,12,65,44,17,4,0
|
||||||
|
"v0","UA","Ukraine",1994,2019,11,35,31,3,1,0
|
||||||
|
"v0","US","United States",1944,2024,2,78,52,26,0,0
|
||||||
|
"v0","UY","Uruguay",1984,2014,3,16,2,0,14,0
|
||||||
|
"v0","ZA","South Africa",1989,2019,6,24,18,4,2,0
|
||||||
|
@@ -0,0 +1,28 @@
|
|||||||
|
file_name,variable_name,label,description,type,allowed_values,range_min,range_max,missing_value_code,unit,scale_direction,constructed_from,construction_rule,notes
|
||||||
|
party_2d_election_year_panel_vN.csv (inside party_2d_election_year_panel_vN.zip),party_id,PartyFacts party identifier,Identifier for the individual party,integer,,,,,identifier,,PartyFacts crosswalk,Assigned during party harmonization,
|
||||||
|
party_2d_election_year_panel_vN.csv (inside party_2d_election_year_panel_vN.zip),party_name_english,English party name,Party name from the harmonized output,string,,,,,name,,Party metadata,,
|
||||||
|
party_2d_election_year_panel_vN.csv (inside party_2d_election_year_panel_vN.zip),party_name_short,Short party name,Short party label from the harmonized output,string,,,,,name,,Party metadata,,May be missing
|
||||||
|
party_2d_election_year_panel_vN.csv (inside party_2d_election_year_panel_vN.zip),country,Country code,ISO2 or historical country-code identifier,string,,,,,identifier,,Source metadata and PartyFacts,,
|
||||||
|
party_2d_election_year_panel_vN.csv (inside party_2d_election_year_panel_vN.zip),year,Calendar year,Calendar year of the election-year estimate,integer,,1944,2025,,year,,Election and model-year metadata,,
|
||||||
|
party_2d_election_year_panel_vN.csv (inside party_2d_election_year_panel_vN.zip),segment_num,Party segment number,Segment number within party after splitting at long evidence gaps,integer,,1,,,segment,,Party history segmentation,Main segment is coded 1,
|
||||||
|
party_2d_election_year_panel_vN.csv (inside party_2d_election_year_panel_vN.zip),union_party_id,Union or alliance PartyFacts identifier,Identifier of parent union or alliance where applicable,integer,,,,,identifier,,Alliance mapping,,Missing if party is not represented through a union or alliance
|
||||||
|
party_2d_election_year_panel_vN.csv (inside party_2d_election_year_panel_vN.zip),in_union,Union membership indicator,Indicator that the row is associated with a union or alliance,boolean,0;1,0,1,,indicator,,Alliance mapping,,
|
||||||
|
party_2d_election_year_panel_vN.csv (inside party_2d_election_year_panel_vN.zip),pervote,Vote share,Vote share at the election year,numeric,,0,100,,percent,,Election metadata,,
|
||||||
|
party_2d_election_year_panel_vN.csv (inside party_2d_election_year_panel_vN.zip),economic_lr,Economic left-right posterior mean,Posterior mean of economic left-right position,numeric,,0,1,,unit interval,0=left; 1=right,Posterior draws,Mean after inverse-logit transformation,
|
||||||
|
party_2d_election_year_panel_vN.csv (inside party_2d_election_year_panel_vN.zip),galtan,Cultural cosmopolitan--traditionalist posterior mean,Posterior mean of cultural cosmopolitan--traditionalist position,numeric,,0,1,,unit interval,0=cosmopolitan; 1=traditionalist,Posterior draws,Mean after inverse-logit transformation,
|
||||||
|
party_2d_election_year_panel_vN.csv (inside party_2d_election_year_panel_vN.zip),economic_lr_se,Economic posterior standard error,Posterior standard deviation for economic_lr,numeric,,0,,,unit interval,,Posterior draws,Standard deviation over posterior draws,
|
||||||
|
party_2d_election_year_panel_vN.csv (inside party_2d_election_year_panel_vN.zip),galtan_se,Cultural posterior standard error,Posterior standard deviation for the cultural estimate,numeric,,0,,,unit interval,,Posterior draws,Standard deviation over posterior draws,
|
||||||
|
party_2d_election_year_panel_vN.csv (inside party_2d_election_year_panel_vN.zip),economic_lr_q025,Economic lower posterior interval,2.5 percent posterior quantile for economic_lr,numeric,,0,1,,unit interval,0=left; 1=right,Posterior draws,Quantile over posterior draws,
|
||||||
|
party_2d_election_year_panel_vN.csv (inside party_2d_election_year_panel_vN.zip),economic_lr_q975,Economic upper posterior interval,97.5 percent posterior quantile for economic_lr,numeric,,0,1,,unit interval,0=left; 1=right,Posterior draws,Quantile over posterior draws,
|
||||||
|
party_2d_election_year_panel_vN.csv (inside party_2d_election_year_panel_vN.zip),galtan_q025,Cultural lower posterior interval,2.5 percent posterior quantile for the cultural estimate,numeric,,0,1,,unit interval,0=cosmopolitan; 1=traditionalist,Posterior draws,Quantile over posterior draws,
|
||||||
|
party_2d_election_year_panel_vN.csv (inside party_2d_election_year_panel_vN.zip),galtan_q975,Cultural upper posterior interval,97.5 percent posterior quantile for the cultural estimate,numeric,,0,1,,unit interval,0=cosmopolitan; 1=traditionalist,Posterior draws,Quantile over posterior draws,
|
||||||
|
party_2d_election_year_panel_vN.csv (inside party_2d_election_year_panel_vN.zip),election_id,Election identifier,Election identifier where available,string,,,,,identifier,,Election metadata,,Missing where no election identifier is available
|
||||||
|
party_2d_election_year_panel_vN.csv (inside party_2d_election_year_panel_vN.zip),election_date,Election date,Election date where available,date,,,,,date,,Election metadata,,Missing in current processed election metadata
|
||||||
|
party_2d_election_year_panel_vN.csv (inside party_2d_election_year_panel_vN.zip),has_text,Text support indicator,Indicator for direct or nearby text evidence,boolean,0;1,0,1,,indicator,,Source-support construction,Nearby threshold documented in source_support_dictionary.csv,
|
||||||
|
party_2d_election_year_panel_vN.csv (inside party_2d_election_year_panel_vN.zip),has_expert,Expert support indicator,Indicator for direct or nearby expert evidence,boolean,0;1,0,1,,indicator,,Source-support construction,Nearby threshold documented in source_support_dictionary.csv,
|
||||||
|
party_2d_election_year_panel_vN.csv (inside party_2d_election_year_panel_vN.zip),n_text_sources,Number of text sources,Number of distinct text source families contributing direct or nearby evidence,integer,,0,,,count,,Source-support construction,,
|
||||||
|
party_2d_election_year_panel_vN.csv (inside party_2d_election_year_panel_vN.zip),n_expert_sources,Number of expert sources,Number of distinct expert source families contributing direct or nearby evidence,integer,,0,,,count,,Source-support construction,,
|
||||||
|
party_2d_election_year_panel_vN.csv (inside party_2d_election_year_panel_vN.zip),nearest_text_distance,Nearest text distance,Absolute distance in years to nearest text observation used to inform trajectory,numeric,,0,,,years,,Source-support construction,,Missing if no text observation exists for the party-country key
|
||||||
|
party_2d_election_year_panel_vN.csv (inside party_2d_election_year_panel_vN.zip),nearest_expert_distance,Nearest expert distance,Absolute distance in years to nearest expert observation used to inform trajectory,numeric,,0,,,years,,Source-support construction,,Missing if no expert observation exists for the party-country key
|
||||||
|
party_2d_election_year_panel_vN.csv (inside party_2d_election_year_panel_vN.zip),source_support_class,Source-support class,Summary class of text/expert support,string,both_direct_or_nearby; text_only_direct_or_nearby; expert_only_direct_or_nearby; temporal_propagation,,,,categorical,,Source-support construction,,
|
||||||
|
party_2d_annual_model_output_vN.csv (inside party_2d_annual_model_output_vN.zip),all fields,Annual model output fields,Same model-output fields as the production annual party-position file plus no source-support augmentation,mixed,,,,,,,Posterior processing,,Secondary model output
|
||||||
|
@@ -0,0 +1,9 @@
|
|||||||
|
field,value,definition,notes
|
||||||
|
direct_support,observation in same calendar year,Observation in the same calendar year as the election-year estimate,
|
||||||
|
nearby_support,observation within threshold,Observation within the stated distance threshold of the election-year estimate,
|
||||||
|
nearby_support_threshold_years,2,Numerical threshold in years used to classify nearby support,
|
||||||
|
temporal_propagation,no nearby observation,No text or expert observation within the nearby threshold,
|
||||||
|
source_support_class,both_direct_or_nearby,Both text and expert evidence are direct or nearby,
|
||||||
|
source_support_class,text_only_direct_or_nearby,Text evidence is direct or nearby and expert evidence is not,
|
||||||
|
source_support_class,expert_only_direct_or_nearby,Expert evidence is direct or nearby and text evidence is not,
|
||||||
|
source_support_class,temporal_propagation,Neither text nor expert evidence is direct or nearby,
|
||||||
|
@@ -0,0 +1,595 @@
|
|||||||
|
// =============================================================================
|
||||||
|
// 2D BIPOLAR LATENT TRAIT MODEL V6 - Hierarchical L-R Weights
|
||||||
|
// =============================================================================
|
||||||
|
//
|
||||||
|
// V6 CHANGE: Hierarchical L-R weights varying by country, source, and decade
|
||||||
|
// - Replaces global simplex[2] lr_weights with additive logit-scale model
|
||||||
|
// - logit_weight[n] = global + country_offset[c] + source_offset[k] + decade_offset[d]
|
||||||
|
// - All offsets use non-centered parameterization with estimated sigma hyperparameters
|
||||||
|
// - Data-poor contexts shrink toward global mean (~55 new scalar parameters)
|
||||||
|
//
|
||||||
|
// PRESERVED FROM V5:
|
||||||
|
// - Beta-Binomial expert likelihood with K-scaling
|
||||||
|
// - V-Party cultural expansion (5 native items + v2pawelf)
|
||||||
|
// - Mean-constituent model (individual party estimates for union members)
|
||||||
|
// - Segment-based indexing (S segments, R segment-years)
|
||||||
|
// - 3-level hierarchical variance (global -> country -> family)
|
||||||
|
// - Random walk dynamics within segments
|
||||||
|
// - CDU anchor for scale identification (CDU=1375, not CDU/CSU=211)
|
||||||
|
// - Non-centered parameterization for efficiency
|
||||||
|
// - Zero-inflation model for manifesto data
|
||||||
|
// - Country-item intercepts (coding convention differences across countries)
|
||||||
|
// - Binomial-logit likelihood for text data (unchanged)
|
||||||
|
//
|
||||||
|
// =============================================================================
|
||||||
|
|
||||||
|
data {
|
||||||
|
// Segment structure
|
||||||
|
int<lower=1> S; // Number of segments
|
||||||
|
int<lower=1> P; // Number of countries
|
||||||
|
int<lower=1> R; // Total unique segment-year combinations
|
||||||
|
int<lower=1> T_year; // Total number of years
|
||||||
|
array[S] int<lower=1> len_theta_ts; // Number of years per segment
|
||||||
|
|
||||||
|
// Country membership for each segment
|
||||||
|
array[S] int<lower=1, upper=P> segment_country;
|
||||||
|
|
||||||
|
// Segment family data
|
||||||
|
int<lower=1> F; // Number of party families
|
||||||
|
array[S] int<lower=1, upper=F> segment_family; // Family index for each segment
|
||||||
|
|
||||||
|
// =========================================================================
|
||||||
|
// Text data (manifesto + PolDem)
|
||||||
|
// =========================================================================
|
||||||
|
int<lower=1> N_man; // Number of text observations
|
||||||
|
int<lower=1> K_man; // Number of unique text items
|
||||||
|
array[N_man] int<lower=1, upper=K_man> kk_man; // Text item index
|
||||||
|
array[N_man] int<lower=1, upper=S> ss_man; // Segment index (first constituent for unions)
|
||||||
|
array[N_man] int<lower=1, upper=P> pp_man; // Country index
|
||||||
|
array[N_man] int<lower=0> positive; // Positive mentions
|
||||||
|
array[N_man] int<lower=0> sample; // Total sample size
|
||||||
|
array[N_man] int<lower=1, upper=T_year> year_for_man; // Year index
|
||||||
|
|
||||||
|
// Dimension and direction for text data
|
||||||
|
array[N_man] int<lower=1, upper=2> dim_idx_man; // 1=economic, 2=galtan
|
||||||
|
array[N_man] int<lower=-1, upper=1> direction_man; // +1=right/TAN, -1=left/GAL
|
||||||
|
|
||||||
|
// V4: Constituent structure for manifesto observations
|
||||||
|
int<lower=1> N_const_man_total; // Total entries in const_rr_man
|
||||||
|
array[N_man] int<lower=1> n_const_man; // Number of constituents per obs
|
||||||
|
array[N_man] int<lower=1> const_offset_man; // Offset into const_rr_man
|
||||||
|
array[N_const_man_total] int<lower=1, upper=R> const_rr_man; // Constituent rr indices
|
||||||
|
|
||||||
|
// Country-item-year data (used by zero-inflation model)
|
||||||
|
int<lower=1> N_ciy; // Number of unique country-item-year combinations
|
||||||
|
array[N_man] int<lower=1, upper=N_ciy> ciy_idx;
|
||||||
|
|
||||||
|
// =========================================================================
|
||||||
|
// Expert dimension-specific data (V5: integer observations + scale size)
|
||||||
|
// =========================================================================
|
||||||
|
int<lower=1> N_exp_dim; // Number of dimension-specific expert observations
|
||||||
|
int<lower=1> K_exp_dim; // Number of unique expert items
|
||||||
|
array[N_exp_dim] int<lower=1, upper=K_exp_dim> kk_exp_dim;
|
||||||
|
array[N_exp_dim] int<lower=1, upper=S> ss_exp_dim;
|
||||||
|
array[N_exp_dim] int<lower=1, upper=P> pp_exp_dim;
|
||||||
|
array[N_exp_dim] int<lower=0> val_dim_int; // V5: rounded sum = round(mean * K * n_scale)
|
||||||
|
array[N_exp_dim] int<lower=1> n_total_exp_dim; // V5: K * n_scale (total trials)
|
||||||
|
array[N_exp_dim] int<lower=1> n_experts_exp_dim; // V5: K (number of experts)
|
||||||
|
|
||||||
|
// Dimension index for expert data
|
||||||
|
array[N_exp_dim] int<lower=1, upper=2> dim_idx_exp; // 1=economic, 2=galtan
|
||||||
|
|
||||||
|
// V4: Constituent structure for expert dim observations
|
||||||
|
int<lower=1> N_const_exp_dim_total;
|
||||||
|
array[N_exp_dim] int<lower=1> n_const_exp_dim;
|
||||||
|
array[N_exp_dim] int<lower=1> const_offset_exp_dim;
|
||||||
|
array[N_const_exp_dim_total] int<lower=1, upper=R> const_rr_exp_dim;
|
||||||
|
|
||||||
|
// =========================================================================
|
||||||
|
// Expert general L-R data (V5: integer observations + scale size)
|
||||||
|
// =========================================================================
|
||||||
|
int<lower=1> N_exp_lr;
|
||||||
|
int<lower=1> K_exp_lr;
|
||||||
|
array[N_exp_lr] int<lower=1, upper=K_exp_lr> kk_exp_lr;
|
||||||
|
array[N_exp_lr] int<lower=1, upper=S> ss_exp_lr;
|
||||||
|
array[N_exp_lr] int<lower=1, upper=P> pp_exp_lr;
|
||||||
|
array[N_exp_lr] int<lower=0> val_lr_int; // V5: rounded sum = round(mean * K * n_scale)
|
||||||
|
array[N_exp_lr] int<lower=1> n_total_exp_lr; // V5: K * n_scale (total trials)
|
||||||
|
array[N_exp_lr] int<lower=1> n_experts_exp_lr; // V5: K (number of experts)
|
||||||
|
|
||||||
|
// V6: Decade indexing for hierarchical L-R weights
|
||||||
|
int<lower=1> D_lr; // Number of decades
|
||||||
|
array[N_exp_lr] int<lower=1, upper=D_lr> dd_exp_lr; // Decade index per LR obs
|
||||||
|
|
||||||
|
// V4: Constituent structure for expert L-R observations
|
||||||
|
int<lower=1> N_const_exp_lr_total;
|
||||||
|
array[N_exp_lr] int<lower=1> n_const_exp_lr;
|
||||||
|
array[N_exp_lr] int<lower=1> const_offset_exp_lr;
|
||||||
|
array[N_const_exp_lr_total] int<lower=1, upper=R> const_rr_exp_lr;
|
||||||
|
|
||||||
|
// Prior information
|
||||||
|
real mn_resp_log_man;
|
||||||
|
real mn_resp_log_exp_dim;
|
||||||
|
real mn_resp_log_exp_lr;
|
||||||
|
|
||||||
|
// Identification anchor
|
||||||
|
int<lower=1, upper=S> anchor_segment; // CDU segment (1375 with unions, 211 without)
|
||||||
|
|
||||||
|
// V3 compatibility: these are still passed but not used in V4+ likelihood
|
||||||
|
// (kept so older data dicts work without modification for backwards compat)
|
||||||
|
array[N_man] int<lower=1, upper=R> rr_man; // Segment-year index (unused in V4+ likelihood)
|
||||||
|
array[N_exp_dim] int<lower=1, upper=R> rr_exp_dim;
|
||||||
|
array[N_exp_lr] int<lower=1, upper=R> rr_exp_lr;
|
||||||
|
}
|
||||||
|
|
||||||
|
transformed data {
|
||||||
|
real eps = 1e-6;
|
||||||
|
real one_minus_eps = 1 - eps;
|
||||||
|
|
||||||
|
// Count zero and non-zero samples
|
||||||
|
int N_man_zero = 0;
|
||||||
|
int N_man_nonzero = 0;
|
||||||
|
for (n in 1:N_man) {
|
||||||
|
if (sample[n] == 0) {
|
||||||
|
N_man_zero += 1;
|
||||||
|
} else {
|
||||||
|
N_man_nonzero += 1;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// Create index arrays for zero/nonzero split
|
||||||
|
array[N_man_zero > 0 ? N_man_zero : 1] int idx_zero;
|
||||||
|
array[N_man_nonzero > 0 ? N_man_nonzero : 1] int idx_nonzero;
|
||||||
|
|
||||||
|
{
|
||||||
|
int pos_zero = 1;
|
||||||
|
int pos_nonzero = 1;
|
||||||
|
for (n in 1:N_man) {
|
||||||
|
if (sample[n] == 0) {
|
||||||
|
idx_zero[pos_zero] = n;
|
||||||
|
pos_zero += 1;
|
||||||
|
} else {
|
||||||
|
idx_nonzero[pos_nonzero] = n;
|
||||||
|
pos_nonzero += 1;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// V4: Pre-compute constituent info for nonzero manifesto obs
|
||||||
|
array[N_man_nonzero > 0 ? N_man_nonzero : 1] int kk_man_nonzero;
|
||||||
|
array[N_man_nonzero > 0 ? N_man_nonzero : 1] int orig_idx_nonzero;
|
||||||
|
array[N_man_nonzero > 0 ? N_man_nonzero : 1] int direction_nonzero;
|
||||||
|
array[N_man_nonzero > 0 ? N_man_nonzero : 1] int pp_man_nonzero;
|
||||||
|
array[N_man_nonzero > 0 ? N_man_nonzero : 1] int dim_idx_nonzero;
|
||||||
|
array[N_man_nonzero > 0 ? N_man_nonzero : 1] int n_const_nonzero;
|
||||||
|
array[N_man_nonzero > 0 ? N_man_nonzero : 1] int const_offset_nonzero;
|
||||||
|
|
||||||
|
{
|
||||||
|
for (i in 1:N_man_nonzero) {
|
||||||
|
int n = idx_nonzero[i];
|
||||||
|
orig_idx_nonzero[i] = n;
|
||||||
|
kk_man_nonzero[i] = kk_man[n];
|
||||||
|
direction_nonzero[i] = direction_man[n];
|
||||||
|
pp_man_nonzero[i] = pp_man[n];
|
||||||
|
dim_idx_nonzero[i] = dim_idx_man[n];
|
||||||
|
n_const_nonzero[i] = n_const_man[n];
|
||||||
|
const_offset_nonzero[i] = const_offset_man[n];
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// Pre-compute segment start positions
|
||||||
|
array[S + 1] int segment_start;
|
||||||
|
segment_start[1] = 1;
|
||||||
|
for (s in 1:S) {
|
||||||
|
segment_start[s + 1] = segment_start[s] + len_theta_ts[s];
|
||||||
|
}
|
||||||
|
|
||||||
|
// Extract sample and positive for nonzero observations
|
||||||
|
array[N_man_nonzero > 0 ? N_man_nonzero : 1] int sample_nonzero;
|
||||||
|
array[N_man_nonzero > 0 ? N_man_nonzero : 1] int positive_nonzero;
|
||||||
|
for (i in 1:N_man_nonzero) {
|
||||||
|
sample_nonzero[i] = sample[idx_nonzero[i]];
|
||||||
|
positive_nonzero[i] = positive[idx_nonzero[i]];
|
||||||
|
}
|
||||||
|
|
||||||
|
// Pre-compute family-to-country mapping
|
||||||
|
array[F] int family_country;
|
||||||
|
{
|
||||||
|
for (f in 1:F) {
|
||||||
|
family_country[f] = 0;
|
||||||
|
}
|
||||||
|
for (s in 1:S) {
|
||||||
|
int f = segment_family[s];
|
||||||
|
if (family_country[f] == 0) {
|
||||||
|
family_country[f] = segment_country[s];
|
||||||
|
}
|
||||||
|
}
|
||||||
|
for (f in 1:F) {
|
||||||
|
if (family_country[f] == 0) {
|
||||||
|
family_country[f] = 1;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
parameters {
|
||||||
|
// =========================================================================
|
||||||
|
// Latent position parameters - 2 dimensions
|
||||||
|
// =========================================================================
|
||||||
|
matrix[2, R] theta_ncp; // Non-centered: [1]=economic_lr, [2]=galtan
|
||||||
|
matrix[2, S] theta_init_raw; // Initial traits per segment
|
||||||
|
vector<lower=0>[2] sigma_theta_init; // SD of initial theta per dimension
|
||||||
|
|
||||||
|
// =========================================================================
|
||||||
|
// Three-level hierarchical variance
|
||||||
|
// =========================================================================
|
||||||
|
vector[2] mu_sigma_global_raw; // Global mean RW variance
|
||||||
|
vector<lower=0>[2] tau_sigma_country; // Country deviation scale
|
||||||
|
matrix[2, P] sigma_country_raw; // Non-centered country deviations
|
||||||
|
vector<lower=0>[2] tau_sigma_family; // Family deviation scale
|
||||||
|
matrix[2, F] sigma_family_raw; // Non-centered family deviations
|
||||||
|
|
||||||
|
// =========================================================================
|
||||||
|
// Country-item intercepts (coding convention differences)
|
||||||
|
// =========================================================================
|
||||||
|
matrix[P, K_man] country_item_raw;
|
||||||
|
real<lower=0> sigma_country_item;
|
||||||
|
|
||||||
|
// =========================================================================
|
||||||
|
// Zero-sample parameters
|
||||||
|
// =========================================================================
|
||||||
|
real alpha_zs;
|
||||||
|
vector[T_year] year_effect_raw;
|
||||||
|
real<lower=0> sigma_year_effect;
|
||||||
|
vector[S] segment_zs_raw;
|
||||||
|
real<lower=0> sigma_segment_zs;
|
||||||
|
vector[N_ciy] ciy_zs_raw;
|
||||||
|
real<lower=0> sigma_ciy_zs;
|
||||||
|
|
||||||
|
// =========================================================================
|
||||||
|
// Item parameters
|
||||||
|
// =========================================================================
|
||||||
|
|
||||||
|
// Text data: intercept + single positive loading (direction handled in data)
|
||||||
|
vector[K_man] gamma_man_intercept_raw;
|
||||||
|
vector<lower=0>[K_man] gamma_man_loading; // Positive loading
|
||||||
|
|
||||||
|
// Expert dimension-specific: intercept + slope (direct mapping to dimension)
|
||||||
|
vector[K_exp_dim] gamma_exp_intercept_raw;
|
||||||
|
vector<lower=0>[K_exp_dim] gamma_exp_slope;
|
||||||
|
|
||||||
|
// Expert general L-R: intercept + slope
|
||||||
|
vector[K_exp_lr] gamma_lr_intercept_raw;
|
||||||
|
vector<lower=0>[K_exp_lr] gamma_lr_slope;
|
||||||
|
|
||||||
|
// V6: Hierarchical L-R weights (replace simplex[2] lr_weights)
|
||||||
|
real lr_weight_global; // Global logit-scale weight
|
||||||
|
vector[P] lr_country_offset_raw; // Country offsets (non-centered)
|
||||||
|
vector[K_exp_lr] lr_source_offset_raw; // Source offsets (non-centered)
|
||||||
|
vector[D_lr] lr_decade_offset_raw; // Decade offsets (non-centered)
|
||||||
|
real<lower=0> sigma_lr_country; // SD of country offsets
|
||||||
|
real<lower=0> sigma_lr_source; // SD of source offsets
|
||||||
|
real<lower=0> sigma_lr_decade; // SD of decade offsets
|
||||||
|
|
||||||
|
// Precision parameters
|
||||||
|
real<lower=0> phi_exp_dim; // Precision for dimension-specific (now: only measurement noise)
|
||||||
|
real<lower=0> phi_exp_lr; // Precision for general L-R (now: only measurement noise)
|
||||||
|
|
||||||
|
// Scale parameters
|
||||||
|
real<lower=0> sigma_intercept_man;
|
||||||
|
real<lower=0> sigma_loading_man;
|
||||||
|
real<lower=0> sigma_intercept_exp;
|
||||||
|
real<lower=0> sigma_slope_exp;
|
||||||
|
real<lower=0> sigma_intercept_lr;
|
||||||
|
real<lower=0> sigma_slope_lr;
|
||||||
|
real mu_lambda_man;
|
||||||
|
real mu_lambda_exp_dim;
|
||||||
|
real mu_lambda_exp_lr;
|
||||||
|
}
|
||||||
|
|
||||||
|
transformed parameters {
|
||||||
|
matrix[2, R] theta; // [1]=economic_lr, [2]=galtan
|
||||||
|
matrix[2, S] theta_init;
|
||||||
|
|
||||||
|
// Three-level hierarchical variance (2D)
|
||||||
|
matrix[2, P] sigma_theta_country;
|
||||||
|
for (d in 1:2) {
|
||||||
|
for (p in 1:P) {
|
||||||
|
sigma_theta_country[d, p] = log1p_exp(
|
||||||
|
mu_sigma_global_raw[d] + tau_sigma_country[d] * sigma_country_raw[d, p]
|
||||||
|
);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
matrix[2, F] sigma_theta_family;
|
||||||
|
for (d in 1:2) {
|
||||||
|
for (f in 1:F) {
|
||||||
|
int c = family_country[f];
|
||||||
|
sigma_theta_family[d, f] = log1p_exp(
|
||||||
|
log(sigma_theta_country[d, c]) + tau_sigma_family[d] * sigma_family_raw[d, f]
|
||||||
|
);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// Non-centered parameterization for theta_init
|
||||||
|
for (dim in 1:2) {
|
||||||
|
theta_init[dim, :] = sigma_theta_init[dim] * theta_init_raw[dim, :];
|
||||||
|
}
|
||||||
|
|
||||||
|
// Country-item intercepts
|
||||||
|
matrix[P, K_man] country_item_intercept = sigma_country_item * country_item_raw;
|
||||||
|
|
||||||
|
// Zero-sample components
|
||||||
|
vector[T_year] year_effect = sigma_year_effect * year_effect_raw;
|
||||||
|
vector[S] segment_zs = sigma_segment_zs * segment_zs_raw;
|
||||||
|
vector[N_ciy] ciy_zs = sigma_ciy_zs * ciy_zs_raw;
|
||||||
|
|
||||||
|
// Zero-sample probability
|
||||||
|
vector[N_man] zero_sample_logit = alpha_zs +
|
||||||
|
year_effect[year_for_man] +
|
||||||
|
segment_zs[ss_man] +
|
||||||
|
ciy_zs[ciy_idx];
|
||||||
|
vector[N_man] zero_sample_prob = inv_logit(zero_sample_logit);
|
||||||
|
zero_sample_prob = fmax(fmin(zero_sample_prob, one_minus_eps), eps);
|
||||||
|
|
||||||
|
// Construct theta using family-specific random walk variance (2D)
|
||||||
|
for (dim in 1:2) {
|
||||||
|
for (s in 1:S) {
|
||||||
|
int start = segment_start[s];
|
||||||
|
int Ts = len_theta_ts[s];
|
||||||
|
int fam = segment_family[s];
|
||||||
|
real sigma_s = sigma_theta_family[dim, fam];
|
||||||
|
|
||||||
|
theta[dim, start] = theta_init[dim, s] + sigma_s * theta_ncp[dim, start];
|
||||||
|
if (Ts > 1) {
|
||||||
|
theta[dim, start + 1 : start + Ts - 1] = theta[dim, start] +
|
||||||
|
cumulative_sum(sigma_s * theta_ncp[dim, start + 1 : start + Ts - 1]);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// Item parameters
|
||||||
|
vector[K_man] gamma_man_intercept = mu_lambda_man + sigma_intercept_man * gamma_man_intercept_raw;
|
||||||
|
vector[K_exp_dim] gamma_exp_intercept = mu_lambda_exp_dim + sigma_intercept_exp * gamma_exp_intercept_raw;
|
||||||
|
vector[K_exp_lr] gamma_lr_intercept = mu_lambda_exp_lr + sigma_intercept_lr * gamma_lr_intercept_raw;
|
||||||
|
}
|
||||||
|
|
||||||
|
model {
|
||||||
|
// =========================================================================
|
||||||
|
// PRIORS
|
||||||
|
// =========================================================================
|
||||||
|
|
||||||
|
// Three-level hierarchical variance priors (2D)
|
||||||
|
mu_sigma_global_raw ~ normal(-0.8, 0.5);
|
||||||
|
tau_sigma_country ~ normal(0, 0.3);
|
||||||
|
to_vector(sigma_country_raw) ~ std_normal();
|
||||||
|
tau_sigma_family ~ normal(0, 0.2);
|
||||||
|
to_vector(sigma_family_raw) ~ std_normal();
|
||||||
|
|
||||||
|
// Other variance priors
|
||||||
|
sigma_theta_init ~ normal(0, 0.5);
|
||||||
|
to_vector(theta_init_raw) ~ std_normal();
|
||||||
|
to_vector(theta_ncp) ~ std_normal();
|
||||||
|
|
||||||
|
// Country-item intercept priors
|
||||||
|
to_vector(country_item_raw) ~ std_normal();
|
||||||
|
sigma_country_item ~ normal(0, 0.3);
|
||||||
|
|
||||||
|
// Zero-sample priors
|
||||||
|
alpha_zs ~ normal(-1, 1);
|
||||||
|
year_effect_raw ~ std_normal();
|
||||||
|
sigma_year_effect ~ normal(0, 0.5);
|
||||||
|
segment_zs_raw ~ std_normal();
|
||||||
|
sigma_segment_zs ~ normal(0, 0.3);
|
||||||
|
ciy_zs_raw ~ std_normal();
|
||||||
|
sigma_ciy_zs ~ normal(0, 0.3);
|
||||||
|
|
||||||
|
// =========================================================================
|
||||||
|
// Item parameter priors
|
||||||
|
// =========================================================================
|
||||||
|
|
||||||
|
// Text data item priors
|
||||||
|
mu_lambda_man ~ normal(mn_resp_log_man, 0.5);
|
||||||
|
sigma_intercept_man ~ normal(0, 1);
|
||||||
|
sigma_loading_man ~ normal(0, 0.5);
|
||||||
|
gamma_man_intercept_raw ~ std_normal();
|
||||||
|
gamma_man_loading ~ normal(1.0, sigma_loading_man);
|
||||||
|
|
||||||
|
// Expert dimension item priors
|
||||||
|
mu_lambda_exp_dim ~ normal(mn_resp_log_exp_dim, 0.5);
|
||||||
|
sigma_intercept_exp ~ normal(0, 1);
|
||||||
|
sigma_slope_exp ~ normal(0, 0.5);
|
||||||
|
gamma_exp_intercept_raw ~ std_normal();
|
||||||
|
gamma_exp_slope ~ normal(1.0, sigma_slope_exp);
|
||||||
|
|
||||||
|
// Expert L-R item priors
|
||||||
|
mu_lambda_exp_lr ~ normal(mn_resp_log_exp_lr, 0.5);
|
||||||
|
sigma_intercept_lr ~ normal(0, 1);
|
||||||
|
sigma_slope_lr ~ normal(0, 0.5);
|
||||||
|
gamma_lr_intercept_raw ~ std_normal();
|
||||||
|
gamma_lr_slope ~ normal(1.0, sigma_slope_lr);
|
||||||
|
|
||||||
|
// V6: Hierarchical L-R weight priors
|
||||||
|
lr_weight_global ~ normal(0, 1);
|
||||||
|
lr_country_offset_raw ~ std_normal();
|
||||||
|
lr_source_offset_raw ~ std_normal();
|
||||||
|
lr_decade_offset_raw ~ std_normal();
|
||||||
|
sigma_lr_country ~ normal(0, 0.5);
|
||||||
|
sigma_lr_source ~ normal(0, 0.5);
|
||||||
|
sigma_lr_decade ~ normal(0, 0.5);
|
||||||
|
|
||||||
|
// Expert data precision priors
|
||||||
|
phi_exp_dim ~ gamma(50, 0.5);
|
||||||
|
phi_exp_lr ~ gamma(10, 0.5);
|
||||||
|
|
||||||
|
// =========================================================================
|
||||||
|
// CDU ANCHOR CONSTRAINT
|
||||||
|
// With unions: anchors CDU (1375) at moderate center-right position
|
||||||
|
// Without unions: anchors CDU/CSU (211) as before
|
||||||
|
// =========================================================================
|
||||||
|
target += normal_lpdf(theta_init[1, anchor_segment] | 0.2, 0.2); // economic_lr
|
||||||
|
target += normal_lpdf(theta_init[2, anchor_segment] | 0.2, 0.2); // galtan
|
||||||
|
|
||||||
|
// =========================================================================
|
||||||
|
// LIKELIHOOD 1: Text data (binomial with zero-inflation)
|
||||||
|
// V4: Mean-constituent averaging for union observations
|
||||||
|
// =========================================================================
|
||||||
|
|
||||||
|
// Zero-sample observations
|
||||||
|
if (N_man_zero > 0) {
|
||||||
|
target += sum(log(zero_sample_prob[idx_zero]));
|
||||||
|
}
|
||||||
|
|
||||||
|
// Non-zero observations with constituent averaging
|
||||||
|
if (N_man_nonzero > 0) {
|
||||||
|
vector[N_man_nonzero] lin_man;
|
||||||
|
|
||||||
|
for (i in 1:N_man_nonzero) {
|
||||||
|
int nc = n_const_nonzero[i];
|
||||||
|
int off = const_offset_nonzero[i];
|
||||||
|
int dim = dim_idx_nonzero[i];
|
||||||
|
real avg_pos;
|
||||||
|
|
||||||
|
if (nc == 1) {
|
||||||
|
// Fast path: single party (>95% of observations)
|
||||||
|
avg_pos = theta[dim, const_rr_man[off]];
|
||||||
|
} else {
|
||||||
|
// Union: average over constituent thetas
|
||||||
|
avg_pos = 0;
|
||||||
|
for (c in 0:(nc-1)) {
|
||||||
|
avg_pos += theta[dim, const_rr_man[off + c]];
|
||||||
|
}
|
||||||
|
avg_pos /= nc;
|
||||||
|
}
|
||||||
|
|
||||||
|
lin_man[i] = gamma_man_intercept[kk_man_nonzero[i]] +
|
||||||
|
direction_nonzero[i] * gamma_man_loading[kk_man_nonzero[i]] * avg_pos +
|
||||||
|
country_item_intercept[pp_man_nonzero[i], kk_man_nonzero[i]];
|
||||||
|
}
|
||||||
|
|
||||||
|
for (i in 1:N_man_nonzero) {
|
||||||
|
target += log1m(zero_sample_prob[orig_idx_nonzero[i]]) +
|
||||||
|
binomial_logit_lpmf(positive_nonzero[i] | sample_nonzero[i], lin_man[i]);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// =========================================================================
|
||||||
|
// LIKELIHOOD 2: Expert dimension-specific data (V5: beta-binomial likelihood)
|
||||||
|
// V4: Constituent averaging for union-level expert obs
|
||||||
|
// =========================================================================
|
||||||
|
{
|
||||||
|
vector[N_exp_dim] pos;
|
||||||
|
for (n in 1:N_exp_dim) {
|
||||||
|
int nc = n_const_exp_dim[n];
|
||||||
|
int off = const_offset_exp_dim[n];
|
||||||
|
int dim = dim_idx_exp[n];
|
||||||
|
|
||||||
|
if (nc == 1) {
|
||||||
|
pos[n] = theta[dim, const_rr_exp_dim[off]];
|
||||||
|
} else {
|
||||||
|
pos[n] = 0;
|
||||||
|
for (c in 0:(nc-1)) {
|
||||||
|
pos[n] += theta[dim, const_rr_exp_dim[off + c]];
|
||||||
|
}
|
||||||
|
pos[n] /= nc;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
vector[N_exp_dim] lin_exp_dim;
|
||||||
|
for (n in 1:N_exp_dim) {
|
||||||
|
lin_exp_dim[n] = gamma_exp_intercept[kk_exp_dim[n]] +
|
||||||
|
gamma_exp_slope[kk_exp_dim[n]] * pos[n];
|
||||||
|
}
|
||||||
|
|
||||||
|
vector[N_exp_dim] mu_exp_dim = inv_logit(lin_exp_dim);
|
||||||
|
mu_exp_dim = fmax(fmin(mu_exp_dim, one_minus_eps), eps);
|
||||||
|
|
||||||
|
// K-scaling: phi * K corrects for independent expert perceptions
|
||||||
|
vector[N_exp_dim] alpha_exp_dim = phi_exp_dim * to_vector(n_experts_exp_dim) .* mu_exp_dim;
|
||||||
|
vector[N_exp_dim] beta_exp_dim = phi_exp_dim * to_vector(n_experts_exp_dim) .* (1 - mu_exp_dim);
|
||||||
|
val_dim_int ~ beta_binomial(n_total_exp_dim, alpha_exp_dim, beta_exp_dim);
|
||||||
|
}
|
||||||
|
|
||||||
|
// =========================================================================
|
||||||
|
// LIKELIHOOD 3: Expert general L-R data (V6: hierarchical per-obs weights)
|
||||||
|
// V4: Constituent averaging + weighted combination of both dimensions
|
||||||
|
// V6: Per-observation weights via country + source + decade offsets
|
||||||
|
// =========================================================================
|
||||||
|
{
|
||||||
|
// V6: Compute per-observation economic weight on logit scale
|
||||||
|
vector[N_exp_lr] logit_w;
|
||||||
|
for (n in 1:N_exp_lr) {
|
||||||
|
logit_w[n] = lr_weight_global
|
||||||
|
+ sigma_lr_country * lr_country_offset_raw[pp_exp_lr[n]]
|
||||||
|
+ sigma_lr_source * lr_source_offset_raw[kk_exp_lr[n]]
|
||||||
|
+ sigma_lr_decade * lr_decade_offset_raw[dd_exp_lr[n]];
|
||||||
|
}
|
||||||
|
vector[N_exp_lr] w_econ = inv_logit(logit_w);
|
||||||
|
|
||||||
|
vector[N_exp_lr] combined_pos;
|
||||||
|
for (n in 1:N_exp_lr) {
|
||||||
|
int nc = n_const_exp_lr[n];
|
||||||
|
int off = const_offset_exp_lr[n];
|
||||||
|
|
||||||
|
if (nc == 1) {
|
||||||
|
int r = const_rr_exp_lr[off];
|
||||||
|
combined_pos[n] = w_econ[n] * theta[1, r] + (1 - w_econ[n]) * theta[2, r];
|
||||||
|
} else {
|
||||||
|
// Average the combined position across constituents
|
||||||
|
combined_pos[n] = 0;
|
||||||
|
for (c in 0:(nc-1)) {
|
||||||
|
int r = const_rr_exp_lr[off + c];
|
||||||
|
combined_pos[n] += w_econ[n] * theta[1, r] + (1 - w_econ[n]) * theta[2, r];
|
||||||
|
}
|
||||||
|
combined_pos[n] /= nc;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
vector[N_exp_lr] lin_exp_lr;
|
||||||
|
for (n in 1:N_exp_lr) {
|
||||||
|
lin_exp_lr[n] = gamma_lr_intercept[kk_exp_lr[n]] +
|
||||||
|
gamma_lr_slope[kk_exp_lr[n]] * combined_pos[n];
|
||||||
|
}
|
||||||
|
|
||||||
|
vector[N_exp_lr] mu_exp_lr = inv_logit(lin_exp_lr);
|
||||||
|
mu_exp_lr = fmax(fmin(mu_exp_lr, one_minus_eps), eps);
|
||||||
|
|
||||||
|
// K-scaling: phi * K corrects for independent expert perceptions
|
||||||
|
vector[N_exp_lr] alpha_exp_lr = phi_exp_lr * to_vector(n_experts_exp_lr) .* mu_exp_lr;
|
||||||
|
vector[N_exp_lr] beta_exp_lr = phi_exp_lr * to_vector(n_experts_exp_lr) .* (1 - mu_exp_lr);
|
||||||
|
val_lr_int ~ beta_binomial(n_total_exp_lr, alpha_exp_lr, beta_exp_lr);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
generated quantities {
|
||||||
|
// =========================================================================
|
||||||
|
// Direct outputs (no derivation needed)
|
||||||
|
// =========================================================================
|
||||||
|
|
||||||
|
// Two bipolar scales on [0,1] via inv_logit
|
||||||
|
vector[R] economic_lr = inv_logit(to_vector(theta[1, :])); // 0=left, 1=right
|
||||||
|
vector[R] galtan = inv_logit(to_vector(theta[2, :])); // 0=GAL, 1=TAN
|
||||||
|
|
||||||
|
// General left-right combining both dimensions (using global weight only)
|
||||||
|
// Per-R country/source/decade info not available in GQ, so use global mean
|
||||||
|
real lr_w_econ_global = inv_logit(lr_weight_global);
|
||||||
|
vector[R] general_lr = inv_logit(
|
||||||
|
lr_w_econ_global * to_vector(theta[1, :]) + (1 - lr_w_econ_global) * to_vector(theta[2, :])
|
||||||
|
);
|
||||||
|
|
||||||
|
// V6: Expose hierarchical L-R weight diagnostics
|
||||||
|
real lr_weight_econ_global = lr_w_econ_global;
|
||||||
|
real lr_sigma_country = sigma_lr_country;
|
||||||
|
real lr_sigma_source = sigma_lr_source;
|
||||||
|
real lr_sigma_decade = sigma_lr_decade;
|
||||||
|
|
||||||
|
// Expose hierarchical variance diagnostics (2D)
|
||||||
|
vector[2] mean_sigma_global;
|
||||||
|
vector[2] mean_sigma_country;
|
||||||
|
vector[2] mean_sigma_family;
|
||||||
|
for (d in 1:2) {
|
||||||
|
mean_sigma_global[d] = log1p_exp(mu_sigma_global_raw[d]);
|
||||||
|
mean_sigma_country[d] = mean(sigma_theta_country[d, :]);
|
||||||
|
mean_sigma_family[d] = mean(sigma_theta_family[d, :]);
|
||||||
|
}
|
||||||
|
}
|
||||||
Executable
+63
@@ -0,0 +1,63 @@
|
|||||||
|
#!/usr/bin/env bash
|
||||||
|
set -euo pipefail
|
||||||
|
|
||||||
|
MODE="${1:-reuse}"
|
||||||
|
repo_root="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd -P)"
|
||||||
|
cd "$repo_root"
|
||||||
|
|
||||||
|
case "$MODE" in
|
||||||
|
full|reuse|dry-run) ;;
|
||||||
|
*)
|
||||||
|
echo "Usage: $0 [full|reuse|dry-run]" >&2
|
||||||
|
exit 1
|
||||||
|
;;
|
||||||
|
esac
|
||||||
|
|
||||||
|
export PARTY2D_RAW_DATA_DIR="${PARTY2D_RAW_DATA_DIR:-$repo_root/_local/raw}"
|
||||||
|
export TMPDIR="${TMPDIR:-$repo_root/_local/tmp}"
|
||||||
|
mkdir -p "$TMPDIR"
|
||||||
|
|
||||||
|
required_model_inputs=(
|
||||||
|
"data/text_data.csv"
|
||||||
|
"data/expert.csv"
|
||||||
|
"data/lr_data.csv"
|
||||||
|
"data/union_mapping.csv"
|
||||||
|
"data/party_families.csv"
|
||||||
|
)
|
||||||
|
|
||||||
|
if [ "$MODE" = "dry-run" ]; then
|
||||||
|
echo "Checking commands..."
|
||||||
|
command -v bash >/dev/null
|
||||||
|
command -v julia >/dev/null
|
||||||
|
echo "Checking key files..."
|
||||||
|
test -f Project.toml
|
||||||
|
test -f Manifest.toml
|
||||||
|
test -f models/stan_model_2dim_v6.stan
|
||||||
|
for input in "${required_model_inputs[@]}"; do
|
||||||
|
test -f "$input"
|
||||||
|
done
|
||||||
|
echo "Checking shell syntax..."
|
||||||
|
bash -n scripts/01_prepare_data.sh
|
||||||
|
bash -n scripts/02_fit_model.sh
|
||||||
|
bash -n scripts/03_extract_estimates.sh
|
||||||
|
bash -n scripts/04_enrich_estimates.sh
|
||||||
|
bash -n scripts/05_validate_estimates.sh
|
||||||
|
bash -n data-setup/run_data_setup.sh
|
||||||
|
bash -n data-setup/check_raw_data.sh
|
||||||
|
echo "Checking Julia project can instantiate without running model code..."
|
||||||
|
julia --project=. -e 'import Pkg; Pkg.instantiate(); println("Julia project OK")'
|
||||||
|
echo "Dry run passed. No model fitting was run."
|
||||||
|
exit 0
|
||||||
|
fi
|
||||||
|
|
||||||
|
bash scripts/01_prepare_data.sh
|
||||||
|
|
||||||
|
if [ "$MODE" = "full" ]; then
|
||||||
|
bash scripts/02_fit_model.sh
|
||||||
|
else
|
||||||
|
echo "reuse: skipping Stan model fitting; using latest existing model output"
|
||||||
|
fi
|
||||||
|
|
||||||
|
bash scripts/03_extract_estimates.sh
|
||||||
|
bash scripts/04_enrich_estimates.sh
|
||||||
|
bash scripts/05_validate_estimates.sh
|
||||||
Executable
+28
@@ -0,0 +1,28 @@
|
|||||||
|
#!/usr/bin/env bash
|
||||||
|
set -euo pipefail
|
||||||
|
|
||||||
|
repo_root="$(cd "$(dirname "${BASH_SOURCE[0]}")/.." && pwd -P)"
|
||||||
|
cd "$repo_root"
|
||||||
|
|
||||||
|
required_model_inputs=(
|
||||||
|
"data/text_data.csv"
|
||||||
|
"data/expert.csv"
|
||||||
|
"data/lr_data.csv"
|
||||||
|
"data/union_mapping.csv"
|
||||||
|
"data/party_families.csv"
|
||||||
|
)
|
||||||
|
|
||||||
|
missing=0
|
||||||
|
for input in "${required_model_inputs[@]}"; do
|
||||||
|
if [ ! -s "$input" ]; then
|
||||||
|
echo "Missing model-ready input: $input" >&2
|
||||||
|
missing=1
|
||||||
|
fi
|
||||||
|
done
|
||||||
|
|
||||||
|
if [ "$missing" -ne 0 ]; then
|
||||||
|
echo "Regenerate model inputs with data-setup/run_data_setup.sh, or restore the included data/ files." >&2
|
||||||
|
exit 1
|
||||||
|
fi
|
||||||
|
|
||||||
|
echo "Model-ready data inputs are present."
|
||||||
Executable
+7
@@ -0,0 +1,7 @@
|
|||||||
|
#!/usr/bin/env bash
|
||||||
|
set -euo pipefail
|
||||||
|
|
||||||
|
repo_root="$(cd "$(dirname "${BASH_SOURCE[0]}")/.." && pwd -P)"
|
||||||
|
cd "$repo_root"
|
||||||
|
|
||||||
|
julia --project=. src/julia/01_run_model.jl
|
||||||
Executable
+7
@@ -0,0 +1,7 @@
|
|||||||
|
#!/usr/bin/env bash
|
||||||
|
set -euo pipefail
|
||||||
|
|
||||||
|
repo_root="$(cd "$(dirname "${BASH_SOURCE[0]}")/.." && pwd -P)"
|
||||||
|
cd "$repo_root"
|
||||||
|
|
||||||
|
julia --project=. src/julia/02_post_estimation.jl
|
||||||
Executable
+7
@@ -0,0 +1,7 @@
|
|||||||
|
#!/usr/bin/env bash
|
||||||
|
set -euo pipefail
|
||||||
|
|
||||||
|
repo_root="$(cd "$(dirname "${BASH_SOURCE[0]}")/.." && pwd -P)"
|
||||||
|
cd "$repo_root"
|
||||||
|
|
||||||
|
julia --project=. src/julia/02_enrich_output.jl
|
||||||
Executable
+9
@@ -0,0 +1,9 @@
|
|||||||
|
#!/usr/bin/env bash
|
||||||
|
set -euo pipefail
|
||||||
|
|
||||||
|
repo_root="$(cd "$(dirname "${BASH_SOURCE[0]}")/.." && pwd -P)"
|
||||||
|
cd "$repo_root"
|
||||||
|
|
||||||
|
julia --project=. src/julia/validate_convergent.jl
|
||||||
|
julia --project=. src/julia/validate_uncertainty.jl
|
||||||
|
julia --project=. src/julia/validate_construct.jl
|
||||||
Executable
+42
@@ -0,0 +1,42 @@
|
|||||||
|
#!/usr/bin/env bash
|
||||||
|
set -euo pipefail
|
||||||
|
|
||||||
|
repo_root="$(git rev-parse --show-toplevel)"
|
||||||
|
cd "$repo_root"
|
||||||
|
|
||||||
|
fail() {
|
||||||
|
printf 'PUBLIC CONTENT CHECK FAILED: %s\n' "$1" >&2
|
||||||
|
exit 1
|
||||||
|
}
|
||||||
|
|
||||||
|
# The bracketed spellings intentionally prevent this guard from matching its
|
||||||
|
# own detection patterns.
|
||||||
|
content_pattern='[Rr][Ee][Vv][Ii][Ee][Ww]([Ee][Rr]|[Ss])?|[Ee][Dd][Ii][Tt][Oo][Rr]([[:space:]_-]*[Cc][Oo][Mm][Mm][Ee][Nn][Tt][Ss]?)?|[Mm][Aa][Jj][Oo][Rr][[:space:]_-]*[Rr][Ee][Vv][Ii][Ss][Ii][Oo][Nn]|[Pp][Oo][Ii][Nn][Tt][[:space:]_-]*[Bb][Yy][[:space:]_-]*[Pp][Oo][Ii][Nn][Tt]|[Rr][Ee][Vv][Ii][Ee][Ww][[:space:]_-]*[Pp][Rr][Oo][Cc][Ee][Ss][Ss]|[Ss][Uu][Bb][Mm][Ii][Ss][Ss][Ii][Oo][Nn][[:space:]_-]*[Dd][Ee][Tt][Aa][Ii][Ll][Ss]'
|
||||||
|
path_pattern='([Rr][Ee][Vv][Ii][Ee][Ww]|[Rr][Ee][Vv][Ii][Ss][Ii][Oo][Nn]|[Rr][Ee][Ss][Pp][Oo][Nn][Ss][Ee][[:space:]_.-]*[Tt][Oo])'
|
||||||
|
|
||||||
|
check_ref() {
|
||||||
|
local ref="$1"
|
||||||
|
local hits
|
||||||
|
hits="$(git grep -n -I -E "$content_pattern" "$ref" -- . ':(exclude)*.pdf' 2>/dev/null || true)"
|
||||||
|
[[ -z "$hits" ]] || fail "non-public wording found in $ref:\n$hits"
|
||||||
|
}
|
||||||
|
|
||||||
|
check_names() {
|
||||||
|
local names
|
||||||
|
names="$(git ls-tree -r --name-only HEAD | grep -E "$path_pattern" || true)"
|
||||||
|
[[ -z "$names" ]] || fail "non-public-looking tracked path(s):\n$names"
|
||||||
|
}
|
||||||
|
|
||||||
|
check_names
|
||||||
|
|
||||||
|
while IFS= read -r commit; do
|
||||||
|
check_ref "$commit"
|
||||||
|
done < <(git rev-list --all)
|
||||||
|
|
||||||
|
while IFS=$'\t' read -r object subject; do
|
||||||
|
if printf '%s\n%s\n' "$object" "$subject" | grep -E -q "$content_pattern"; then
|
||||||
|
fail "non-public wording found in reachable ref or commit subject: $object $subject"
|
||||||
|
fi
|
||||||
|
done < <(git for-each-ref --format='%(refname)%09%(subject)'; git log --all --format='%H%x09%s')
|
||||||
|
|
||||||
|
printf 'OK: public-content audit passed.\n'
|
||||||
Executable
+24
@@ -0,0 +1,24 @@
|
|||||||
|
#!/usr/bin/env bash
|
||||||
|
set -euo pipefail
|
||||||
|
|
||||||
|
repo_root="$(git rev-parse --show-toplevel)"
|
||||||
|
hook="$repo_root/.git/hooks/pre-push"
|
||||||
|
|
||||||
|
if [[ -e "$hook" && ! -f "$hook" ]]; then
|
||||||
|
printf 'Cannot install guard: %s is not a regular file.\n' "$hook" >&2
|
||||||
|
exit 1
|
||||||
|
fi
|
||||||
|
|
||||||
|
if [[ -f "$hook" ]] && ! grep -Fq 'scripts/check_public_content.sh' "$hook"; then
|
||||||
|
printf 'Cannot install guard: existing pre-push hook is not managed by this repository.\n' >&2
|
||||||
|
exit 1
|
||||||
|
fi
|
||||||
|
|
||||||
|
cat > "$hook" <<'EOF'
|
||||||
|
#!/usr/bin/env bash
|
||||||
|
set -euo pipefail
|
||||||
|
repo_root="$(git rev-parse --show-toplevel)"
|
||||||
|
exec bash "$repo_root/scripts/check_public_content.sh"
|
||||||
|
EOF
|
||||||
|
chmod +x "$hook"
|
||||||
|
printf 'Installed public-content pre-push guard.\n'
|
||||||
@@ -0,0 +1,340 @@
|
|||||||
|
#!/usr/bin/env julia
|
||||||
|
#############################################################################
|
||||||
|
## run_model.jl
|
||||||
|
## Main runner for latent trait model
|
||||||
|
## Executes the complete pipeline: data loading → preparation → model fitting
|
||||||
|
##
|
||||||
|
## Supports two model versions:
|
||||||
|
## - "2dim": 2D bipolar model (V1) - estimates economic left-right and cultural cosmopolitan--traditionalist positions directly
|
||||||
|
## - "4dim": 4D unipolar model (V10) - estimates 4 traits, derives 2 scales
|
||||||
|
##
|
||||||
|
## Default is 2D model (better identification, faster convergence)
|
||||||
|
#############################################################################
|
||||||
|
|
||||||
|
using Dates
|
||||||
|
|
||||||
|
#############################################################################
|
||||||
|
## EXECUTION CONFIGURATION (Change these values as needed)
|
||||||
|
#############################################################################
|
||||||
|
const MODEL_VERSION = "2dim" # "2dim" (recommended) or "4dim"
|
||||||
|
const STAN_MODEL_FILE = MODEL_VERSION == "2dim" ? "models/stan_model_2dim_v6.stan" : "models/stan_model_4dim_v10.stan"
|
||||||
|
const NUM_CHAINS = parse(Int, get(ENV, "PARTY2D_NUM_CHAINS", "4"))
|
||||||
|
const NUM_WARMUP = parse(Int, get(ENV, "PARTY2D_NUM_WARMUP", "1000"))
|
||||||
|
const NUM_SAMPLES = parse(Int, get(ENV, "PARTY2D_NUM_SAMPLES", "2000"))
|
||||||
|
const ADAPT_DELTA = 0.95 # Target acceptance probability
|
||||||
|
const MAX_DEPTH = 15 # Maximum tree depth
|
||||||
|
const START_YEAR = 1944 # First year to include (avoid sparse early data)
|
||||||
|
|
||||||
|
println("=" ^ 70)
|
||||||
|
println("Latent Trait Model - Estimation Pipeline")
|
||||||
|
println("=" ^ 70)
|
||||||
|
println("Started at: ", Dates.now())
|
||||||
|
println("Configuration:")
|
||||||
|
println(" Model version: $(MODEL_VERSION)")
|
||||||
|
println(" Stan model: $(STAN_MODEL_FILE)")
|
||||||
|
println(" Chains: $(NUM_CHAINS)")
|
||||||
|
println(" Warmup iterations: $(NUM_WARMUP)")
|
||||||
|
println(" Sampling iterations: $(NUM_SAMPLES)")
|
||||||
|
println(" Total iterations per chain: $(NUM_WARMUP + NUM_SAMPLES)")
|
||||||
|
println(" Adapt delta: $(ADAPT_DELTA)")
|
||||||
|
println(" Max depth: $(MAX_DEPTH)")
|
||||||
|
println(" Start year: $(START_YEAR)")
|
||||||
|
if MODEL_VERSION == "2dim"
|
||||||
|
println("\n 2D MODEL: Estimates economic left-right and cultural cosmopolitan--traditionalist positions directly")
|
||||||
|
println(" (Half the parameters, better convergence)")
|
||||||
|
else
|
||||||
|
println("\n 4D MODEL: Estimates 4 traits, derives 2 scales")
|
||||||
|
println(" (Known identification issues - see VERSION_HISTORY.md)")
|
||||||
|
end
|
||||||
|
println("=" ^ 70)
|
||||||
|
|
||||||
|
# Include pipeline modules
|
||||||
|
include("pipeline/00_validation.jl") # Validation checks
|
||||||
|
include("pipeline/02_data_loading.jl")
|
||||||
|
include("pipeline/03_data_preparation.jl")
|
||||||
|
include("pipeline/04_model_execution.jl")
|
||||||
|
include("pipeline/05_results_processing.jl")
|
||||||
|
|
||||||
|
# Load robust save module
|
||||||
|
include("pipeline/06_save_model.jl")
|
||||||
|
import .RobustSave: robust_save_model
|
||||||
|
|
||||||
|
function run_model(;
|
||||||
|
num_chains=NUM_CHAINS,
|
||||||
|
num_warmup=NUM_WARMUP,
|
||||||
|
num_samples=NUM_SAMPLES,
|
||||||
|
adapt_delta=ADAPT_DELTA,
|
||||||
|
max_depth=MAX_DEPTH,
|
||||||
|
model_file=STAN_MODEL_FILE,
|
||||||
|
start_year=START_YEAR,
|
||||||
|
data_dir="data"
|
||||||
|
)
|
||||||
|
"""Run the complete latent trait model pipeline (2D or 4D based on MODEL_VERSION)"""
|
||||||
|
|
||||||
|
try
|
||||||
|
# Step 1: Load and preprocess data
|
||||||
|
println("\n" * "="^50)
|
||||||
|
println("STEP 1: DATA LOADING")
|
||||||
|
println("="^50)
|
||||||
|
|
||||||
|
manifesto, expert_dim, expert_lr, year0, union_to_constituents, constituent_to_union = load_and_preprocess_4dim_data(start_year; data_dir=data_dir)
|
||||||
|
|
||||||
|
# Step 2: Prepare Stan data structure
|
||||||
|
println("\n" * "="^50)
|
||||||
|
println("STEP 2: DATA PREPARATION")
|
||||||
|
println("="^50)
|
||||||
|
|
||||||
|
# Prepare indices and mappings (V4: union-aware)
|
||||||
|
data_prep = prepare_4dim_stan_data(manifesto, expert_dim, expert_lr, year0;
|
||||||
|
union_to_constituents=union_to_constituents,
|
||||||
|
constituent_to_union=constituent_to_union)
|
||||||
|
|
||||||
|
# Finalize Stan data dictionary (V4: includes constituent arrays)
|
||||||
|
final_data = finalize_4dim_stan_data(
|
||||||
|
data_prep.manifesto, data_prep.expert_dim, data_prep.expert_lr,
|
||||||
|
data_prep.segment_year, data_prep.segment_info,
|
||||||
|
data_prep.all_parties, data_prep.all_groups,
|
||||||
|
data_prep.group_to_index, year0, data_prep.S, data_prep.J, data_prep.P, data_prep.R,
|
||||||
|
data_prep.N_ciy, data_prep.len_theta_ts, data_prep.segment_country_idx,
|
||||||
|
data_prep.F, data_prep.segment_family_idx, data_prep.anchor_segment_idx;
|
||||||
|
N_const_man_total=data_prep.N_const_man_total,
|
||||||
|
n_const_man=data_prep.n_const_man,
|
||||||
|
const_offset_man=data_prep.const_offset_man,
|
||||||
|
const_rr_man=data_prep.const_rr_man,
|
||||||
|
N_const_exp_dim_total=data_prep.N_const_exp_dim_total,
|
||||||
|
n_const_exp_dim=data_prep.n_const_exp_dim,
|
||||||
|
const_offset_exp_dim=data_prep.const_offset_exp_dim,
|
||||||
|
const_rr_exp_dim=data_prep.const_rr_exp_dim,
|
||||||
|
N_const_exp_lr_total=data_prep.N_const_exp_lr_total,
|
||||||
|
n_const_exp_lr=data_prep.n_const_exp_lr,
|
||||||
|
const_offset_exp_lr=data_prep.const_offset_exp_lr,
|
||||||
|
const_rr_exp_lr=data_prep.const_rr_exp_lr
|
||||||
|
)
|
||||||
|
|
||||||
|
dat_4dim = final_data.dat_4dim
|
||||||
|
|
||||||
|
# Step 3: Validate data BEFORE running Stan
|
||||||
|
println("\n" * "="^50)
|
||||||
|
println("STEP 3: DATA VALIDATION")
|
||||||
|
println("="^50)
|
||||||
|
|
||||||
|
if !validate_stan_data(dat_4dim; verbose=true)
|
||||||
|
error("Data validation failed - see errors above")
|
||||||
|
end
|
||||||
|
|
||||||
|
estimate_memory_requirements(dat_4dim; verbose=true)
|
||||||
|
|
||||||
|
# Step 4: Create initialization function
|
||||||
|
println("\n" * "="^50)
|
||||||
|
println("STEP 4: MODEL INITIALIZATION")
|
||||||
|
println("="^50)
|
||||||
|
|
||||||
|
# Determine model version for initialization
|
||||||
|
model_init_version = MODEL_VERSION == "2dim" ? "v1_2dim" : "v10"
|
||||||
|
|
||||||
|
# Use S (segments) for initialization, not J (parties)
|
||||||
|
init_fn = create_init_function(dat_4dim, data_prep.S, data_prep.P,
|
||||||
|
data_prep.R, final_data.T_year, data_prep.N_ciy;
|
||||||
|
model_version=model_init_version)
|
||||||
|
|
||||||
|
# Validate initialization values
|
||||||
|
println("\nValidating initialization for chain 1...")
|
||||||
|
test_init = init_fn()
|
||||||
|
if !validate_init_values(test_init; verbose=true)
|
||||||
|
error("Initialization validation failed - see errors above")
|
||||||
|
end
|
||||||
|
|
||||||
|
# Step 5: Run Stan model
|
||||||
|
println("\n" * "="^50)
|
||||||
|
println("STEP 5: MODEL EXECUTION")
|
||||||
|
println("="^50)
|
||||||
|
|
||||||
|
# Create temp folder for output
|
||||||
|
temp_folder = mktempdir()
|
||||||
|
println("Temporary folder for Stan output: $temp_folder")
|
||||||
|
|
||||||
|
# Run Stan model
|
||||||
|
stanmodel = run_4dim_stan_model(
|
||||||
|
dat_4dim, init_fn, temp_folder;
|
||||||
|
num_chains=num_chains,
|
||||||
|
num_warmup=num_warmup,
|
||||||
|
num_samples=num_samples,
|
||||||
|
adapt_delta=adapt_delta,
|
||||||
|
max_depth=max_depth,
|
||||||
|
model_file=model_file
|
||||||
|
)
|
||||||
|
|
||||||
|
# Step 6: Results Processing & Diagnostics
|
||||||
|
println("\n" * "="^50)
|
||||||
|
println("STEP 6: RESULTS PROCESSING & DIAGNOSTICS")
|
||||||
|
println("="^50)
|
||||||
|
|
||||||
|
results = extract_model_results_4dim(stanmodel)
|
||||||
|
diagnostics = compute_model_diagnostics(stanmodel)
|
||||||
|
|
||||||
|
println("\nCONVERGENCE DIAGNOSTICS:")
|
||||||
|
println(" Max R-hat: $(round(diagnostics.max_rhat, digits=4))")
|
||||||
|
println(" Mean R-hat: $(round(diagnostics.mean_rhat, digits=4))")
|
||||||
|
println(" High R-hat count: $(diagnostics.high_rhat_count)")
|
||||||
|
println(" Min ESS: $(round(diagnostics.min_ess, digits=0))")
|
||||||
|
println(" Mean ESS: $(round(diagnostics.mean_ess, digits=0))")
|
||||||
|
println(" Convergence Status: $(diagnostics.convergence_status)")
|
||||||
|
|
||||||
|
# Step 7: Save results
|
||||||
|
println("\n" * "="^50)
|
||||||
|
println("STEP 7: SAVING RESULTS")
|
||||||
|
println("="^50)
|
||||||
|
println(" Using save-local-then-move strategy...")
|
||||||
|
|
||||||
|
# Prepare data for saving (SINGLE copy of stanmodel, not multiple!)
|
||||||
|
model_data_to_save = Dict{String, Any}(
|
||||||
|
# StanModel object (contains all MCMC samples)
|
||||||
|
"stanmodel_object" => stanmodel,
|
||||||
|
|
||||||
|
# Processed diagnostics
|
||||||
|
"diagnostics_summary" => diagnostics.diagnostics_summary,
|
||||||
|
"data_dict" => dat_4dim,
|
||||||
|
|
||||||
|
# Original data for reference
|
||||||
|
"manifesto" => final_data.manifesto,
|
||||||
|
"expert_dim" => final_data.expert_dim,
|
||||||
|
"expert_lr" => final_data.expert_lr,
|
||||||
|
"segment_year" => final_data.segment_year, # V10: segment-year mapping
|
||||||
|
"segment_info" => final_data.segment_info, # V10: segment metadata (party_id, segment_num, year range)
|
||||||
|
|
||||||
|
# Metadata with convergence info
|
||||||
|
"model_info" => Dict(
|
||||||
|
"timestamp" => Dates.format(Dates.now(), "yyyy-mm-dd_HH-MM-SS"),
|
||||||
|
"max_rhat" => diagnostics.max_rhat,
|
||||||
|
"mean_rhat" => diagnostics.mean_rhat,
|
||||||
|
"min_ess" => diagnostics.min_ess,
|
||||||
|
"mean_ess" => diagnostics.mean_ess,
|
||||||
|
"convergence_status" => diagnostics.convergence_status,
|
||||||
|
"model_file" => model_file,
|
||||||
|
"model_version" => MODEL_VERSION,
|
||||||
|
"num_chains" => num_chains,
|
||||||
|
"num_warmup" => num_warmup,
|
||||||
|
"num_samples" => num_samples,
|
||||||
|
"adapt_delta" => adapt_delta,
|
||||||
|
"max_depth" => max_depth,
|
||||||
|
"year0" => year0,
|
||||||
|
"dimensions" => MODEL_VERSION == "2dim" ?
|
||||||
|
["economic_lr", "galtan"] :
|
||||||
|
["pro_market", "pro_welfare", "cosmopolitan", "traditional"]
|
||||||
|
)
|
||||||
|
)
|
||||||
|
|
||||||
|
# Save using CSV-first robust system. Chains have already been secured
|
||||||
|
# by model execution; robust_save_model adds metadata/data and verifies.
|
||||||
|
save_dir = data_dir != "data" ? joinpath(data_dir, "model_run") : "outputs/model_outputs"
|
||||||
|
output_file = robust_save_model(
|
||||||
|
stanmodel,
|
||||||
|
model_data_to_save,
|
||||||
|
save_dir;
|
||||||
|
compress=true, # Ignored by CSV-first save implementation
|
||||||
|
keep_local_backups=2 # Ignored; chains are already saved before this step
|
||||||
|
)
|
||||||
|
|
||||||
|
# Robust save module already verified everything!
|
||||||
|
|
||||||
|
println("\n" * "="^70)
|
||||||
|
println("MODEL EXECUTION COMPLETED SUCCESSFULLY!")
|
||||||
|
println("="^70)
|
||||||
|
println(" Max R-hat: $(round(diagnostics.max_rhat, digits=4))")
|
||||||
|
println(" Mean R-hat: $(round(diagnostics.mean_rhat, digits=4))")
|
||||||
|
println(" Convergence: $(diagnostics.convergence_status)")
|
||||||
|
println(" Output file: $output_file")
|
||||||
|
|
||||||
|
# Print summary statistics
|
||||||
|
println("\nModel Summary ($(MODEL_VERSION == "2dim" ? "2D Direct Bipolar" : "4D Unipolar")):")
|
||||||
|
println(" Segments: $(data_prep.S)")
|
||||||
|
println(" Parties with valid segments: $(data_prep.J)")
|
||||||
|
println(" Countries: $(data_prep.P)")
|
||||||
|
println(" Segment-year combinations: $(data_prep.R)")
|
||||||
|
println(" Years: $(final_data.T_year)")
|
||||||
|
println(" Manifesto observations: $(dat_4dim["N_man"])")
|
||||||
|
println(" Expert dimension-specific observations: $(dat_4dim["N_exp_dim"])")
|
||||||
|
println(" Expert general L-R observations: $(dat_4dim["N_exp_lr"])")
|
||||||
|
if MODEL_VERSION == "2dim"
|
||||||
|
println(" Dimensions estimated: 2 (economic left-right, cultural cosmopolitan--traditionalist)")
|
||||||
|
println(" Theta parameters: $(2 * data_prep.R) (2 × R)")
|
||||||
|
else
|
||||||
|
println(" Dimensions estimated: 4 (pro_market, pro_welfare, cosmopolitan, traditional)")
|
||||||
|
println(" Theta parameters: $(4 * data_prep.R) (4 × R)")
|
||||||
|
end
|
||||||
|
println("=" ^ 70)
|
||||||
|
|
||||||
|
# Cleanup temp folder after successful save
|
||||||
|
println("\nCLEANING UP TEMPORARY FILES...")
|
||||||
|
try
|
||||||
|
if isdir(temp_folder)
|
||||||
|
rm(temp_folder, recursive=true, force=true)
|
||||||
|
println(" Removed temporary folder: $temp_folder")
|
||||||
|
end
|
||||||
|
catch cleanup_error
|
||||||
|
println(" Warning: Could not remove temp folder: $cleanup_error")
|
||||||
|
println(" (This won't affect your saved results)")
|
||||||
|
end
|
||||||
|
|
||||||
|
return true
|
||||||
|
|
||||||
|
catch e
|
||||||
|
println("\nERROR in model pipeline: $e")
|
||||||
|
println("Stack trace:")
|
||||||
|
showerror(stdout, e, catch_backtrace())
|
||||||
|
rethrow(e)
|
||||||
|
end
|
||||||
|
end
|
||||||
|
|
||||||
|
function main(args=ARGS)
|
||||||
|
# Parse --data-dir argument
|
||||||
|
data_dir = "data"
|
||||||
|
for (i, arg) in enumerate(args)
|
||||||
|
if arg == "--data-dir" && i < length(args)
|
||||||
|
data_dir = args[i + 1]
|
||||||
|
elseif startswith(arg, "--data-dir=")
|
||||||
|
data_dir = split(arg, "=", limit=2)[2]
|
||||||
|
end
|
||||||
|
end
|
||||||
|
|
||||||
|
if data_dir != "."
|
||||||
|
println("Using data directory: $data_dir")
|
||||||
|
end
|
||||||
|
println("Executing $(MODEL_VERSION) latent trait model pipeline...")
|
||||||
|
|
||||||
|
# Check that required data files exist (in data_dir)
|
||||||
|
required_data = [joinpath(data_dir, f) for f in ["text_data.csv", "expert.csv", "lr_data.csv"]]
|
||||||
|
required_files = vcat(required_data, [STAN_MODEL_FILE])
|
||||||
|
missing_files = []
|
||||||
|
|
||||||
|
for file in required_files
|
||||||
|
if !isfile(file)
|
||||||
|
push!(missing_files, file)
|
||||||
|
end
|
||||||
|
end
|
||||||
|
|
||||||
|
if !isempty(missing_files)
|
||||||
|
println("ERROR: Missing required files:")
|
||||||
|
for file in missing_files
|
||||||
|
println(" - $file")
|
||||||
|
end
|
||||||
|
|
||||||
|
if any(f -> endswith(f, "text_data.csv") || endswith(f, "expert.csv") || endswith(f, "lr_data.csv"), missing_files)
|
||||||
|
println("\nTo generate data files, run:")
|
||||||
|
println(" bash scripts/01_prepare_data.sh")
|
||||||
|
end
|
||||||
|
|
||||||
|
error("Cannot proceed without required files")
|
||||||
|
end
|
||||||
|
|
||||||
|
# Run the complete pipeline
|
||||||
|
results = run_model(data_dir=data_dir)
|
||||||
|
|
||||||
|
println("\n$(MODEL_VERSION) latent trait model pipeline completed successfully!")
|
||||||
|
println("Check outputs/model_outputs/latest/ for chain CSVs and metadata.")
|
||||||
|
end
|
||||||
|
|
||||||
|
# Main execution
|
||||||
|
if abspath(PROGRAM_FILE) == @__FILE__
|
||||||
|
main()
|
||||||
|
end
|
||||||
@@ -0,0 +1,138 @@
|
|||||||
|
#!/usr/bin/env julia
|
||||||
|
#=
|
||||||
|
02_enrich_output.jl - Enrich party positions CSV with model-input-derived metadata
|
||||||
|
|
||||||
|
Fast post-processing script that adds union membership status from the model-ready
|
||||||
|
inputs. Operates purely on CSV files (no chain loading). Takes seconds, not minutes.
|
||||||
|
|
||||||
|
Usage:
|
||||||
|
julia 02_enrich_output.jl # enriches latest party_positions_*.csv
|
||||||
|
julia 02_enrich_output.jl somefile.csv # enriches a specific file
|
||||||
|
|
||||||
|
Adds columns:
|
||||||
|
in_union - 1 if party's union had a joint manifesto that year, 0 otherwise
|
||||||
|
=#
|
||||||
|
|
||||||
|
using CSV
|
||||||
|
using DataFrames
|
||||||
|
using Dates
|
||||||
|
|
||||||
|
function find_latest_output()
|
||||||
|
outdir = "outputs/estimations/latest"
|
||||||
|
files = filter(f -> startswith(f, "party_positions_") && endswith(f, ".csv") &&
|
||||||
|
!contains(f, "metadata") && !contains(f, "tables"),
|
||||||
|
readdir(outdir))
|
||||||
|
isempty(files) && error("No party_positions_*.csv found in $outdir/")
|
||||||
|
sort!(files, rev=true)
|
||||||
|
return joinpath(outdir, files[1])
|
||||||
|
end
|
||||||
|
|
||||||
|
function enrich(input_file::String)
|
||||||
|
println("="^60)
|
||||||
|
println("ENRICH OUTPUT")
|
||||||
|
println("="^60)
|
||||||
|
println("Input: $input_file")
|
||||||
|
|
||||||
|
output = CSV.read(input_file, DataFrame)
|
||||||
|
println(" Rows: $(nrow(output)), Columns: $(ncol(output))")
|
||||||
|
|
||||||
|
# --- Union mapping (for in_union) ---
|
||||||
|
union_mapping_file = joinpath("data", "union_mapping.csv")
|
||||||
|
constituent_to_union = Dict{Int, Int}()
|
||||||
|
if isfile(union_mapping_file)
|
||||||
|
union_df = CSV.read(union_mapping_file, DataFrame)
|
||||||
|
for row in eachrow(union_df)
|
||||||
|
constituent_to_union[row.expert_pf_id] = row.manifesto_pf_id
|
||||||
|
end
|
||||||
|
end
|
||||||
|
|
||||||
|
# --- in_union dummy (year-varying) ---
|
||||||
|
text_data_file = "data/text_data.csv"
|
||||||
|
output.in_union = zeros(Int, nrow(output))
|
||||||
|
|
||||||
|
if isfile(text_data_file) && !isempty(constituent_to_union)
|
||||||
|
text_df = CSV.read(text_data_file, DataFrame)
|
||||||
|
|
||||||
|
# Build set of (party_pf_id, year) pairs for manifesto data
|
||||||
|
manifesto_text = filter(r -> r.project == "Manifesto Project", text_df)
|
||||||
|
manifesto_party_years = Set{Tuple{Int, Int}}()
|
||||||
|
for row in eachrow(manifesto_text)
|
||||||
|
push!(manifesto_party_years, (row.party, row.year))
|
||||||
|
end
|
||||||
|
|
||||||
|
# Process each (party, segment) group
|
||||||
|
gdf = groupby(output, [:party_id, :segment_num])
|
||||||
|
for subdf in gdf
|
||||||
|
pid = subdf.party_id[1]
|
||||||
|
!haskey(constituent_to_union, pid) && continue
|
||||||
|
union_id = constituent_to_union[pid]
|
||||||
|
|
||||||
|
# Get row indices in the full output for this group
|
||||||
|
idxs = parentindices(subdf)[1]
|
||||||
|
|
||||||
|
# Determine in_union at election years
|
||||||
|
election_year_vals = Dict{Int, Int}()
|
||||||
|
for (j, row) in enumerate(eachrow(subdf))
|
||||||
|
if (union_id, row.year) in manifesto_party_years
|
||||||
|
election_year_vals[row.year] = 1
|
||||||
|
elseif (pid, row.year) in manifesto_party_years
|
||||||
|
election_year_vals[row.year] = 0
|
||||||
|
end
|
||||||
|
end
|
||||||
|
|
||||||
|
# Forward-fill within segment
|
||||||
|
sorted_pairs = sort(collect(zip(subdf.year, idxs)))
|
||||||
|
last_val = 0
|
||||||
|
for (yr, idx) in sorted_pairs
|
||||||
|
if haskey(election_year_vals, yr)
|
||||||
|
last_val = election_year_vals[yr]
|
||||||
|
end
|
||||||
|
output.in_union[idx] = last_val
|
||||||
|
end
|
||||||
|
end
|
||||||
|
|
||||||
|
n_in_union = count(x -> x == 1, output.in_union)
|
||||||
|
println(" in_union: $n_in_union rows flagged as union members")
|
||||||
|
else
|
||||||
|
println(" WARNING: Could not compute in_union (missing files)")
|
||||||
|
end
|
||||||
|
|
||||||
|
# --- Reorder columns ---
|
||||||
|
estimate_cols = Symbol[]
|
||||||
|
for base in ["economic_lr", "galtan", "pro_market", "pro_welfare", "cosmopolitan", "traditional"]
|
||||||
|
sym = Symbol(base)
|
||||||
|
if hasproperty(output, sym)
|
||||||
|
push!(estimate_cols, sym)
|
||||||
|
push!(estimate_cols, Symbol("$(base)_se"))
|
||||||
|
push!(estimate_cols, Symbol("$(base)_q025"))
|
||||||
|
push!(estimate_cols, Symbol("$(base)_q975"))
|
||||||
|
end
|
||||||
|
end
|
||||||
|
|
||||||
|
# Fix country code: MO (Macau) → MK (North Macedonia) — GPS uses wrong ISO2
|
||||||
|
output.country = replace(output.country, "MO" => "MK")
|
||||||
|
|
||||||
|
col_order = vcat(
|
||||||
|
[:party_id, :country, :year, :segment_num, :union_party_id, :in_union],
|
||||||
|
estimate_cols
|
||||||
|
)
|
||||||
|
col_order = filter(c -> hasproperty(output, c), col_order)
|
||||||
|
select!(output, col_order)
|
||||||
|
|
||||||
|
# --- Write back ---
|
||||||
|
CSV.write(input_file, output)
|
||||||
|
println("\n Wrote: $input_file")
|
||||||
|
println(" Columns ($(ncol(output))): $(join(string.(names(output)), ", "))")
|
||||||
|
|
||||||
|
return output
|
||||||
|
end
|
||||||
|
|
||||||
|
function main(args=ARGS)
|
||||||
|
input = length(args) >= 1 ? args[1] : find_latest_output()
|
||||||
|
enrich(input)
|
||||||
|
println("\nDone.")
|
||||||
|
end
|
||||||
|
|
||||||
|
if abspath(PROGRAM_FILE) == @__FILE__
|
||||||
|
main()
|
||||||
|
end
|
||||||
@@ -0,0 +1,986 @@
|
|||||||
|
#!/usr/bin/env julia
|
||||||
|
#=
|
||||||
|
02_post_estimation.jl - Extract party position estimates from Stan model output
|
||||||
|
|
||||||
|
Supports both model versions:
|
||||||
|
- 2D model (V1): Extracts economic left-right and cultural cosmopolitan--traditionalist positions directly
|
||||||
|
- 4D model (V10): Extracts 4 traits + 2 derived scales
|
||||||
|
|
||||||
|
V10/V1 UPDATE: Handles segment-based indexing
|
||||||
|
- Maps segment results back to original party IDs
|
||||||
|
- Adds segment_num column to indicate which segment of the party
|
||||||
|
- Flags discontinuities for parties with multiple segments
|
||||||
|
|
||||||
|
This script:
|
||||||
|
1. Auto-detects the latest model run in model_outputs/
|
||||||
|
2. Loads the chain CSV files and data mappings
|
||||||
|
3. Detects model version from metadata or column names
|
||||||
|
4. Extracts posterior summaries for all segment-year positions
|
||||||
|
5. Maps Stan parameter indices back to real party IDs, segment numbers, and years
|
||||||
|
6. Saves output as wide-format CSV with uncertainty estimates
|
||||||
|
|
||||||
|
Usage:
|
||||||
|
julia 02_post_estimation.jl
|
||||||
|
|
||||||
|
Output:
|
||||||
|
party_positions_YYYY-MM-DD_HH-MM-SS.csv
|
||||||
|
=#
|
||||||
|
|
||||||
|
using CSV
|
||||||
|
using DataFrames
|
||||||
|
using Statistics
|
||||||
|
using JSON
|
||||||
|
using Dates
|
||||||
|
using Printf
|
||||||
|
|
||||||
|
# =============================================================================
|
||||||
|
# STEP 0: Auto-detect latest run
|
||||||
|
# =============================================================================
|
||||||
|
|
||||||
|
function find_latest_run(base_dir::String="outputs/model_outputs/latest")
|
||||||
|
if !isdir(base_dir)
|
||||||
|
error("Model outputs directory not found: $base_dir")
|
||||||
|
end
|
||||||
|
|
||||||
|
runs = filter(d -> startswith(d, "run_") && isdir(joinpath(base_dir, d)), readdir(base_dir))
|
||||||
|
|
||||||
|
if isempty(runs)
|
||||||
|
error("No runs found in $base_dir")
|
||||||
|
end
|
||||||
|
|
||||||
|
# Sort by timestamp in directory name (format: run_YYYY-MM-DD_HH-MM-SS)
|
||||||
|
sort!(runs, rev=true)
|
||||||
|
|
||||||
|
latest = joinpath(base_dir, runs[1])
|
||||||
|
println("Found $(length(runs)) run(s). Using latest: $latest")
|
||||||
|
return latest
|
||||||
|
end
|
||||||
|
|
||||||
|
# =============================================================================
|
||||||
|
# STEP 1: Load data and build segment-year lookup
|
||||||
|
# =============================================================================
|
||||||
|
|
||||||
|
function load_run_data(run_dir::String)
|
||||||
|
println("\n" * "="^60)
|
||||||
|
println("LOADING RUN DATA")
|
||||||
|
println("="^60)
|
||||||
|
|
||||||
|
data_dir = joinpath(run_dir, "data")
|
||||||
|
chains_dir = joinpath(run_dir, "chains")
|
||||||
|
|
||||||
|
# Check required files exist
|
||||||
|
required_files = [
|
||||||
|
joinpath(data_dir, "text_data.csv"),
|
||||||
|
joinpath(data_dir, "expert_dim.csv"),
|
||||||
|
joinpath(data_dir, "expert_lr.csv"),
|
||||||
|
joinpath(run_dir, "metadata.json")
|
||||||
|
]
|
||||||
|
|
||||||
|
for f in required_files
|
||||||
|
if !isfile(f)
|
||||||
|
error("Required file not found: $f")
|
||||||
|
end
|
||||||
|
end
|
||||||
|
|
||||||
|
# Load data files
|
||||||
|
println("Loading text_data.csv...")
|
||||||
|
text_data = CSV.read(joinpath(data_dir, "text_data.csv"), DataFrame)
|
||||||
|
println(" Rows: $(nrow(text_data))")
|
||||||
|
|
||||||
|
println("Loading expert_dim.csv...")
|
||||||
|
expert_dim = CSV.read(joinpath(data_dir, "expert_dim.csv"), DataFrame)
|
||||||
|
println(" Rows: $(nrow(expert_dim))")
|
||||||
|
|
||||||
|
println("Loading expert_lr.csv...")
|
||||||
|
expert_lr = CSV.read(joinpath(data_dir, "expert_lr.csv"), DataFrame)
|
||||||
|
println(" Rows: $(nrow(expert_lr))")
|
||||||
|
|
||||||
|
println("Loading metadata.json...")
|
||||||
|
metadata = JSON.parsefile(joinpath(run_dir, "metadata.json"))
|
||||||
|
println(" year0: $(metadata["year0"])")
|
||||||
|
println(" Model: $(metadata["model_file"])")
|
||||||
|
|
||||||
|
# V10: Load segment_info if available
|
||||||
|
segment_info_file = joinpath(data_dir, "segment_info.csv")
|
||||||
|
segment_info = nothing
|
||||||
|
if isfile(segment_info_file)
|
||||||
|
println("Loading segment_info.csv (V10)...")
|
||||||
|
segment_info = CSV.read(segment_info_file, DataFrame)
|
||||||
|
println(" Segments: $(nrow(segment_info))")
|
||||||
|
end
|
||||||
|
|
||||||
|
# V10: Load segment_year_map if available
|
||||||
|
segment_year_file = joinpath(data_dir, "segment_year_map.csv")
|
||||||
|
segment_year_map = nothing
|
||||||
|
if isfile(segment_year_file)
|
||||||
|
println("Loading segment_year_map.csv (V10)...")
|
||||||
|
segment_year_map = CSV.read(segment_year_file, DataFrame)
|
||||||
|
println(" Segment-years: $(nrow(segment_year_map))")
|
||||||
|
end
|
||||||
|
|
||||||
|
# Find chain files
|
||||||
|
chain_files = filter(f -> endswith(f, ".csv") && startswith(f, "chain_"), readdir(chains_dir))
|
||||||
|
println("\nFound $(length(chain_files)) chain file(s)")
|
||||||
|
|
||||||
|
return (
|
||||||
|
text_data = text_data,
|
||||||
|
expert_dim = expert_dim,
|
||||||
|
expert_lr = expert_lr,
|
||||||
|
metadata = metadata,
|
||||||
|
segment_info = segment_info,
|
||||||
|
segment_year_map = segment_year_map,
|
||||||
|
chain_files = [joinpath(chains_dir, f) for f in sort(chain_files)],
|
||||||
|
run_dir = run_dir
|
||||||
|
)
|
||||||
|
end
|
||||||
|
|
||||||
|
function normalize_country_value(value)
|
||||||
|
if ismissing(value)
|
||||||
|
return missing
|
||||||
|
end
|
||||||
|
txt = strip(string(value))
|
||||||
|
return isempty(txt) ? missing : txt
|
||||||
|
end
|
||||||
|
|
||||||
|
function build_party_country_map(text_data::DataFrame, expert_dim::DataFrame, expert_lr::DataFrame)
|
||||||
|
merged = unique(vcat(
|
||||||
|
select(text_data, :party, :country),
|
||||||
|
select(expert_dim, :party, :country),
|
||||||
|
select(expert_lr, :party, :country)
|
||||||
|
))
|
||||||
|
|
||||||
|
party_to_country = Dict{Int, String}()
|
||||||
|
for row in eachrow(merged)
|
||||||
|
pid = tryparse(Int, string(row.party))
|
||||||
|
if pid === nothing
|
||||||
|
continue
|
||||||
|
end
|
||||||
|
c = normalize_country_value(row.country)
|
||||||
|
if !ismissing(c)
|
||||||
|
party_to_country[pid] = c
|
||||||
|
end
|
||||||
|
end
|
||||||
|
return party_to_country
|
||||||
|
end
|
||||||
|
|
||||||
|
function load_constituent_to_union_map()::Dict{Int, Int}
|
||||||
|
mapping_file = joinpath("data", "union_mapping.csv")
|
||||||
|
constituent_to_union = Dict{Int, Int}()
|
||||||
|
if isfile(mapping_file)
|
||||||
|
union_df = CSV.read(mapping_file, DataFrame)
|
||||||
|
for row in eachrow(union_df)
|
||||||
|
constituent_to_union[row.expert_pf_id] = row.manifesto_pf_id
|
||||||
|
end
|
||||||
|
end
|
||||||
|
return constituent_to_union
|
||||||
|
end
|
||||||
|
|
||||||
|
function resolve_party_country(pid_value,
|
||||||
|
party_to_country::Dict{Int, String},
|
||||||
|
constituent_to_union::Dict{Int, Int})
|
||||||
|
pid = tryparse(Int, string(pid_value))
|
||||||
|
if pid === nothing
|
||||||
|
return missing, "unresolved"
|
||||||
|
end
|
||||||
|
|
||||||
|
if haskey(party_to_country, pid)
|
||||||
|
return party_to_country[pid], "direct"
|
||||||
|
end
|
||||||
|
|
||||||
|
if haskey(constituent_to_union, pid)
|
||||||
|
uid = constituent_to_union[pid]
|
||||||
|
if haskey(party_to_country, uid)
|
||||||
|
return party_to_country[uid], "union_fallback"
|
||||||
|
end
|
||||||
|
end
|
||||||
|
|
||||||
|
return missing, "unresolved"
|
||||||
|
end
|
||||||
|
|
||||||
|
function apply_country_resolution!(df::DataFrame,
|
||||||
|
party_col::Symbol,
|
||||||
|
country_col::Symbol,
|
||||||
|
party_to_country::Dict{Int, String},
|
||||||
|
constituent_to_union::Dict{Int, Int})
|
||||||
|
resolved_country = Union{Missing, String}[]
|
||||||
|
source_counts = Dict("direct" => 0, "union_fallback" => 0, "unresolved" => 0)
|
||||||
|
unresolved_parties = Set{Int}()
|
||||||
|
|
||||||
|
for pid in df[!, party_col]
|
||||||
|
country, source = resolve_party_country(pid, party_to_country, constituent_to_union)
|
||||||
|
push!(resolved_country, country)
|
||||||
|
source_counts[source] += 1
|
||||||
|
if source == "unresolved"
|
||||||
|
pid_int = tryparse(Int, string(pid))
|
||||||
|
if pid_int !== nothing
|
||||||
|
push!(unresolved_parties, pid_int)
|
||||||
|
end
|
||||||
|
end
|
||||||
|
end
|
||||||
|
|
||||||
|
df[!, country_col] = resolved_country
|
||||||
|
return source_counts, sort!(collect(unresolved_parties))
|
||||||
|
end
|
||||||
|
|
||||||
|
function fill_missing_countries!(df::DataFrame,
|
||||||
|
segment_info::Union{DataFrame, Nothing},
|
||||||
|
party_to_country::Dict{Int, String},
|
||||||
|
constituent_to_union::Dict{Int, Int})
|
||||||
|
if !hasproperty(df, :country)
|
||||||
|
source_counts, unresolved = apply_country_resolution!(
|
||||||
|
df, :party_id, :country, party_to_country, constituent_to_union
|
||||||
|
)
|
||||||
|
return source_counts, unresolved
|
||||||
|
end
|
||||||
|
|
||||||
|
normalized = Union{Missing, String}[]
|
||||||
|
for val in df.country
|
||||||
|
push!(normalized, normalize_country_value(val))
|
||||||
|
end
|
||||||
|
df.country = normalized
|
||||||
|
|
||||||
|
source_counts = Dict("direct" => 0, "union_fallback" => 0, "segment_info" => 0, "unresolved" => 0)
|
||||||
|
|
||||||
|
segment_country_by_id = Dict{Int, String}()
|
||||||
|
if segment_info !== nothing && hasproperty(segment_info, :country)
|
||||||
|
for row in eachrow(segment_info)
|
||||||
|
c = normalize_country_value(row.country)
|
||||||
|
if !ismissing(c)
|
||||||
|
segment_country_by_id[Int(row.segment_id)] = c
|
||||||
|
end
|
||||||
|
end
|
||||||
|
end
|
||||||
|
|
||||||
|
unresolved_parties = Set{Int}()
|
||||||
|
for i in 1:nrow(df)
|
||||||
|
if !ismissing(df.country[i])
|
||||||
|
continue
|
||||||
|
end
|
||||||
|
|
||||||
|
if hasproperty(df, :segment_id) && haskey(segment_country_by_id, Int(df.segment_id[i]))
|
||||||
|
df.country[i] = segment_country_by_id[Int(df.segment_id[i])]
|
||||||
|
source_counts["segment_info"] += 1
|
||||||
|
continue
|
||||||
|
end
|
||||||
|
|
||||||
|
country, source = resolve_party_country(df.party_id[i], party_to_country, constituent_to_union)
|
||||||
|
if !ismissing(country)
|
||||||
|
df.country[i] = country
|
||||||
|
source_counts[source] += 1
|
||||||
|
else
|
||||||
|
source_counts["unresolved"] += 1
|
||||||
|
pid_int = tryparse(Int, string(df.party_id[i]))
|
||||||
|
pid_int !== nothing && push!(unresolved_parties, pid_int)
|
||||||
|
end
|
||||||
|
end
|
||||||
|
|
||||||
|
return source_counts, sort!(collect(unresolved_parties))
|
||||||
|
end
|
||||||
|
|
||||||
|
# =============================================================================
|
||||||
|
# STEP 2: Build complete segment-year mapping (V10) or party-year mapping (V9)
|
||||||
|
# =============================================================================
|
||||||
|
|
||||||
|
function build_segment_year_map(text_data::DataFrame, expert_dim::DataFrame, expert_lr::DataFrame,
|
||||||
|
segment_info::Union{DataFrame, Nothing},
|
||||||
|
segment_year_map::Union{DataFrame, Nothing},
|
||||||
|
run_dir::String,
|
||||||
|
year0::Int)
|
||||||
|
println("\n" * "="^60)
|
||||||
|
println("BUILDING SEGMENT-YEAR MAPPING")
|
||||||
|
println("="^60)
|
||||||
|
|
||||||
|
party_to_country = build_party_country_map(text_data, expert_dim, expert_lr)
|
||||||
|
constituent_to_union = load_constituent_to_union_map()
|
||||||
|
|
||||||
|
# V10: Use segment_year_map if available
|
||||||
|
if segment_year_map !== nothing && segment_info !== nothing
|
||||||
|
println("Using segment_year_map.csv (V10 mode)")
|
||||||
|
|
||||||
|
# Convert relative Year to absolute year
|
||||||
|
if hasproperty(segment_year_map, :Year)
|
||||||
|
segment_year_map.year = segment_year_map.Year .+ year0
|
||||||
|
elseif !hasproperty(segment_year_map, :year)
|
||||||
|
error("segment_year_map has no Year or year column")
|
||||||
|
end
|
||||||
|
|
||||||
|
# Add party_id from segment_info if not already present
|
||||||
|
if !hasproperty(segment_year_map, :party_id)
|
||||||
|
segment_id_to_party = Dict(row.segment_id => row.party_id for row in eachrow(segment_info))
|
||||||
|
segment_year_map.party_id = [segment_id_to_party[sid] for sid in segment_year_map.segment_id]
|
||||||
|
end
|
||||||
|
|
||||||
|
# Add segment_num from segment_info if not already present
|
||||||
|
if !hasproperty(segment_year_map, :segment_num)
|
||||||
|
segment_id_to_segnum = Dict(row.segment_id => row.segment_num for row in eachrow(segment_info))
|
||||||
|
segment_year_map.segment_num = [segment_id_to_segnum[sid] for sid in segment_year_map.segment_id]
|
||||||
|
end
|
||||||
|
|
||||||
|
# Resolve/fill country column using segment metadata first, then direct and union-fallback lookup.
|
||||||
|
source_counts, unresolved = fill_missing_countries!(
|
||||||
|
segment_year_map, segment_info, party_to_country, constituent_to_union
|
||||||
|
)
|
||||||
|
direct_count = get(source_counts, "direct", 0)
|
||||||
|
union_count = get(source_counts, "union_fallback", 0)
|
||||||
|
segment_info_count = get(source_counts, "segment_info", 0)
|
||||||
|
unresolved_count = count(ismissing, segment_year_map.country)
|
||||||
|
println(" Country resolution fill counts: direct=$direct_count, union_fallback=$union_count, segment_info=$segment_info_count, unresolved_rows=$unresolved_count")
|
||||||
|
if !isempty(unresolved)
|
||||||
|
println(" Warning: unresolved country party IDs (first 20): $(unresolved[1:min(20, length(unresolved))])")
|
||||||
|
end
|
||||||
|
|
||||||
|
R = maximum(segment_year_map.rr)
|
||||||
|
n_segments = length(unique(segment_year_map.segment_id))
|
||||||
|
n_parties = length(unique(segment_year_map.party_id))
|
||||||
|
|
||||||
|
println("Loaded segment_year_map: $(nrow(segment_year_map)) segment-years (R=$R)")
|
||||||
|
println(" Unique segments: $n_segments")
|
||||||
|
println(" Unique parties: $n_parties")
|
||||||
|
|
||||||
|
# Count observed vs interpolated
|
||||||
|
observed_rrs = Set{Int}()
|
||||||
|
if hasproperty(text_data, :rr_man)
|
||||||
|
union!(observed_rrs, Set(text_data.rr_man))
|
||||||
|
end
|
||||||
|
if hasproperty(expert_dim, :rr_exp_dim)
|
||||||
|
union!(observed_rrs, Set(expert_dim.rr_exp_dim))
|
||||||
|
end
|
||||||
|
if hasproperty(expert_lr, :rr_exp_lr)
|
||||||
|
union!(observed_rrs, Set(expert_lr.rr_exp_lr))
|
||||||
|
end
|
||||||
|
|
||||||
|
n_observed = length(intersect(Set(segment_year_map.rr), observed_rrs))
|
||||||
|
n_interpolated = nrow(segment_year_map) - n_observed
|
||||||
|
println(" Observed segment-years: $n_observed")
|
||||||
|
println(" Interpolated segment-years: $n_interpolated")
|
||||||
|
|
||||||
|
return segment_year_map, R, segment_info
|
||||||
|
end
|
||||||
|
|
||||||
|
# V9 fallback: Use party_year_map
|
||||||
|
party_year_file = joinpath(run_dir, "data", "party_year_map.csv")
|
||||||
|
if isfile(party_year_file)
|
||||||
|
println("Loading party_year_map.csv (V9 fallback mode)")
|
||||||
|
party_year_map = CSV.read(party_year_file, DataFrame)
|
||||||
|
|
||||||
|
# Add party_id column (same as party for V9)
|
||||||
|
if !hasproperty(party_year_map, :party_id)
|
||||||
|
party_year_map.party_id = party_year_map.party
|
||||||
|
end
|
||||||
|
|
||||||
|
# Add segment_num column (always 1 for V9)
|
||||||
|
if !hasproperty(party_year_map, :segment_num)
|
||||||
|
party_year_map.segment_num = ones(Int, nrow(party_year_map))
|
||||||
|
end
|
||||||
|
|
||||||
|
# Add segment_id column (same as party index for V9)
|
||||||
|
if !hasproperty(party_year_map, :segment_id)
|
||||||
|
party_year_map.segment_id = party_year_map.party
|
||||||
|
end
|
||||||
|
|
||||||
|
# Convert relative Year to absolute year
|
||||||
|
if hasproperty(party_year_map, :Year)
|
||||||
|
party_year_map.year = party_year_map.Year .+ year0
|
||||||
|
elseif !hasproperty(party_year_map, :year)
|
||||||
|
error("party_year_map has no Year or year column")
|
||||||
|
end
|
||||||
|
|
||||||
|
# Resolve/fill country column
|
||||||
|
if !hasproperty(party_year_map, :country)
|
||||||
|
source_counts, unresolved = apply_country_resolution!(
|
||||||
|
party_year_map, :party_id, :country, party_to_country, constituent_to_union
|
||||||
|
)
|
||||||
|
direct_count = source_counts["direct"]
|
||||||
|
union_count = source_counts["union_fallback"]
|
||||||
|
unresolved_count = source_counts["unresolved"]
|
||||||
|
println(" Country resolution sources: direct=$direct_count, union_fallback=$union_count, unresolved=$unresolved_count")
|
||||||
|
if !isempty(unresolved)
|
||||||
|
println(" Warning: unresolved country party IDs (first 20): $(unresolved[1:min(20, length(unresolved))])")
|
||||||
|
end
|
||||||
|
else
|
||||||
|
normalized = Union{Missing, String}[]
|
||||||
|
for val in party_year_map.country
|
||||||
|
push!(normalized, normalize_country_value(val))
|
||||||
|
end
|
||||||
|
party_year_map.country = normalized
|
||||||
|
end
|
||||||
|
|
||||||
|
R = maximum(party_year_map.rr)
|
||||||
|
println("Loaded party_year_map: $(nrow(party_year_map)) party-years (R=$R)")
|
||||||
|
|
||||||
|
return party_year_map, R, nothing
|
||||||
|
end
|
||||||
|
|
||||||
|
# Fallback: Reconstruct from data files
|
||||||
|
@warn "No mapping file found, reconstructing from data (observed years only)"
|
||||||
|
|
||||||
|
# Extract unique party-year-rr combinations from text_data
|
||||||
|
text_map = unique(select(text_data, :party, :country, :year, :rr_man))
|
||||||
|
rename!(text_map, :rr_man => :rr)
|
||||||
|
text_map.party_id = text_map.party
|
||||||
|
text_map.segment_num = ones(Int, nrow(text_map))
|
||||||
|
|
||||||
|
expert_dim_map = unique(select(expert_dim, :party, :country, :year, :rr_exp_dim))
|
||||||
|
rename!(expert_dim_map, :rr_exp_dim => :rr)
|
||||||
|
expert_dim_map.party_id = expert_dim_map.party
|
||||||
|
expert_dim_map.segment_num = ones(Int, nrow(expert_dim_map))
|
||||||
|
|
||||||
|
expert_lr_map = unique(select(expert_lr, :party, :country, :year, :rr_exp_lr))
|
||||||
|
rename!(expert_lr_map, :rr_exp_lr => :rr)
|
||||||
|
expert_lr_map.party_id = expert_lr_map.party
|
||||||
|
expert_lr_map.segment_num = ones(Int, nrow(expert_lr_map))
|
||||||
|
|
||||||
|
combined = vcat(text_map, expert_dim_map, expert_lr_map)
|
||||||
|
segment_year_map = unique(combined)
|
||||||
|
sort!(segment_year_map, :rr)
|
||||||
|
|
||||||
|
R = maximum(segment_year_map.rr)
|
||||||
|
println("Reconstructed mapping: $(nrow(segment_year_map)) segment-years (R=$R)")
|
||||||
|
|
||||||
|
return segment_year_map, R, nothing
|
||||||
|
end
|
||||||
|
|
||||||
|
# =============================================================================
|
||||||
|
# STEP 3: Load and combine chains
|
||||||
|
# =============================================================================
|
||||||
|
|
||||||
|
function load_chains(chain_files::Vector{String})
|
||||||
|
println("\n" * "="^60)
|
||||||
|
println("LOADING STAN CHAINS")
|
||||||
|
println("="^60)
|
||||||
|
flush(stdout)
|
||||||
|
|
||||||
|
chains = DataFrame[]
|
||||||
|
|
||||||
|
# The full Stan CSVs are very wide (hundreds of thousands of columns). For
|
||||||
|
# post-estimation we only need party-position generated quantities. Reading
|
||||||
|
# all columns can take hours and allocate many GB of irrelevant parameters.
|
||||||
|
post_estimation_prefixes = (
|
||||||
|
"economic_lr.",
|
||||||
|
"galtan.",
|
||||||
|
"pro_market.",
|
||||||
|
"pro_welfare.",
|
||||||
|
"cosmopolitan.",
|
||||||
|
"traditional.",
|
||||||
|
)
|
||||||
|
keep_post_estimation_col(_i, name) = any(startswith(String(name), p) for p in post_estimation_prefixes)
|
||||||
|
|
||||||
|
for (i, f) in enumerate(chain_files)
|
||||||
|
println("Loading chain $i: $(basename(f))...")
|
||||||
|
flush(stdout)
|
||||||
|
# Skip comment lines (Stan header) and parse only needed quantities.
|
||||||
|
chain = CSV.read(f, DataFrame; comment="#", select=keep_post_estimation_col)
|
||||||
|
println(" Samples: $(nrow(chain)), Parameters: $(ncol(chain))")
|
||||||
|
flush(stdout)
|
||||||
|
push!(chains, chain)
|
||||||
|
end
|
||||||
|
|
||||||
|
# Combine chains
|
||||||
|
println("Combining selected chain columns...")
|
||||||
|
flush(stdout)
|
||||||
|
combined = vcat(chains...)
|
||||||
|
println("\nCombined: $(nrow(combined)) total samples")
|
||||||
|
println("Selected parameters: $(ncol(combined))")
|
||||||
|
flush(stdout)
|
||||||
|
|
||||||
|
return combined
|
||||||
|
end
|
||||||
|
|
||||||
|
# =============================================================================
|
||||||
|
# STEP 4: Extract generated quantities
|
||||||
|
# =============================================================================
|
||||||
|
|
||||||
|
"""
|
||||||
|
Detect model version from chain column names.
|
||||||
|
Returns "2dim" or "4dim".
|
||||||
|
"""
|
||||||
|
function detect_model_version(chains::DataFrame)
|
||||||
|
cols = names(chains)
|
||||||
|
# 2D model has economic_lr but NOT pro_market
|
||||||
|
has_economic_lr = any(c -> startswith(string(c), "economic_lr."), cols)
|
||||||
|
has_pro_market = any(c -> startswith(string(c), "pro_market."), cols)
|
||||||
|
|
||||||
|
if has_economic_lr && !has_pro_market
|
||||||
|
return "2dim"
|
||||||
|
elseif has_pro_market
|
||||||
|
return "4dim"
|
||||||
|
else
|
||||||
|
error("Could not detect model version from chain columns")
|
||||||
|
end
|
||||||
|
end
|
||||||
|
|
||||||
|
function extract_estimates(chains::DataFrame, segment_year_map::DataFrame, R::Int)
|
||||||
|
println("\n" * "="^60)
|
||||||
|
println("EXTRACTING POSTERIOR ESTIMATES")
|
||||||
|
println("="^60)
|
||||||
|
|
||||||
|
# Auto-detect model version from columns
|
||||||
|
model_version = detect_model_version(chains)
|
||||||
|
println("Detected model version: $model_version")
|
||||||
|
|
||||||
|
# Select quantities based on model version
|
||||||
|
if model_version == "2dim"
|
||||||
|
# 2D model: economic left-right and cultural cosmopolitan--traditionalist positions are directly estimated
|
||||||
|
# (general_lr is computed in Stan for anchoring but not extracted as output)
|
||||||
|
quantities = ["economic_lr", "galtan"]
|
||||||
|
test_col = "economic_lr.1"
|
||||||
|
else
|
||||||
|
# 4D model: 4 traits + 2 derived scales
|
||||||
|
quantities = ["pro_market", "pro_welfare", "cosmopolitan", "traditional", "economic_lr", "galtan"]
|
||||||
|
test_col = "pro_market.1"
|
||||||
|
end
|
||||||
|
|
||||||
|
# Check that columns exist
|
||||||
|
if !hasproperty(chains, Symbol(test_col))
|
||||||
|
error("Column $test_col not found in chains. Available columns: $(first(names(chains), 10))...")
|
||||||
|
end
|
||||||
|
|
||||||
|
n_samples = nrow(chains)
|
||||||
|
println("Samples per parameter: $n_samples")
|
||||||
|
|
||||||
|
# Load union mapping for adding union_party_id column
|
||||||
|
union_mapping_file = joinpath("data", "union_mapping.csv")
|
||||||
|
constituent_to_union_pf = Dict{Int, Int}()
|
||||||
|
if isfile(union_mapping_file)
|
||||||
|
union_df = CSV.read(union_mapping_file, DataFrame)
|
||||||
|
for row in eachrow(union_df)
|
||||||
|
constituent_to_union_pf[row.expert_pf_id] = row.manifesto_pf_id
|
||||||
|
end
|
||||||
|
end
|
||||||
|
|
||||||
|
# Pre-allocate output DataFrame
|
||||||
|
n_rows = nrow(segment_year_map)
|
||||||
|
|
||||||
|
# Add union_party_id column: NA for standalone parties, union PF ID for constituents
|
||||||
|
union_ids = Union{Int, Missing}[]
|
||||||
|
for pid in segment_year_map.party_id
|
||||||
|
pid_int = isa(pid, Integer) ? pid : tryparse(Int, string(pid))
|
||||||
|
if pid_int !== nothing && haskey(constituent_to_union_pf, pid_int)
|
||||||
|
push!(union_ids, constituent_to_union_pf[pid_int])
|
||||||
|
else
|
||||||
|
push!(union_ids, missing)
|
||||||
|
end
|
||||||
|
end
|
||||||
|
|
||||||
|
output = DataFrame(
|
||||||
|
party_id = segment_year_map.party_id,
|
||||||
|
union_party_id = union_ids,
|
||||||
|
segment_num = segment_year_map.segment_num,
|
||||||
|
country = segment_year_map.country,
|
||||||
|
year = segment_year_map.year,
|
||||||
|
rr = segment_year_map.rr
|
||||||
|
)
|
||||||
|
|
||||||
|
# Add columns for each quantity
|
||||||
|
for q in quantities
|
||||||
|
output[!, Symbol(q)] = zeros(Float64, n_rows)
|
||||||
|
output[!, Symbol("$(q)_se")] = zeros(Float64, n_rows)
|
||||||
|
output[!, Symbol("$(q)_q025")] = zeros(Float64, n_rows)
|
||||||
|
output[!, Symbol("$(q)_q975")] = zeros(Float64, n_rows)
|
||||||
|
end
|
||||||
|
|
||||||
|
println("Extracting estimates for $(n_rows) segment-year positions...")
|
||||||
|
|
||||||
|
# Progress tracking
|
||||||
|
prog_interval = max(1, n_rows ÷ 20)
|
||||||
|
|
||||||
|
for (i, row) in enumerate(eachrow(segment_year_map))
|
||||||
|
r = row.rr
|
||||||
|
|
||||||
|
# Progress
|
||||||
|
if i % prog_interval == 0 || i == n_rows
|
||||||
|
pct = round(100 * i / n_rows, digits=1)
|
||||||
|
print("\r Progress: $pct% ($i / $n_rows)")
|
||||||
|
end
|
||||||
|
|
||||||
|
for q in quantities
|
||||||
|
col_name = Symbol("$q.$r")
|
||||||
|
|
||||||
|
if !hasproperty(chains, col_name)
|
||||||
|
@warn "Column $col_name not found (rr=$r)" maxlog=5
|
||||||
|
continue
|
||||||
|
end
|
||||||
|
|
||||||
|
samples = chains[!, col_name]
|
||||||
|
|
||||||
|
# Compute summary statistics
|
||||||
|
output[i, Symbol(q)] = mean(samples)
|
||||||
|
output[i, Symbol("$(q)_se")] = std(samples)
|
||||||
|
output[i, Symbol("$(q)_q025")] = quantile(samples, 0.025)
|
||||||
|
output[i, Symbol("$(q)_q975")] = quantile(samples, 0.975)
|
||||||
|
end
|
||||||
|
end
|
||||||
|
println() # Newline after progress
|
||||||
|
|
||||||
|
# Remove the rr column from final output (internal only)
|
||||||
|
select!(output, Not(:rr))
|
||||||
|
|
||||||
|
return output
|
||||||
|
end
|
||||||
|
|
||||||
|
# =============================================================================
|
||||||
|
# STEP 5: Validation
|
||||||
|
# =============================================================================
|
||||||
|
|
||||||
|
function validate_output(output::DataFrame, segment_info::Union{DataFrame, Nothing})
|
||||||
|
println("\n" * "="^60)
|
||||||
|
println("VALIDATION CHECKS")
|
||||||
|
println("="^60)
|
||||||
|
|
||||||
|
all_passed = true
|
||||||
|
|
||||||
|
# Detect which columns are present (2D vs 4D model)
|
||||||
|
has_4d = hasproperty(output, :pro_market)
|
||||||
|
|
||||||
|
# Check 1: Range check - all estimates should be in [0, 1]
|
||||||
|
println("\n1. Range check (all values in [0, 1]):")
|
||||||
|
|
||||||
|
if has_4d
|
||||||
|
check_cols = [:pro_market, :pro_welfare, :cosmopolitan, :traditional, :economic_lr, :galtan]
|
||||||
|
else
|
||||||
|
check_cols = [:economic_lr, :galtan]
|
||||||
|
end
|
||||||
|
|
||||||
|
for col in check_cols
|
||||||
|
if !hasproperty(output, col)
|
||||||
|
continue
|
||||||
|
end
|
||||||
|
vals = output[!, col]
|
||||||
|
min_val, max_val = extrema(vals)
|
||||||
|
in_range = min_val >= 0 && max_val <= 1
|
||||||
|
status = in_range ? "PASS" : "FAIL"
|
||||||
|
println(" $col: [$(@sprintf("%.4f", min_val)), $(@sprintf("%.4f", max_val))] - $status")
|
||||||
|
all_passed = all_passed && in_range
|
||||||
|
end
|
||||||
|
|
||||||
|
# Check 2: Anchor party checks
|
||||||
|
println("\n2. Anchor party checks:")
|
||||||
|
|
||||||
|
# Define anchor parties with expected ranges (for 2D model)
|
||||||
|
# Includes both union IDs (V3) and individual constituent IDs (V4)
|
||||||
|
anchor_parties = [
|
||||||
|
(id=211, name="CDU/CSU", country="DE", econ=(0.50, 0.70), galtan=(0.45, 0.70)),
|
||||||
|
(id=1375, name="CDU", country="DE", econ=(0.50, 0.70), galtan=(0.45, 0.65)),
|
||||||
|
(id=1731, name="CSU", country="DE", econ=(0.50, 0.70), galtan=(0.55, 0.75)),
|
||||||
|
(id=383, name="SPD", country="DE", econ=(0.30, 0.50), galtan=(0.30, 0.55)),
|
||||||
|
(id=1516, name="Labour", country="GB", econ=(0.30, 0.55), galtan=(0.30, 0.55)),
|
||||||
|
(id=1567, name="Conservatives", country="GB", econ=(0.55, 0.80), galtan=(0.50, 0.75)),
|
||||||
|
(id=487, name="SAP", country="SE", econ=(0.30, 0.50), galtan=(0.35, 0.55)),
|
||||||
|
]
|
||||||
|
|
||||||
|
n_checked = 0
|
||||||
|
n_passed = 0
|
||||||
|
|
||||||
|
for anchor in anchor_parties
|
||||||
|
party_rows = filter(r -> r.party_id == anchor.id, output)
|
||||||
|
|
||||||
|
if nrow(party_rows) == 0
|
||||||
|
println(" $(anchor.name) ($(anchor.id)): NOT FOUND")
|
||||||
|
continue
|
||||||
|
end
|
||||||
|
|
||||||
|
# Use most recent 20 years of data as reference period
|
||||||
|
max_year = maximum(party_rows.year)
|
||||||
|
ref_rows = filter(r -> r.year >= max_year - 20, party_rows)
|
||||||
|
if nrow(ref_rows) == 0
|
||||||
|
ref_rows = party_rows
|
||||||
|
end
|
||||||
|
|
||||||
|
n_checked += 1
|
||||||
|
|
||||||
|
mean_econ = mean(ref_rows.economic_lr)
|
||||||
|
mean_galtan = mean(ref_rows.galtan)
|
||||||
|
|
||||||
|
econ_ok = anchor.econ[1] <= mean_econ <= anchor.econ[2]
|
||||||
|
galtan_ok = anchor.galtan[1] <= mean_galtan <= anchor.galtan[2]
|
||||||
|
all_ok = econ_ok && galtan_ok
|
||||||
|
|
||||||
|
if all_ok
|
||||||
|
n_passed += 1
|
||||||
|
end
|
||||||
|
|
||||||
|
status = all_ok ? "PASS" : "WARN"
|
||||||
|
econ_marker = econ_ok ? "" : "*"
|
||||||
|
galtan_marker = galtan_ok ? "" : "*"
|
||||||
|
|
||||||
|
@printf(" %-15s economic=%.2f%s [%.2f-%.2f] cultural=%.2f%s [%.2f-%.2f] %s\n",
|
||||||
|
anchor.name, mean_econ, econ_marker, anchor.econ[1], anchor.econ[2],
|
||||||
|
mean_galtan, galtan_marker, anchor.galtan[1], anchor.galtan[2], status)
|
||||||
|
end
|
||||||
|
|
||||||
|
if n_checked > 0
|
||||||
|
println(" Anchor check: $n_passed/$n_checked within expected ranges")
|
||||||
|
println(" Note: Model integrates text + expert data; deviations from expert-only expectations are normal")
|
||||||
|
end
|
||||||
|
|
||||||
|
# Check 3: Coverage check
|
||||||
|
println("\n3. Coverage check:")
|
||||||
|
println(" Total segment-year positions: $(nrow(output))")
|
||||||
|
println(" Unique parties: $(length(unique(output.party_id)))")
|
||||||
|
blank_country_rows = count(ismissing, output.country)
|
||||||
|
println(" Unique countries: $(length(unique(skipmissing(output.country))))")
|
||||||
|
if blank_country_rows == 0
|
||||||
|
println(" Blank country rows: 0 - PASS")
|
||||||
|
else
|
||||||
|
println(" Blank country rows: $blank_country_rows - FAIL")
|
||||||
|
all_passed = false
|
||||||
|
end
|
||||||
|
println(" Year range: $(minimum(output.year)) - $(maximum(output.year))")
|
||||||
|
|
||||||
|
# V10: Check segment distribution
|
||||||
|
segment_counts = combine(groupby(output, :party_id), nrow => :n_years,
|
||||||
|
:segment_num => (x -> length(unique(x))) => :n_segments)
|
||||||
|
parties_multi_segment = filter(:n_segments => >(1), segment_counts)
|
||||||
|
if nrow(parties_multi_segment) > 0
|
||||||
|
println("\n Parties with multiple segments: $(length(unique(parties_multi_segment.party_id)))")
|
||||||
|
end
|
||||||
|
|
||||||
|
# Check 4: No duplicates (party_id, segment_num, year should be unique)
|
||||||
|
println("\n4. Duplicate check:")
|
||||||
|
dup_count = nrow(output) - nrow(unique(select(output, :party_id, :segment_num, :year)))
|
||||||
|
if dup_count == 0
|
||||||
|
println(" No duplicate (party_id, segment_num, year) combinations - PASS")
|
||||||
|
else
|
||||||
|
println(" WARNING: Found $dup_count duplicate combinations!")
|
||||||
|
all_passed = false
|
||||||
|
end
|
||||||
|
|
||||||
|
# Check 5: SE reasonableness
|
||||||
|
println("\n5. Standard error check:")
|
||||||
|
se_cols = has_4d ?
|
||||||
|
[:pro_market_se, :pro_welfare_se, :cosmopolitan_se, :traditional_se] :
|
||||||
|
[:economic_lr_se, :galtan_se]
|
||||||
|
|
||||||
|
for col in se_cols
|
||||||
|
if !hasproperty(output, col)
|
||||||
|
continue
|
||||||
|
end
|
||||||
|
vals = output[!, col]
|
||||||
|
mean_se = mean(vals)
|
||||||
|
max_se = maximum(vals)
|
||||||
|
println(" $col: mean=$(@sprintf("%.4f", mean_se)), max=$(@sprintf("%.4f", max_se))")
|
||||||
|
end
|
||||||
|
|
||||||
|
println("\n" * "-"^60)
|
||||||
|
if all_passed
|
||||||
|
println("All validation checks PASSED")
|
||||||
|
else
|
||||||
|
println("Some validation checks FAILED - please inspect output carefully")
|
||||||
|
end
|
||||||
|
|
||||||
|
return all_passed
|
||||||
|
end
|
||||||
|
|
||||||
|
# =============================================================================
|
||||||
|
# STEP 6: Save output
|
||||||
|
# =============================================================================
|
||||||
|
|
||||||
|
function save_output(output::DataFrame, metadata::Dict, segment_info::Union{DataFrame, Nothing}, run_dir::String; outdir::String="outputs/estimations/latest")
|
||||||
|
println("\n" * "="^60)
|
||||||
|
println("SAVING OUTPUT")
|
||||||
|
println("="^60)
|
||||||
|
|
||||||
|
timestamp = Dates.format(now(), "yyyy-mm-dd_HH-MM-SS")
|
||||||
|
mkpath(outdir)
|
||||||
|
|
||||||
|
# Delete previous output files
|
||||||
|
for f in readdir(outdir)
|
||||||
|
if startswith(f, "party_positions_") && (endswith(f, ".csv") || endswith(f, ".txt") || endswith(f, ".tex"))
|
||||||
|
rm(joinpath(outdir, f))
|
||||||
|
println(" Deleted old: $f")
|
||||||
|
end
|
||||||
|
end
|
||||||
|
|
||||||
|
# Save main CSV
|
||||||
|
csv_file = joinpath(outdir, "party_positions_$timestamp.csv")
|
||||||
|
CSV.write(csv_file, output)
|
||||||
|
println("Saved: $csv_file")
|
||||||
|
println(" Rows: $(nrow(output))")
|
||||||
|
println(" Columns: $(ncol(output))")
|
||||||
|
|
||||||
|
# Count parties with multiple segments
|
||||||
|
n_multi_segment = 0
|
||||||
|
if segment_info !== nothing
|
||||||
|
party_segment_counts = combine(groupby(segment_info, :party_id), nrow => :n_segments)
|
||||||
|
n_multi_segment = count(r -> r.n_segments > 1, eachrow(party_segment_counts))
|
||||||
|
end
|
||||||
|
|
||||||
|
# Save metadata
|
||||||
|
meta_file = joinpath(outdir, "party_positions_$(timestamp)_metadata.txt")
|
||||||
|
open(meta_file, "w") do f
|
||||||
|
println(f, "Party Positions Dataset - Metadata")
|
||||||
|
println(f, "="^50)
|
||||||
|
println(f, "")
|
||||||
|
println(f, "Generated: $(Dates.format(now(), "yyyy-mm-dd HH:MM:SS"))")
|
||||||
|
println(f, "Source run: $(basename(run_dir))")
|
||||||
|
println(f, "Model file: $(get(metadata, "model_file", "unknown"))")
|
||||||
|
println(f, "")
|
||||||
|
println(f, "Dataset size:")
|
||||||
|
println(f, " Segment-year observations: $(nrow(output))")
|
||||||
|
println(f, " Unique parties: $(length(unique(output.party_id)))")
|
||||||
|
if n_multi_segment > 0
|
||||||
|
println(f, " Parties with multiple segments: $n_multi_segment")
|
||||||
|
end
|
||||||
|
println(f, " Unique countries: $(length(unique(output.country)))")
|
||||||
|
println(f, " Year range: $(minimum(output.year)) - $(maximum(output.year))")
|
||||||
|
println(f, "")
|
||||||
|
println(f, "Columns:")
|
||||||
|
println(f, " party_id: PartyFacts ID (integer) - individual party (e.g., CDU=1375, CSU=1731)")
|
||||||
|
println(f, " union_party_id: PartyFacts ID of parent union (NA for standalone parties)")
|
||||||
|
println(f, " segment_num: Segment number within party (1, 2, 3...)")
|
||||||
|
println(f, " country: ISO2 country code")
|
||||||
|
println(f, " year: Calendar year")
|
||||||
|
println(f, "")
|
||||||
|
println(f, "Segment-Based Indexing:")
|
||||||
|
println(f, " - Parties are split into segments at gaps > 7 years")
|
||||||
|
println(f, " - Each segment is estimated independently (no continuity across gaps)")
|
||||||
|
println(f, " - Segments with < 3 observations are dropped")
|
||||||
|
println(f, " - segment_num=1 is the main segment; higher numbers indicate gaps in data")
|
||||||
|
println(f, "")
|
||||||
|
|
||||||
|
# Check if this is 2D or 4D output
|
||||||
|
is_2d = !hasproperty(output, :pro_market)
|
||||||
|
|
||||||
|
if is_2d
|
||||||
|
println(f, "Model: 2D Direct Bipolar")
|
||||||
|
println(f, "")
|
||||||
|
println(f, "Bipolar scales (0 = left/cosmopolitan, 1 = right/traditionalist):")
|
||||||
|
println(f, " economic_lr: Economic left-right position (directly estimated)")
|
||||||
|
println(f, " galtan: Cultural cosmopolitan--traditionalist position (directly estimated)")
|
||||||
|
# Note: general_lr is computed internally for cross-dimensional anchoring
|
||||||
|
# but not reported as output (the two dimension-specific estimates are preferred)
|
||||||
|
else
|
||||||
|
println(f, "Model: 4D Unipolar")
|
||||||
|
println(f, "")
|
||||||
|
println(f, "Dimensions (0 = low, 1 = high):")
|
||||||
|
println(f, " pro_market: Pro-market economic position")
|
||||||
|
println(f, " pro_welfare: Pro-welfare state position")
|
||||||
|
println(f, " cosmopolitan: Cosmopolitan cultural position")
|
||||||
|
println(f, " traditional: Traditionalist cultural position")
|
||||||
|
println(f, "")
|
||||||
|
println(f, "Derived bipolar scales (0 = left/cosmopolitan, 1 = right/traditionalist):")
|
||||||
|
println(f, " economic_lr: Economic left-right (derived from pro_market - pro_welfare)")
|
||||||
|
println(f, " galtan: Cultural cosmopolitan--traditionalist (derived from traditional - cosmopolitan)")
|
||||||
|
end
|
||||||
|
println(f, "")
|
||||||
|
println(f, "Uncertainty columns:")
|
||||||
|
println(f, " *_se: Standard error (posterior SD)")
|
||||||
|
println(f, " *_q025: 2.5th percentile (lower 95% CI)")
|
||||||
|
println(f, " *_q975: 97.5th percentile (upper 95% CI)")
|
||||||
|
println(f, "")
|
||||||
|
println(f, "Model convergence:")
|
||||||
|
println(f, " Mean R-hat: $(get(metadata, "mean_rhat", "N/A"))")
|
||||||
|
println(f, " Max R-hat: $(get(metadata, "max_rhat", "N/A"))")
|
||||||
|
println(f, " Mean ESS: $(get(metadata, "mean_ess", "N/A"))")
|
||||||
|
println(f, " Min ESS: $(get(metadata, "min_ess", "N/A"))")
|
||||||
|
end
|
||||||
|
println("Saved: $meta_file")
|
||||||
|
|
||||||
|
return csv_file, meta_file
|
||||||
|
end
|
||||||
|
|
||||||
|
# =============================================================================
|
||||||
|
# STEP 5b: Verify no union/alliance IDs in output
|
||||||
|
# =============================================================================
|
||||||
|
|
||||||
|
function verify_no_unions_in_output(output::DataFrame)
|
||||||
|
println("\n" * "="^60)
|
||||||
|
println("UNION ID VERIFICATION")
|
||||||
|
println("="^60)
|
||||||
|
|
||||||
|
union_mapping_file = joinpath("data", "union_mapping.csv")
|
||||||
|
if !isfile(union_mapping_file)
|
||||||
|
println(" No union_mapping.csv found — skipping verification")
|
||||||
|
return
|
||||||
|
end
|
||||||
|
|
||||||
|
union_df = CSV.read(union_mapping_file, DataFrame)
|
||||||
|
union_pf_ids = Set(union_df.manifesto_pf_id)
|
||||||
|
output_pf_ids = Set(output.party_id)
|
||||||
|
|
||||||
|
violations = intersect(union_pf_ids, output_pf_ids)
|
||||||
|
|
||||||
|
if isempty(violations)
|
||||||
|
println(" PASS: No union/alliance PF IDs found in output")
|
||||||
|
println(" Checked $(length(union_pf_ids)) union IDs against $(length(output_pf_ids)) output parties")
|
||||||
|
else
|
||||||
|
println(" WARNING: $(length(violations)) union PF IDs found in output")
|
||||||
|
println(" (This is expected if union_mapping.csv was updated after the model run)")
|
||||||
|
for v in sort(collect(violations))
|
||||||
|
n_rows = count(r -> r.party_id == v, eachrow(output))
|
||||||
|
println(" PF $v: $n_rows rows")
|
||||||
|
end
|
||||||
|
end
|
||||||
|
end
|
||||||
|
|
||||||
|
# =============================================================================
|
||||||
|
# MAIN
|
||||||
|
# =============================================================================
|
||||||
|
|
||||||
|
function main()
|
||||||
|
println("="^60)
|
||||||
|
println("POST-ESTIMATION: Party-position model")
|
||||||
|
println("="^60)
|
||||||
|
println("Started: $(Dates.format(now(), "yyyy-mm-dd HH:MM:SS"))")
|
||||||
|
|
||||||
|
# Step 0: Find run directory (CLI --run-dir or auto-detect latest)
|
||||||
|
run_dir = nothing
|
||||||
|
output_dir = nothing
|
||||||
|
for (i, arg) in enumerate(ARGS)
|
||||||
|
if arg == "--run-dir" && i < length(ARGS)
|
||||||
|
run_dir = ARGS[i + 1]
|
||||||
|
elseif startswith(arg, "--run-dir=")
|
||||||
|
run_dir = split(arg, "=", limit=2)[2]
|
||||||
|
elseif arg == "--output-dir" && i < length(ARGS)
|
||||||
|
output_dir = ARGS[i + 1]
|
||||||
|
elseif startswith(arg, "--output-dir=")
|
||||||
|
output_dir = split(arg, "=", limit=2)[2]
|
||||||
|
end
|
||||||
|
end
|
||||||
|
if run_dir === nothing
|
||||||
|
run_dir = find_latest_run()
|
||||||
|
else
|
||||||
|
println("Using specified run directory: $run_dir")
|
||||||
|
end
|
||||||
|
|
||||||
|
# Step 1: Load run data
|
||||||
|
data = load_run_data(run_dir)
|
||||||
|
|
||||||
|
# Step 2: Build segment-year mapping
|
||||||
|
year0 = data.metadata["year0"]
|
||||||
|
segment_year_map, R, segment_info = build_segment_year_map(
|
||||||
|
data.text_data, data.expert_dim, data.expert_lr,
|
||||||
|
data.segment_info, data.segment_year_map, data.run_dir, year0
|
||||||
|
)
|
||||||
|
|
||||||
|
# Step 3: Load chains
|
||||||
|
chains = load_chains(data.chain_files)
|
||||||
|
|
||||||
|
# Step 4: Extract estimates
|
||||||
|
output = extract_estimates(chains, segment_year_map, R)
|
||||||
|
|
||||||
|
# Step 5: Validate
|
||||||
|
validate_output(output, segment_info)
|
||||||
|
|
||||||
|
# Step 5b: Verify no union/alliance IDs in output
|
||||||
|
verify_no_unions_in_output(output)
|
||||||
|
|
||||||
|
# Step 6: Save output
|
||||||
|
effective_output_dir = output_dir !== nothing ? output_dir : "outputs/estimations/latest"
|
||||||
|
csv_file, meta_file = save_output(output, data.metadata, segment_info, run_dir; outdir=effective_output_dir)
|
||||||
|
|
||||||
|
println("\n" * "="^60)
|
||||||
|
println("COMPLETE")
|
||||||
|
println("="^60)
|
||||||
|
println("Output files:")
|
||||||
|
println(" $csv_file")
|
||||||
|
println(" $meta_file")
|
||||||
|
println("\nFinished: $(Dates.format(now(), "yyyy-mm-dd HH:MM:SS"))")
|
||||||
|
|
||||||
|
return output
|
||||||
|
end
|
||||||
|
|
||||||
|
# Run if executed directly
|
||||||
|
if abspath(PROGRAM_FILE) == @__FILE__
|
||||||
|
main()
|
||||||
|
end
|
||||||
@@ -0,0 +1,332 @@
|
|||||||
|
#!/usr/bin/env julia
|
||||||
|
#############################################################################
|
||||||
|
## 00_validation.jl
|
||||||
|
## Pre-flight validation checks for Stan data and initialization
|
||||||
|
## Prevents cryptic Stan errors by catching issues early
|
||||||
|
#############################################################################
|
||||||
|
|
||||||
|
using Statistics
|
||||||
|
|
||||||
|
"""
|
||||||
|
Validate Stan data dictionary before passing to Stan.
|
||||||
|
Catches common issues that cause Stan to crash with cryptic errors.
|
||||||
|
"""
|
||||||
|
function validate_stan_data(dat::Dict; verbose=true)
|
||||||
|
verbose && println("\n" * "="^70)
|
||||||
|
verbose && println("VALIDATING STAN DATA")
|
||||||
|
verbose && println("="^70)
|
||||||
|
|
||||||
|
errors = String[]
|
||||||
|
warnings = String[]
|
||||||
|
|
||||||
|
# Check for NaN/Inf in all numeric data
|
||||||
|
for (key, value) in dat
|
||||||
|
if isa(value, AbstractArray) && eltype(value) <: Number
|
||||||
|
if any(isnan, value)
|
||||||
|
push!(errors, "Data '$key' contains NaN values")
|
||||||
|
end
|
||||||
|
if any(isinf, value)
|
||||||
|
push!(errors, "Data '$key' contains Inf values")
|
||||||
|
end
|
||||||
|
elseif isa(value, Number)
|
||||||
|
if isnan(value)
|
||||||
|
push!(errors, "Data '$key' is NaN")
|
||||||
|
end
|
||||||
|
if isinf(value)
|
||||||
|
push!(errors, "Data '$key' is Inf")
|
||||||
|
end
|
||||||
|
end
|
||||||
|
end
|
||||||
|
|
||||||
|
# Validate expert data is in open interval (0, 1)
|
||||||
|
if haskey(dat, "val_dim")
|
||||||
|
val_dim = dat["val_dim"]
|
||||||
|
if any(x -> x <= 0 || x >= 1, val_dim)
|
||||||
|
n_boundary = count(x -> x <= 0 || x >= 1, val_dim)
|
||||||
|
push!(errors, "Expert dimension data has $n_boundary values at boundaries (must be in (0,1))")
|
||||||
|
if verbose
|
||||||
|
println(" Dimension-specific expert data range: [$(minimum(val_dim)), $(maximum(val_dim))]")
|
||||||
|
end
|
||||||
|
end
|
||||||
|
end
|
||||||
|
|
||||||
|
if haskey(dat, "val_lr")
|
||||||
|
val_lr = dat["val_lr"]
|
||||||
|
if any(x -> x <= 0 || x >= 1, val_lr)
|
||||||
|
n_boundary = count(x -> x <= 0 || x >= 1, val_lr)
|
||||||
|
push!(errors, "Expert L-R data has $n_boundary values at boundaries (must be in (0,1))")
|
||||||
|
if verbose
|
||||||
|
println(" L-R expert data range: [$(minimum(val_lr)), $(maximum(val_lr))]")
|
||||||
|
end
|
||||||
|
end
|
||||||
|
end
|
||||||
|
|
||||||
|
# Validate manifesto data
|
||||||
|
if haskey(dat, "positive") && haskey(dat, "sample")
|
||||||
|
positive = dat["positive"]
|
||||||
|
sample = dat["sample"]
|
||||||
|
|
||||||
|
if any(positive .> sample)
|
||||||
|
push!(errors, "Manifesto: positive counts exceed sample sizes")
|
||||||
|
end
|
||||||
|
|
||||||
|
if any(positive .< 0)
|
||||||
|
push!(errors, "Manifesto: negative positive counts found")
|
||||||
|
end
|
||||||
|
|
||||||
|
if any(sample .< 0)
|
||||||
|
push!(errors, "Manifesto: negative sample sizes found")
|
||||||
|
end
|
||||||
|
end
|
||||||
|
|
||||||
|
# Validate indices are within bounds
|
||||||
|
# V10: Check segment indices (ss_man) if present, otherwise party indices (jj_man)
|
||||||
|
if haskey(dat, "S") && haskey(dat, "ss_man")
|
||||||
|
S = dat["S"]
|
||||||
|
ss_man = dat["ss_man"]
|
||||||
|
if any(ss_man .< 1) || any(ss_man .> S)
|
||||||
|
push!(errors, "Manifesto segment indices out of bounds [1, $S]")
|
||||||
|
end
|
||||||
|
elseif haskey(dat, "J") && haskey(dat, "jj_man")
|
||||||
|
J = dat["J"]
|
||||||
|
jj_man = dat["jj_man"]
|
||||||
|
if any(jj_man .< 1) || any(jj_man .> J)
|
||||||
|
push!(errors, "Manifesto party indices out of bounds [1, $J]")
|
||||||
|
end
|
||||||
|
end
|
||||||
|
|
||||||
|
if haskey(dat, "R") && haskey(dat, "rr_man")
|
||||||
|
R = dat["R"]
|
||||||
|
rr_man = dat["rr_man"]
|
||||||
|
if any(rr_man .< 1) || any(rr_man .> R)
|
||||||
|
push!(errors, "Manifesto party-year indices out of bounds [1, $R]")
|
||||||
|
end
|
||||||
|
end
|
||||||
|
|
||||||
|
# V4: Validate constituent arrays
|
||||||
|
if haskey(dat, "const_rr_man") && haskey(dat, "R")
|
||||||
|
R = dat["R"]
|
||||||
|
const_rr = dat["const_rr_man"]
|
||||||
|
if any(const_rr .< 1) || any(const_rr .> R)
|
||||||
|
push!(errors, "const_rr_man out of bounds [1, $R]")
|
||||||
|
end
|
||||||
|
# Verify offsets are valid
|
||||||
|
if haskey(dat, "const_offset_man") && haskey(dat, "n_const_man")
|
||||||
|
offsets = dat["const_offset_man"]
|
||||||
|
n_consts = dat["n_const_man"]
|
||||||
|
total = dat["N_const_man_total"]
|
||||||
|
for i in eachindex(offsets)
|
||||||
|
if offsets[i] + n_consts[i] - 1 > total
|
||||||
|
push!(errors, "const_offset_man[$i] + n_const_man[$i] exceeds N_const_man_total")
|
||||||
|
break
|
||||||
|
end
|
||||||
|
end
|
||||||
|
end
|
||||||
|
end
|
||||||
|
|
||||||
|
# Print summary
|
||||||
|
if verbose
|
||||||
|
println("\nDATA SUMMARY:")
|
||||||
|
# V10: Show segments if present
|
||||||
|
if haskey(dat, "S")
|
||||||
|
println(" Segments (S): $(dat["S"])")
|
||||||
|
println(" Parties with segments (J): $(get(dat, "J", "N/A"))")
|
||||||
|
else
|
||||||
|
println(" Parties (J): $(get(dat, "J", "N/A"))")
|
||||||
|
end
|
||||||
|
println(" Countries (P): $(get(dat, "P", "N/A"))")
|
||||||
|
println(" Segment-years (R): $(get(dat, "R", "N/A"))")
|
||||||
|
println(" Years (T_year): $(get(dat, "T_year", "N/A"))")
|
||||||
|
println(" Manifesto obs: $(get(dat, "N_man", "N/A"))")
|
||||||
|
println(" Expert dim obs: $(get(dat, "N_exp_dim", "N/A"))")
|
||||||
|
println(" Expert L-R obs: $(get(dat, "N_exp_lr", "N/A"))")
|
||||||
|
|
||||||
|
if haskey(dat, "mn_resp_log_man")
|
||||||
|
println("\nPRIOR MEANS:")
|
||||||
|
println(" Manifesto: $(round(dat["mn_resp_log_man"], digits=3))")
|
||||||
|
println(" Expert dim: $(round(dat["mn_resp_log_exp_dim"], digits=3))")
|
||||||
|
println(" Expert L-R: $(round(dat["mn_resp_log_exp_lr"], digits=3))")
|
||||||
|
end
|
||||||
|
end
|
||||||
|
|
||||||
|
# Report results
|
||||||
|
if !isempty(errors)
|
||||||
|
println("\n❌ VALIDATION FAILED - $(length(errors)) ERROR(S):")
|
||||||
|
for (i, err) in enumerate(errors)
|
||||||
|
println(" $i. $err")
|
||||||
|
end
|
||||||
|
return false
|
||||||
|
end
|
||||||
|
|
||||||
|
if !isempty(warnings)
|
||||||
|
println("\n⚠️ $(length(warnings)) WARNING(S):")
|
||||||
|
for (i, warn) in enumerate(warnings)
|
||||||
|
println(" $i. $warn")
|
||||||
|
end
|
||||||
|
end
|
||||||
|
|
||||||
|
if verbose
|
||||||
|
println("\n✓ DATA VALIDATION PASSED")
|
||||||
|
println("="^70)
|
||||||
|
end
|
||||||
|
|
||||||
|
return true
|
||||||
|
end
|
||||||
|
|
||||||
|
"""
|
||||||
|
Validate initialization values before passing to Stan.
|
||||||
|
Checks for common issues that cause immediate Stan crashes.
|
||||||
|
"""
|
||||||
|
function validate_init_values(init_dict::Dict; verbose=true)
|
||||||
|
verbose && println("\n" * "="^70)
|
||||||
|
verbose && println("VALIDATING INITIALIZATION VALUES")
|
||||||
|
verbose && println("="^70)
|
||||||
|
|
||||||
|
errors = String[]
|
||||||
|
warnings = String[]
|
||||||
|
|
||||||
|
for (key, value) in init_dict
|
||||||
|
# Check for NaN/Inf
|
||||||
|
if isa(value, AbstractArray) && eltype(value) <: Number
|
||||||
|
if any(isnan, value)
|
||||||
|
push!(errors, "Init '$key' contains NaN")
|
||||||
|
end
|
||||||
|
if any(isinf, value)
|
||||||
|
push!(errors, "Init '$key' contains Inf")
|
||||||
|
end
|
||||||
|
|
||||||
|
if verbose && length(value) > 0
|
||||||
|
val_array = vec(value)
|
||||||
|
println(" $key: range [$(round(minimum(val_array), digits=3)), $(round(maximum(val_array), digits=3))]")
|
||||||
|
end
|
||||||
|
elseif isa(value, Number)
|
||||||
|
if isnan(value)
|
||||||
|
push!(errors, "Init '$key' is NaN")
|
||||||
|
end
|
||||||
|
if isinf(value)
|
||||||
|
push!(errors, "Init '$key' is Inf")
|
||||||
|
end
|
||||||
|
|
||||||
|
if verbose
|
||||||
|
println(" $key: $(round(value, digits=3))")
|
||||||
|
end
|
||||||
|
end
|
||||||
|
|
||||||
|
# Check positive constraints (common Stan constraints)
|
||||||
|
# Exception: *_raw parameters are non-centered and can be any real
|
||||||
|
if (contains(string(key), "sigma") || contains(string(key), "tau") || contains(string(key), "phi")) &&
|
||||||
|
!endswith(string(key), "_raw")
|
||||||
|
if isa(value, Number) && value <= 0
|
||||||
|
push!(errors, "Init '$key' = $value violates constraint > 0")
|
||||||
|
elseif isa(value, AbstractArray) && any(value .<= 0)
|
||||||
|
push!(errors, "Init '$key' has values ≤ 0 (violates constraint > 0)")
|
||||||
|
end
|
||||||
|
end
|
||||||
|
|
||||||
|
# Check Cholesky factors are valid
|
||||||
|
if contains(string(key), "L_Omega")
|
||||||
|
if isa(value, AbstractMatrix)
|
||||||
|
# Check it's lower triangular with positive diagonal
|
||||||
|
n = size(value, 1)
|
||||||
|
if size(value, 2) != n
|
||||||
|
push!(errors, "Init '$key' is not square")
|
||||||
|
end
|
||||||
|
for i in 1:n
|
||||||
|
if value[i, i] <= 0
|
||||||
|
push!(errors, "Init '$key' has non-positive diagonal at position $i")
|
||||||
|
end
|
||||||
|
for j in (i+1):n
|
||||||
|
if abs(value[i, j]) > 1e-10
|
||||||
|
push!(warnings, "Init '$key' is not lower triangular")
|
||||||
|
break
|
||||||
|
end
|
||||||
|
end
|
||||||
|
end
|
||||||
|
end
|
||||||
|
end
|
||||||
|
|
||||||
|
# Check slope parameters for positive constraint (V2/V3 feature)
|
||||||
|
if key == "Gamma_man_slope_raw"
|
||||||
|
if isa(value, AbstractArray) && any(value .< 0)
|
||||||
|
push!(errors, "Init 'Gamma_man_slope_raw' has negative values (must be ≥ 0)")
|
||||||
|
end
|
||||||
|
end
|
||||||
|
end
|
||||||
|
|
||||||
|
# Report results
|
||||||
|
if !isempty(errors)
|
||||||
|
println("\n❌ INIT VALIDATION FAILED - $(length(errors)) ERROR(S):")
|
||||||
|
for (i, err) in enumerate(errors)
|
||||||
|
println(" $i. $err")
|
||||||
|
end
|
||||||
|
return false
|
||||||
|
end
|
||||||
|
|
||||||
|
if !isempty(warnings)
|
||||||
|
println("\n⚠️ $(length(warnings)) WARNING(S):")
|
||||||
|
for (i, warn) in enumerate(warnings)
|
||||||
|
println(" $i. $warn")
|
||||||
|
end
|
||||||
|
end
|
||||||
|
|
||||||
|
if verbose
|
||||||
|
println("\n✓ INIT VALIDATION PASSED")
|
||||||
|
println("="^70)
|
||||||
|
end
|
||||||
|
|
||||||
|
return true
|
||||||
|
end
|
||||||
|
|
||||||
|
"""
|
||||||
|
Estimate memory requirements for model
|
||||||
|
"""
|
||||||
|
function estimate_memory_requirements(dat::Dict; verbose=true, num_chains::Int=4, num_samples::Int=1000, num_threads_per_chain::Int=1)
|
||||||
|
if !verbose
|
||||||
|
return
|
||||||
|
end
|
||||||
|
|
||||||
|
println("\n" * "="^70)
|
||||||
|
println("MEMORY ESTIMATE")
|
||||||
|
println("="^70)
|
||||||
|
|
||||||
|
R = get(dat, "R", 0)
|
||||||
|
J = get(dat, "J", 0)
|
||||||
|
K_man = get(dat, "K_man", 0)
|
||||||
|
K_exp_dim = get(dat, "K_exp_dim", 0)
|
||||||
|
K_exp_lr = get(dat, "K_exp_lr", 0)
|
||||||
|
N_man = get(dat, "N_man", 0)
|
||||||
|
N_ciy = get(dat, "N_ciy", 0)
|
||||||
|
T_year = get(dat, "T_year", 0)
|
||||||
|
|
||||||
|
# Rough parameter count
|
||||||
|
theta_params = 4 * R
|
||||||
|
item_params = 4 * K_man * 2 + K_exp_dim * 3 + K_exp_lr * 2
|
||||||
|
strategic_params = get(dat, "P", 0) * get(dat, "K_man", 0) # Country-item intercepts
|
||||||
|
other_params = 4 * J + T_year + J + N_ciy + 50
|
||||||
|
|
||||||
|
total_params = theta_params + item_params + strategic_params + other_params
|
||||||
|
|
||||||
|
# Memory estimate (very rough)
|
||||||
|
# Each parameter: ~8 bytes (float64) × samples × chains
|
||||||
|
total_draws_per_param = num_samples * num_chains
|
||||||
|
bytes_per_param = 8 * total_draws_per_param
|
||||||
|
total_mb = (total_params * bytes_per_param) / (1024 * 1024)
|
||||||
|
|
||||||
|
println(" Configuration:")
|
||||||
|
println(" Chains: $num_chains")
|
||||||
|
println(" Samples per chain: $num_samples")
|
||||||
|
println(" Threads per chain: $num_threads_per_chain")
|
||||||
|
println(" Total parallel workers: $(num_chains * num_threads_per_chain)")
|
||||||
|
|
||||||
|
println(" Total parameters: ~$(total_params)")
|
||||||
|
println(" Estimated memory (samples only): ~$(round(total_mb, digits=0)) MB")
|
||||||
|
thread_scaling = max(1, num_threads_per_chain)
|
||||||
|
println(" With thread overhead (×$(thread_scaling)): ~$(round(total_mb * thread_scaling, digits=0)) MB")
|
||||||
|
println(" With safety margin (×3): ~$(round(3 * total_mb * thread_scaling, digits=0)) MB")
|
||||||
|
|
||||||
|
if total_mb * thread_scaling * 3 > 8000
|
||||||
|
println("\n⚠️ WARNING: Model may require > 8GB RAM")
|
||||||
|
end
|
||||||
|
|
||||||
|
println("="^70)
|
||||||
|
end
|
||||||
@@ -0,0 +1,277 @@
|
|||||||
|
#!/usr/bin/env julia
|
||||||
|
#############################################################################
|
||||||
|
## 02_data_loading.jl
|
||||||
|
## Load and preprocess data for latent trait model
|
||||||
|
## Loads three datasets: text_data (manifesto + PolDem), expert dimension-specific, expert general L-R
|
||||||
|
##
|
||||||
|
## Supports both:
|
||||||
|
## - 4D model (V10): type_high/type_low columns for bipolar bridges
|
||||||
|
## - 2D model (V1): dim_idx + direction columns for direct estimation
|
||||||
|
#############################################################################
|
||||||
|
|
||||||
|
using DataFrames, CSV, CategoricalArrays, Statistics
|
||||||
|
|
||||||
|
#############################################################################
|
||||||
|
## UNION MAPPING: Individual party estimates via mean-constituent model
|
||||||
|
## Loads data/union_mapping.csv and builds lookup structures
|
||||||
|
#############################################################################
|
||||||
|
|
||||||
|
"""
|
||||||
|
load_union_mapping(project_root::String)
|
||||||
|
|
||||||
|
Load union_mapping.csv and build lookup dictionaries.
|
||||||
|
Returns (union_to_constituents, constituent_to_union) dicts.
|
||||||
|
If file is missing or empty, returns empty dicts (backwards compatible).
|
||||||
|
"""
|
||||||
|
function load_union_mapping(project_root::String=".")
|
||||||
|
mapping_file = joinpath(project_root, "data", "union_mapping.csv")
|
||||||
|
|
||||||
|
union_to_constituents = Dict{Int, Vector{Int}}()
|
||||||
|
constituent_to_union = Dict{Int, Int}()
|
||||||
|
|
||||||
|
if !isfile(mapping_file)
|
||||||
|
println(" No union_mapping.csv found - running without union decomposition")
|
||||||
|
return union_to_constituents, constituent_to_union
|
||||||
|
end
|
||||||
|
|
||||||
|
df = CSV.read(mapping_file, DataFrame)
|
||||||
|
if nrow(df) == 0
|
||||||
|
println(" union_mapping.csv is empty - running without union decomposition")
|
||||||
|
return union_to_constituents, constituent_to_union
|
||||||
|
end
|
||||||
|
|
||||||
|
for row in eachrow(df)
|
||||||
|
union_id = row.manifesto_pf_id
|
||||||
|
expert_id = row.expert_pf_id
|
||||||
|
|
||||||
|
if !haskey(union_to_constituents, union_id)
|
||||||
|
union_to_constituents[union_id] = Int[]
|
||||||
|
end
|
||||||
|
if !(expert_id in union_to_constituents[union_id])
|
||||||
|
push!(union_to_constituents[union_id], expert_id)
|
||||||
|
end
|
||||||
|
constituent_to_union[expert_id] = union_id
|
||||||
|
end
|
||||||
|
|
||||||
|
println(" Union mapping loaded: $(length(union_to_constituents)) unions, $(length(constituent_to_union)) constituents")
|
||||||
|
return union_to_constituents, constituent_to_union
|
||||||
|
end
|
||||||
|
|
||||||
|
#############################################################################
|
||||||
|
## SEGMENT-BASED INDEXING CONFIGURATION
|
||||||
|
## Split parties at gaps > MAX_GAP years to avoid flat posteriors
|
||||||
|
#############################################################################
|
||||||
|
const MAX_GAP = 7 # Maximum years between observations within a segment
|
||||||
|
const MIN_OBS = 2 # Minimum observations per segment (drop segments with fewer)
|
||||||
|
|
||||||
|
#############################################################################
|
||||||
|
## 2D MODEL MAPPING CONFIGURATION
|
||||||
|
## Maps type_high/type_low pairs to dim_idx + direction
|
||||||
|
#############################################################################
|
||||||
|
const TYPE_TO_DIM_DIRECTION = Dict(
|
||||||
|
# Economic dimension: pro_market = right (+1), pro_welfare = left (-1)
|
||||||
|
("pro_market", "pro_welfare") => (dim_idx=1, direction=1), # Right
|
||||||
|
("pro_welfare", "pro_market") => (dim_idx=1, direction=-1), # Left
|
||||||
|
|
||||||
|
# Cultural dimension: traditional = TAN (+1), cosmopolitan = GAL (-1)
|
||||||
|
("traditional", "cosmopolitan") => (dim_idx=2, direction=1), # TAN
|
||||||
|
("cosmopolitan", "traditional") => (dim_idx=2, direction=-1) # GAL
|
||||||
|
)
|
||||||
|
|
||||||
|
# Expert dimension mapping (lrecon -> economic, galtan/cultural -> galtan)
|
||||||
|
const EXPERT_VAR_TO_DIM = Dict(
|
||||||
|
"lrecon_ches" => 1,
|
||||||
|
"lrecon_vparty" => 1,
|
||||||
|
"welf_vparty" => 1,
|
||||||
|
"lrecon_gps" => 1,
|
||||||
|
"lrecon_poppa" => 1,
|
||||||
|
"galtan_ches" => 2,
|
||||||
|
"libcon_gps" => 2,
|
||||||
|
"immig_vparty" => 2,
|
||||||
|
"lgbt_vparty" => 2,
|
||||||
|
"culsup_vparty" => 2,
|
||||||
|
"relig_vparty" => 2,
|
||||||
|
"gender_vparty" => 2
|
||||||
|
)
|
||||||
|
|
||||||
|
function load_and_preprocess_4dim_data(start_year=1950; data_dir::String=".")
|
||||||
|
println("Loading party-position data files...")
|
||||||
|
println("Start year filter: $start_year")
|
||||||
|
data_dir != "." && println("Data directory: $data_dir")
|
||||||
|
|
||||||
|
# Load union mapping (check data_dir first, fall back to project root)
|
||||||
|
println("\nLoading union mapping...")
|
||||||
|
union_mapping_dir = isfile(joinpath(data_dir, "data", "union_mapping.csv")) ? data_dir : "."
|
||||||
|
union_to_constituents, constituent_to_union = load_union_mapping(union_mapping_dir)
|
||||||
|
|
||||||
|
# Load the three datasets
|
||||||
|
text_data_raw = CSV.read(joinpath(data_dir, "text_data.csv"), DataFrame)
|
||||||
|
expert_raw = CSV.read(joinpath(data_dir, "expert.csv"), DataFrame)
|
||||||
|
lr_data_raw = CSV.read(joinpath(data_dir, "lr_data.csv"), DataFrame)
|
||||||
|
|
||||||
|
# Filter to start year BEFORE calculating year0
|
||||||
|
text_data_raw = text_data_raw[text_data_raw.year .>= start_year, :]
|
||||||
|
expert_raw = expert_raw[expert_raw.year .>= start_year, :]
|
||||||
|
lr_data_raw = lr_data_raw[lr_data_raw.year .>= start_year, :]
|
||||||
|
|
||||||
|
println("Data filtered to $start_year onwards:")
|
||||||
|
println(" Text data (manifesto + PolDem): $(nrow(text_data_raw)) observations")
|
||||||
|
println(" Expert: $(nrow(expert_raw)) observations")
|
||||||
|
println(" L-R data: $(nrow(lr_data_raw)) observations")
|
||||||
|
|
||||||
|
# Define base year for relative time indexing
|
||||||
|
year0 = Int(minimum([minimum(text_data_raw.year), minimum(expert_raw.year), minimum(lr_data_raw.year)])) - 1
|
||||||
|
println("Base year set to: $year0")
|
||||||
|
|
||||||
|
# Create type mapping for the four dimensions
|
||||||
|
type_map = Dict(
|
||||||
|
"pro_market" => 1,
|
||||||
|
"pro_welfare" => 2,
|
||||||
|
"cosmopolitan" => 3,
|
||||||
|
"traditional" => 4
|
||||||
|
)
|
||||||
|
|
||||||
|
println("Type mapping: pro_market=1, pro_welfare=2, cosmopolitan=3, traditional=4")
|
||||||
|
|
||||||
|
# Process text data (manifesto + PolDem media)
|
||||||
|
text_data = copy(text_data_raw)
|
||||||
|
text_data = text_data[text_data.year .> year0, :]
|
||||||
|
|
||||||
|
# Add type indices for text items (V4/V10: bipolar bridge structure)
|
||||||
|
if !("type_high" in names(text_data)) || !("type_low" in names(text_data))
|
||||||
|
error("Text data must contain 'type_high' and 'type_low' columns with values: pro_market, pro_welfare, cosmopolitan, traditional")
|
||||||
|
end
|
||||||
|
|
||||||
|
text_data.type_high_idx = [type_map[t] for t in text_data.type_high]
|
||||||
|
text_data.type_low_idx = [type_map[t] for t in text_data.type_low]
|
||||||
|
|
||||||
|
# V1 (2D model): Add dim_idx and direction columns
|
||||||
|
# Maps type_high/type_low to single dimension + direction
|
||||||
|
dim_idx_man = Int[]
|
||||||
|
direction_man = Int[]
|
||||||
|
for row in eachrow(text_data)
|
||||||
|
key = (row.type_high, row.type_low)
|
||||||
|
if haskey(TYPE_TO_DIM_DIRECTION, key)
|
||||||
|
mapping = TYPE_TO_DIM_DIRECTION[key]
|
||||||
|
push!(dim_idx_man, mapping.dim_idx)
|
||||||
|
push!(direction_man, mapping.direction)
|
||||||
|
else
|
||||||
|
# Unknown mapping - this should not happen with valid data
|
||||||
|
error("Unknown type_high/type_low pair: $(row.type_high) / $(row.type_low)")
|
||||||
|
end
|
||||||
|
end
|
||||||
|
text_data.dim_idx_man = dim_idx_man
|
||||||
|
text_data.direction_man = direction_man
|
||||||
|
|
||||||
|
# Standard processing
|
||||||
|
text_data.country = categorical(text_data.country)
|
||||||
|
text_data.party = categorical(text_data.party)
|
||||||
|
text_data.var = categorical(text_data.var)
|
||||||
|
text_data.Year = Int.(text_data.year) .- year0
|
||||||
|
sort!(text_data, [:country, :party, :year, :var])
|
||||||
|
println("Text data processed: $(nrow(text_data)) observations with bipolar bridge structure")
|
||||||
|
|
||||||
|
# Process expert dimension-specific data (bipolar items like lrecon_ches, galtan_ches)
|
||||||
|
expert_dim_vars = ["lrecon_ches", "galtan_ches", "lrecon_vparty", "welf_vparty",
|
||||||
|
"lrecon_gps", "libcon_gps", "lrecon_poppa",
|
||||||
|
"immig_vparty", "lgbt_vparty", "culsup_vparty", "relig_vparty", "gender_vparty"]
|
||||||
|
expert_dim = expert_raw[in.(expert_raw.var, Ref(expert_dim_vars)), :]
|
||||||
|
expert_dim = expert_dim[(expert_dim.year .> year0) .& (expert_dim.val .>= 0) .& (expert_dim.val .<= 1), :]
|
||||||
|
|
||||||
|
# V5: Load integer observations, scale sizes, and expert counts for beta-binomial likelihood
|
||||||
|
expert_dim.val_int = Int.(expert_dim.val_int)
|
||||||
|
expert_dim.n_scale = Int.(expert_dim.n_scale)
|
||||||
|
expert_dim.n_experts = Int.(expert_dim.n_experts)
|
||||||
|
|
||||||
|
# Add type mappings for dimension-specific expert data
|
||||||
|
if !("type_low" in names(expert_dim)) || !("type_high" in names(expert_dim))
|
||||||
|
error("Expert data must contain 'type_low' and 'type_high' columns")
|
||||||
|
end
|
||||||
|
|
||||||
|
expert_dim.type_high_idx = [type_map[t] for t in expert_dim.type_high]
|
||||||
|
expert_dim.type_low_idx = [type_map[t] for t in expert_dim.type_low]
|
||||||
|
|
||||||
|
# V1 (2D model): Add dim_idx for expert dimension data
|
||||||
|
dim_idx_exp = Int[]
|
||||||
|
for row in eachrow(expert_dim)
|
||||||
|
var_name = string(row.var)
|
||||||
|
if haskey(EXPERT_VAR_TO_DIM, var_name)
|
||||||
|
push!(dim_idx_exp, EXPERT_VAR_TO_DIM[var_name])
|
||||||
|
else
|
||||||
|
# Fallback: infer from type_high/type_low
|
||||||
|
key = (row.type_high, row.type_low)
|
||||||
|
if haskey(TYPE_TO_DIM_DIRECTION, key)
|
||||||
|
push!(dim_idx_exp, TYPE_TO_DIM_DIRECTION[key].dim_idx)
|
||||||
|
else
|
||||||
|
error("Unknown expert variable: $var_name with type pair $(row.type_high) / $(row.type_low)")
|
||||||
|
end
|
||||||
|
end
|
||||||
|
end
|
||||||
|
expert_dim.dim_idx_exp = dim_idx_exp
|
||||||
|
|
||||||
|
expert_dim.country = categorical(expert_dim.country)
|
||||||
|
expert_dim.party = categorical(expert_dim.party)
|
||||||
|
expert_dim.var = categorical(expert_dim.var)
|
||||||
|
expert_dim.Year = Int.(expert_dim.year) .- year0
|
||||||
|
sort!(expert_dim, [:country, :party, :year, :var])
|
||||||
|
println("Expert dimension-specific data processed: $(nrow(expert_dim)) observations")
|
||||||
|
|
||||||
|
# Process expert general L-R data (cross-dimensional anchoring)
|
||||||
|
lr_vars = ["lr_ches", "lr_poppa", "lr_morgan"] # General left-right items
|
||||||
|
expert_lr = lr_data_raw[in.(lr_data_raw.var, Ref(lr_vars)), :]
|
||||||
|
expert_lr = expert_lr[(expert_lr.year .> year0) .& (expert_lr.val .>= 0) .& (expert_lr.val .<= 1), :]
|
||||||
|
|
||||||
|
# V5: Load integer observations, scale sizes, and expert counts for beta-binomial likelihood
|
||||||
|
expert_lr.val_int = Int.(expert_lr.val_int)
|
||||||
|
expert_lr.n_scale = Int.(expert_lr.n_scale)
|
||||||
|
expert_lr.n_experts = Int.(expert_lr.n_experts)
|
||||||
|
|
||||||
|
expert_lr.country = categorical(expert_lr.country)
|
||||||
|
expert_lr.party = categorical(expert_lr.party)
|
||||||
|
expert_lr.var = categorical(expert_lr.var)
|
||||||
|
expert_lr.Year = Int.(expert_lr.year) .- year0
|
||||||
|
sort!(expert_lr, [:country, :party, :year, :var])
|
||||||
|
println("Expert general L-R data processed: $(nrow(expert_lr)) observations")
|
||||||
|
|
||||||
|
# Validate data integrity
|
||||||
|
println("\nData validation:")
|
||||||
|
|
||||||
|
# Check text data dimension pair distribution (V4: bipolar bridges)
|
||||||
|
type_pair_counts = combine(groupby(text_data, [:type_high, :type_low]), nrow => :count)
|
||||||
|
for row in eachrow(type_pair_counts)
|
||||||
|
println(" $(row.type_high) ↔ $(row.type_low): $(row.count) text data observations")
|
||||||
|
end
|
||||||
|
|
||||||
|
# Check expert dimension-specific type pairs
|
||||||
|
type_pair_counts = combine(groupby(expert_dim, [:type_high, :type_low]), nrow => :count)
|
||||||
|
for row in eachrow(type_pair_counts)
|
||||||
|
println(" $(row.type_high) - $(row.type_low): $(row.count) expert dimension-specific observations")
|
||||||
|
end
|
||||||
|
|
||||||
|
# Check general L-R items
|
||||||
|
lr_var_counts = combine(groupby(expert_lr, :var), nrow => :count)
|
||||||
|
for row in eachrow(lr_var_counts)
|
||||||
|
println(" $(row.var): $(row.count) general L-R observations")
|
||||||
|
end
|
||||||
|
|
||||||
|
# Check overlapping parties across datasets
|
||||||
|
text_data_parties = Set(text_data.party)
|
||||||
|
expert_dim_parties = Set(expert_dim.party)
|
||||||
|
expert_lr_parties = Set(expert_lr.party)
|
||||||
|
|
||||||
|
all_parties = union(text_data_parties, expert_dim_parties, expert_lr_parties)
|
||||||
|
println("\nParty coverage:")
|
||||||
|
println(" Total unique parties: $(length(all_parties))")
|
||||||
|
println(" In text data: $(length(text_data_parties))")
|
||||||
|
println(" In expert dimension-specific: $(length(expert_dim_parties))")
|
||||||
|
println(" In expert general L-R: $(length(expert_lr_parties))")
|
||||||
|
println(" In all three datasets: $(length(intersect(text_data_parties, expert_dim_parties, expert_lr_parties)))")
|
||||||
|
|
||||||
|
return text_data, expert_dim, expert_lr, year0, union_to_constituents, constituent_to_union
|
||||||
|
end
|
||||||
|
|
||||||
|
# Execute if run directly
|
||||||
|
if abspath(PROGRAM_FILE) == @__FILE__
|
||||||
|
text_data, expert_dim, expert_lr, year0, u2c, c2u = load_and_preprocess_4dim_data()
|
||||||
|
println("4D data loading test completed successfully")
|
||||||
|
end
|
||||||
@@ -0,0 +1,947 @@
|
|||||||
|
#!/usr/bin/env julia
|
||||||
|
#############################################################################
|
||||||
|
## 03_data_preparation_4dim.jl
|
||||||
|
## V10: Segment-based indexing to fix long gap interpolation issues
|
||||||
|
##
|
||||||
|
## Key change: Split parties at gaps > MAX_GAP years into independent segments.
|
||||||
|
## Each segment has its own random walk (restarts at segment start).
|
||||||
|
## Segments with < MIN_OBS observations are dropped.
|
||||||
|
#############################################################################
|
||||||
|
|
||||||
|
using DataFrames, CSV, CategoricalArrays, Statistics, StatsFuns
|
||||||
|
|
||||||
|
# Import configuration from data loading module
|
||||||
|
include("02_data_loading.jl")
|
||||||
|
|
||||||
|
"""
|
||||||
|
split_party_years_into_segments(years::Vector{Int}, max_gap::Int)
|
||||||
|
|
||||||
|
Split a party's observation years into segments based on gaps.
|
||||||
|
Returns a vector of vectors, where each inner vector contains consecutive years
|
||||||
|
with max `max_gap` years between observations.
|
||||||
|
"""
|
||||||
|
function split_party_years_into_segments(years::Vector{Int}, max_gap::Int)
|
||||||
|
if isempty(years)
|
||||||
|
return Vector{Vector{Int}}()
|
||||||
|
end
|
||||||
|
|
||||||
|
sorted_years = sort(unique(years))
|
||||||
|
segments = [Int[sorted_years[1]]]
|
||||||
|
|
||||||
|
for y in sorted_years[2:end]
|
||||||
|
if y - segments[end][end] <= max_gap
|
||||||
|
push!(segments[end], y)
|
||||||
|
else
|
||||||
|
# Gap too large - start new segment
|
||||||
|
push!(segments, [y])
|
||||||
|
end
|
||||||
|
end
|
||||||
|
|
||||||
|
return segments
|
||||||
|
end
|
||||||
|
|
||||||
|
function prepare_4dim_stan_data(manifesto, expert_dim, expert_lr, year0;
|
||||||
|
union_to_constituents=Dict{Int,Vector{Int}}(),
|
||||||
|
constituent_to_union=Dict{Int,Int}())
|
||||||
|
println("Preparing data for Stan model (Segment-based indexing)...")
|
||||||
|
println(" MAX_GAP = $MAX_GAP years, MIN_OBS = $MIN_OBS observations")
|
||||||
|
|
||||||
|
has_unions = !isempty(union_to_constituents)
|
||||||
|
if has_unions
|
||||||
|
println(" Union mapping: $(length(union_to_constituents)) unions, $(length(constituent_to_union)) constituents")
|
||||||
|
else
|
||||||
|
println(" No union mapping - standard party indexing")
|
||||||
|
end
|
||||||
|
|
||||||
|
# =========================================================================
|
||||||
|
# STEP 1: Collect observation years per party (union-aware)
|
||||||
|
# For union parties: create segments for each CONSTITUENT, not the union.
|
||||||
|
# Union manifesto years are assigned to all constituents.
|
||||||
|
# =========================================================================
|
||||||
|
|
||||||
|
# Identify which party IDs in data are unions vs standalone
|
||||||
|
# NOTE: levels() returns raw types (Int64 for integer party IDs).
|
||||||
|
# We consistently use String keys for all party lookups to avoid type mismatches.
|
||||||
|
manifesto_party_strs = Set(string.(levels(manifesto.party)))
|
||||||
|
union_ids_in_data = Set{String}()
|
||||||
|
if has_unions
|
||||||
|
for uid in keys(union_to_constituents)
|
||||||
|
uid_str = string(uid)
|
||||||
|
if uid_str in manifesto_party_strs
|
||||||
|
push!(union_ids_in_data, uid_str)
|
||||||
|
end
|
||||||
|
end
|
||||||
|
println(" Union party IDs found in manifesto data: $(length(union_ids_in_data))")
|
||||||
|
end
|
||||||
|
|
||||||
|
# Collect all party IDs that need segments
|
||||||
|
# For unions: constituents get segments; union itself does NOT
|
||||||
|
# For standalone: party gets segment as before
|
||||||
|
party_obs_years = Dict{String, Set{Int}}()
|
||||||
|
|
||||||
|
# First pass: collect non-union parties from all data sources (as strings)
|
||||||
|
all_data_parties = Set{String}()
|
||||||
|
for p in levels(manifesto.party)
|
||||||
|
push!(all_data_parties, string(p))
|
||||||
|
end
|
||||||
|
for p in levels(expert_dim.party)
|
||||||
|
push!(all_data_parties, string(p))
|
||||||
|
end
|
||||||
|
for p in levels(expert_lr.party)
|
||||||
|
push!(all_data_parties, string(p))
|
||||||
|
end
|
||||||
|
|
||||||
|
# Initialize observation years for standalone parties and constituents
|
||||||
|
for p in all_data_parties
|
||||||
|
p_int = tryparse(Int, string(p))
|
||||||
|
if p_int !== nothing && has_unions && haskey(union_to_constituents, p_int) && string(p) in union_ids_in_data
|
||||||
|
# This is a union ID in manifesto - skip it, create entries for constituents instead
|
||||||
|
continue
|
||||||
|
end
|
||||||
|
party_obs_years[p] = Set{Int}()
|
||||||
|
end
|
||||||
|
|
||||||
|
# For unions: ensure all constituents have entries
|
||||||
|
if has_unions
|
||||||
|
for (uid, constituents) in union_to_constituents
|
||||||
|
uid_str = string(uid)
|
||||||
|
if uid_str in union_ids_in_data
|
||||||
|
for cid in constituents
|
||||||
|
cid_str = string(cid)
|
||||||
|
if !haskey(party_obs_years, cid_str)
|
||||||
|
party_obs_years[cid_str] = Set{Int}()
|
||||||
|
end
|
||||||
|
end
|
||||||
|
end
|
||||||
|
end
|
||||||
|
end
|
||||||
|
|
||||||
|
# Add years from manifesto
|
||||||
|
for row in eachrow(manifesto)
|
||||||
|
p_str = string(row.party)
|
||||||
|
p_int = tryparse(Int, p_str)
|
||||||
|
if p_int !== nothing && has_unions && haskey(union_to_constituents, p_int) && p_str in union_ids_in_data
|
||||||
|
# Union manifesto: add year to ALL constituents
|
||||||
|
for cid in union_to_constituents[p_int]
|
||||||
|
cid_str = string(cid)
|
||||||
|
if haskey(party_obs_years, cid_str)
|
||||||
|
push!(party_obs_years[cid_str], row.Year)
|
||||||
|
end
|
||||||
|
end
|
||||||
|
else
|
||||||
|
if haskey(party_obs_years, p_str)
|
||||||
|
push!(party_obs_years[p_str], row.Year)
|
||||||
|
end
|
||||||
|
end
|
||||||
|
end
|
||||||
|
|
||||||
|
# Add years from expert_dim (individual party data - direct)
|
||||||
|
for row in eachrow(expert_dim)
|
||||||
|
p_str = string(row.party)
|
||||||
|
if haskey(party_obs_years, p_str)
|
||||||
|
push!(party_obs_years[p_str], row.Year)
|
||||||
|
end
|
||||||
|
end
|
||||||
|
|
||||||
|
# Add years from expert_lr (individual party data - direct)
|
||||||
|
for row in eachrow(expert_lr)
|
||||||
|
p_str = string(row.party)
|
||||||
|
if haskey(party_obs_years, p_str)
|
||||||
|
push!(party_obs_years[p_str], row.Year)
|
||||||
|
end
|
||||||
|
end
|
||||||
|
|
||||||
|
J_original = length(party_obs_years)
|
||||||
|
println("Total number of parties to create segments for: $J_original")
|
||||||
|
|
||||||
|
# =========================================================================
|
||||||
|
# STEP 2: Split parties into segments and filter by MIN_OBS
|
||||||
|
# =========================================================================
|
||||||
|
println("\nCreating segments (splitting at gaps > $MAX_GAP years)...")
|
||||||
|
|
||||||
|
segment_data = []
|
||||||
|
segment_id = 0
|
||||||
|
parties_split = 0
|
||||||
|
segments_dropped = 0
|
||||||
|
observations_dropped = 0
|
||||||
|
|
||||||
|
for (p, years_set) in party_obs_years
|
||||||
|
years = collect(years_set)
|
||||||
|
if isempty(years)
|
||||||
|
continue
|
||||||
|
end
|
||||||
|
|
||||||
|
segments = split_party_years_into_segments(years, MAX_GAP)
|
||||||
|
|
||||||
|
if length(segments) > 1
|
||||||
|
parties_split += 1
|
||||||
|
end
|
||||||
|
|
||||||
|
for (seg_num, seg_years) in enumerate(segments)
|
||||||
|
n_obs = length(seg_years)
|
||||||
|
if n_obs >= MIN_OBS
|
||||||
|
segment_id += 1
|
||||||
|
push!(segment_data, (
|
||||||
|
segment_id = segment_id,
|
||||||
|
party_id = p,
|
||||||
|
segment_num = seg_num,
|
||||||
|
year_start = minimum(seg_years),
|
||||||
|
year_end = maximum(seg_years),
|
||||||
|
n_obs = n_obs
|
||||||
|
))
|
||||||
|
else
|
||||||
|
segments_dropped += 1
|
||||||
|
observations_dropped += n_obs
|
||||||
|
end
|
||||||
|
end
|
||||||
|
end
|
||||||
|
|
||||||
|
segment_info = DataFrame(segment_data)
|
||||||
|
S = nrow(segment_info) # Number of valid segments
|
||||||
|
|
||||||
|
println(" Segments created: $S (from $J_original parties)")
|
||||||
|
println(" Parties split into multiple segments: $parties_split")
|
||||||
|
println(" Segments dropped (< $MIN_OBS obs): $segments_dropped")
|
||||||
|
println(" Observations dropped: $observations_dropped")
|
||||||
|
|
||||||
|
# Get unique parties that have at least one valid segment
|
||||||
|
all_parties = unique(segment_info.party_id)
|
||||||
|
J = length(all_parties)
|
||||||
|
println(" Parties with valid segments: $J")
|
||||||
|
|
||||||
|
# Create party-to-index mapping for the valid parties
|
||||||
|
party_to_index = Dict(all_parties .=> 1:J)
|
||||||
|
|
||||||
|
# =========================================================================
|
||||||
|
# STEP 3: Create segment-year index space (R) - consecutive within segment
|
||||||
|
# =========================================================================
|
||||||
|
println("\nCreating segment-year index space...")
|
||||||
|
|
||||||
|
segment_year_data = []
|
||||||
|
for row in eachrow(segment_info)
|
||||||
|
for y in row.year_start:row.year_end
|
||||||
|
push!(segment_year_data, (
|
||||||
|
segment_id = row.segment_id,
|
||||||
|
party_id = row.party_id,
|
||||||
|
Year = y
|
||||||
|
))
|
||||||
|
end
|
||||||
|
end
|
||||||
|
|
||||||
|
segment_year = DataFrame(segment_year_data)
|
||||||
|
segment_year.rr = 1:nrow(segment_year)
|
||||||
|
R = nrow(segment_year)
|
||||||
|
println("Total segment-year positions (R): $R")
|
||||||
|
|
||||||
|
# Compute len_theta_ts for segments (years per segment)
|
||||||
|
len_theta_ts = [row.year_end - row.year_start + 1 for row in eachrow(segment_info)]
|
||||||
|
@assert sum(len_theta_ts) == R "sum(len_theta_ts)=$(sum(len_theta_ts)) must equal R=$R"
|
||||||
|
|
||||||
|
# =========================================================================
|
||||||
|
# STEP 4: Map observations to segment-year indices (union-aware)
|
||||||
|
# =========================================================================
|
||||||
|
println("\nMapping observations to segment-year indices...")
|
||||||
|
|
||||||
|
# Create lookup: (party_str, year) -> segment_id (for valid segments only)
|
||||||
|
party_year_to_segment = Dict{Tuple{String, Int}, Int}()
|
||||||
|
for row in eachrow(segment_info)
|
||||||
|
for y in row.year_start:row.year_end
|
||||||
|
party_year_to_segment[(string(row.party_id), y)] = row.segment_id
|
||||||
|
end
|
||||||
|
end
|
||||||
|
|
||||||
|
# --- MANIFESTO: union-aware mapping ---
|
||||||
|
# For union manifesto obs: map to first constituent's segment (for ss_man).
|
||||||
|
# The actual theta averaging is handled via constituent arrays.
|
||||||
|
manifesto_segment_ids = Union{Int, Missing}[]
|
||||||
|
for row in eachrow(manifesto)
|
||||||
|
p_str = string(row.party)
|
||||||
|
p_int = tryparse(Int, p_str)
|
||||||
|
key = (p_str, row.Year)
|
||||||
|
|
||||||
|
if haskey(party_year_to_segment, key)
|
||||||
|
# Direct mapping (non-union or constituent with own segment)
|
||||||
|
push!(manifesto_segment_ids, party_year_to_segment[key])
|
||||||
|
elseif p_int !== nothing && has_unions && haskey(union_to_constituents, p_int)
|
||||||
|
# Union party: use first constituent's segment
|
||||||
|
found = false
|
||||||
|
for cid in union_to_constituents[p_int]
|
||||||
|
ckey = (string(cid), row.Year)
|
||||||
|
if haskey(party_year_to_segment, ckey)
|
||||||
|
push!(manifesto_segment_ids, party_year_to_segment[ckey])
|
||||||
|
found = true
|
||||||
|
break
|
||||||
|
end
|
||||||
|
end
|
||||||
|
if !found
|
||||||
|
push!(manifesto_segment_ids, missing)
|
||||||
|
end
|
||||||
|
else
|
||||||
|
push!(manifesto_segment_ids, missing)
|
||||||
|
end
|
||||||
|
end
|
||||||
|
manifesto.segment_id = manifesto_segment_ids
|
||||||
|
n_manifesto_before = nrow(manifesto)
|
||||||
|
manifesto = manifesto[.!ismissing.(manifesto.segment_id), :]
|
||||||
|
manifesto.segment_id = Int.(manifesto.segment_id)
|
||||||
|
println(" Manifesto: $(nrow(manifesto))/$n_manifesto_before observations (dropped $(n_manifesto_before - nrow(manifesto)) in invalid segments)")
|
||||||
|
|
||||||
|
# --- EXPERT DIM: direct mapping (individual party data) ---
|
||||||
|
expert_dim_segment_ids = Union{Int, Missing}[]
|
||||||
|
for row in eachrow(expert_dim)
|
||||||
|
key = (string(row.party), row.Year)
|
||||||
|
if haskey(party_year_to_segment, key)
|
||||||
|
push!(expert_dim_segment_ids, party_year_to_segment[key])
|
||||||
|
else
|
||||||
|
push!(expert_dim_segment_ids, missing)
|
||||||
|
end
|
||||||
|
end
|
||||||
|
expert_dim.segment_id = expert_dim_segment_ids
|
||||||
|
n_expert_dim_before = nrow(expert_dim)
|
||||||
|
expert_dim = expert_dim[.!ismissing.(expert_dim.segment_id), :]
|
||||||
|
expert_dim.segment_id = Int.(expert_dim.segment_id)
|
||||||
|
println(" Expert dim: $(nrow(expert_dim))/$n_expert_dim_before observations (dropped $(n_expert_dim_before - nrow(expert_dim)) in invalid segments)")
|
||||||
|
|
||||||
|
# --- EXPERT LR: direct mapping (individual party data) ---
|
||||||
|
expert_lr_segment_ids = Union{Int, Missing}[]
|
||||||
|
for row in eachrow(expert_lr)
|
||||||
|
key = (string(row.party), row.Year)
|
||||||
|
if haskey(party_year_to_segment, key)
|
||||||
|
push!(expert_lr_segment_ids, party_year_to_segment[key])
|
||||||
|
else
|
||||||
|
push!(expert_lr_segment_ids, missing)
|
||||||
|
end
|
||||||
|
end
|
||||||
|
expert_lr.segment_id = expert_lr_segment_ids
|
||||||
|
n_expert_lr_before = nrow(expert_lr)
|
||||||
|
expert_lr = expert_lr[.!ismissing.(expert_lr.segment_id), :]
|
||||||
|
expert_lr.segment_id = Int.(expert_lr.segment_id)
|
||||||
|
println(" Expert LR: $(nrow(expert_lr))/$n_expert_lr_before observations (dropped $(n_expert_lr_before - nrow(expert_lr)) in invalid segments)")
|
||||||
|
|
||||||
|
# =========================================================================
|
||||||
|
# STEP 5: Create segment indices (ss) for each observation
|
||||||
|
# ss indexes into 1:S (segment space), used for party-level parameters
|
||||||
|
# =========================================================================
|
||||||
|
|
||||||
|
# Create segment_id to ss mapping
|
||||||
|
segment_to_ss = Dict(row.segment_id => i for (i, row) in enumerate(eachrow(segment_info)))
|
||||||
|
|
||||||
|
manifesto.ss_man = [segment_to_ss[sid] for sid in manifesto.segment_id]
|
||||||
|
expert_dim.ss_exp_dim = [segment_to_ss[sid] for sid in expert_dim.segment_id]
|
||||||
|
expert_lr.ss_exp_lr = [segment_to_ss[sid] for sid in expert_lr.segment_id]
|
||||||
|
|
||||||
|
# Validate segment indices
|
||||||
|
@assert all(1 .<= manifesto.ss_man .<= S)
|
||||||
|
@assert all(1 .<= expert_dim.ss_exp_dim .<= S)
|
||||||
|
@assert all(1 .<= expert_lr.ss_exp_lr .<= S)
|
||||||
|
|
||||||
|
# =========================================================================
|
||||||
|
# STEP 6: Map rr indices (segment-year) to datasets via leftjoin
|
||||||
|
# =========================================================================
|
||||||
|
|
||||||
|
# Create (segment_id, Year) -> rr lookup
|
||||||
|
seg_year_to_rr = Dict{Tuple{Int, Int}, Int}()
|
||||||
|
for row in eachrow(segment_year)
|
||||||
|
seg_year_to_rr[(row.segment_id, row.Year)] = row.rr
|
||||||
|
end
|
||||||
|
|
||||||
|
# Join manifesto with segment_year to get rr indices
|
||||||
|
manifesto = leftjoin(manifesto, segment_year, on=[:segment_id, :Year])
|
||||||
|
rename!(manifesto, :rr => :rr_man)
|
||||||
|
|
||||||
|
expert_dim = leftjoin(expert_dim, segment_year, on=[:segment_id, :Year])
|
||||||
|
rename!(expert_dim, :rr => :rr_exp_dim)
|
||||||
|
|
||||||
|
expert_lr = leftjoin(expert_lr, segment_year, on=[:segment_id, :Year])
|
||||||
|
rename!(expert_lr, :rr => :rr_exp_lr)
|
||||||
|
|
||||||
|
# Validate rr mappings
|
||||||
|
@assert all(!ismissing, manifesto.rr_man) "Some manifesto observations have no rr_man mapping"
|
||||||
|
@assert all(!ismissing, expert_dim.rr_exp_dim) "Some expert_dim observations have no rr_exp_dim mapping"
|
||||||
|
@assert all(!ismissing, expert_lr.rr_exp_lr) "Some expert_lr observations have no rr_exp_lr mapping"
|
||||||
|
|
||||||
|
# Convert to Int
|
||||||
|
manifesto.rr_man = Int.(manifesto.rr_man)
|
||||||
|
expert_dim.rr_exp_dim = Int.(expert_dim.rr_exp_dim)
|
||||||
|
expert_lr.rr_exp_lr = Int.(expert_lr.rr_exp_lr)
|
||||||
|
|
||||||
|
# Validate bounds
|
||||||
|
@assert all(1 .<= manifesto.rr_man .<= R) "rr_man out of bounds"
|
||||||
|
@assert all(1 .<= expert_dim.rr_exp_dim .<= R) "rr_exp_dim out of bounds"
|
||||||
|
@assert all(1 .<= expert_lr.rr_exp_lr .<= R) "rr_exp_lr out of bounds"
|
||||||
|
|
||||||
|
# Print diagnostics
|
||||||
|
n_observed = length(unique(vcat(manifesto.rr_man, expert_dim.rr_exp_dim, expert_lr.rr_exp_lr)))
|
||||||
|
println("\nSegment-years with data: $n_observed / $R ($(round(100*n_observed/R, digits=1))%)")
|
||||||
|
println("Segment-years for interpolation: $(R - n_observed)")
|
||||||
|
|
||||||
|
# =========================================================================
|
||||||
|
# STEP 6b: Build constituent arrays for union manifesto/expert observations
|
||||||
|
# For each manifesto obs: store list of constituent rr indices for averaging
|
||||||
|
# Non-union obs: single rr (n_const=1)
|
||||||
|
# Union obs: multiple rr values (n_const=len(constituents))
|
||||||
|
# =========================================================================
|
||||||
|
println("\nBuilding constituent arrays for mean-constituent model...")
|
||||||
|
|
||||||
|
# --- Manifesto constituent arrays ---
|
||||||
|
n_const_man_vec = Int[] # n_const for each manifesto obs
|
||||||
|
const_rr_man_vec = Int[] # flat array of constituent rr values
|
||||||
|
const_offset_man_vec = Int[] # offset into const_rr for each obs
|
||||||
|
|
||||||
|
offset = 1
|
||||||
|
for row in eachrow(manifesto)
|
||||||
|
p_str = string(row.party)
|
||||||
|
p_int = tryparse(Int, p_str)
|
||||||
|
|
||||||
|
if p_int !== nothing && has_unions && haskey(union_to_constituents, p_int) && p_str in union_ids_in_data
|
||||||
|
# Union manifesto obs: find rr for each constituent in this year
|
||||||
|
constituent_rrs = Int[]
|
||||||
|
for cid in union_to_constituents[p_int]
|
||||||
|
ckey = (string(cid), row.Year)
|
||||||
|
if haskey(party_year_to_segment, ckey)
|
||||||
|
sid = party_year_to_segment[ckey]
|
||||||
|
rr_key = (sid, row.Year)
|
||||||
|
if haskey(seg_year_to_rr, rr_key)
|
||||||
|
push!(constituent_rrs, seg_year_to_rr[rr_key])
|
||||||
|
end
|
||||||
|
end
|
||||||
|
end
|
||||||
|
if isempty(constituent_rrs)
|
||||||
|
# Fallback: use the rr_man already assigned
|
||||||
|
push!(constituent_rrs, row.rr_man)
|
||||||
|
end
|
||||||
|
push!(n_const_man_vec, length(constituent_rrs))
|
||||||
|
push!(const_offset_man_vec, offset)
|
||||||
|
append!(const_rr_man_vec, constituent_rrs)
|
||||||
|
offset += length(constituent_rrs)
|
||||||
|
else
|
||||||
|
# Non-union: single constituent (itself)
|
||||||
|
push!(n_const_man_vec, 1)
|
||||||
|
push!(const_offset_man_vec, offset)
|
||||||
|
push!(const_rr_man_vec, row.rr_man)
|
||||||
|
offset += 1
|
||||||
|
end
|
||||||
|
end
|
||||||
|
|
||||||
|
N_const_man_total = length(const_rr_man_vec)
|
||||||
|
n_union_man = count(x -> x > 1, n_const_man_vec)
|
||||||
|
println(" Manifesto: $(nrow(manifesto)) obs, $n_union_man union obs, $N_const_man_total total constituent entries")
|
||||||
|
|
||||||
|
# --- Expert dim constituent arrays ---
|
||||||
|
# Individual expert obs always have n_const=1 (direct mapping)
|
||||||
|
# Union-level expert obs (if any) would average — but typically expert data is at individual party level
|
||||||
|
n_const_exp_dim_vec = Int[]
|
||||||
|
const_rr_exp_dim_vec = Int[]
|
||||||
|
const_offset_exp_dim_vec = Int[]
|
||||||
|
|
||||||
|
offset = 1
|
||||||
|
for row in eachrow(expert_dim)
|
||||||
|
p_str = string(row.party)
|
||||||
|
p_int = tryparse(Int, p_str)
|
||||||
|
|
||||||
|
if p_int !== nothing && has_unions && haskey(union_to_constituents, p_int) && p_str in union_ids_in_data
|
||||||
|
# Union-level expert obs: average over constituents
|
||||||
|
constituent_rrs = Int[]
|
||||||
|
for cid in union_to_constituents[p_int]
|
||||||
|
ckey = (string(cid), row.Year)
|
||||||
|
if haskey(party_year_to_segment, ckey)
|
||||||
|
sid = party_year_to_segment[ckey]
|
||||||
|
rr_key = (sid, row.Year)
|
||||||
|
if haskey(seg_year_to_rr, rr_key)
|
||||||
|
push!(constituent_rrs, seg_year_to_rr[rr_key])
|
||||||
|
end
|
||||||
|
end
|
||||||
|
end
|
||||||
|
if isempty(constituent_rrs)
|
||||||
|
push!(constituent_rrs, row.rr_exp_dim)
|
||||||
|
end
|
||||||
|
push!(n_const_exp_dim_vec, length(constituent_rrs))
|
||||||
|
push!(const_offset_exp_dim_vec, offset)
|
||||||
|
append!(const_rr_exp_dim_vec, constituent_rrs)
|
||||||
|
offset += length(constituent_rrs)
|
||||||
|
else
|
||||||
|
# Individual party obs
|
||||||
|
push!(n_const_exp_dim_vec, 1)
|
||||||
|
push!(const_offset_exp_dim_vec, offset)
|
||||||
|
push!(const_rr_exp_dim_vec, row.rr_exp_dim)
|
||||||
|
offset += 1
|
||||||
|
end
|
||||||
|
end
|
||||||
|
|
||||||
|
N_const_exp_dim_total = length(const_rr_exp_dim_vec)
|
||||||
|
n_union_exp_dim = count(x -> x > 1, n_const_exp_dim_vec)
|
||||||
|
println(" Expert dim: $(nrow(expert_dim)) obs, $n_union_exp_dim union obs, $N_const_exp_dim_total total constituent entries")
|
||||||
|
|
||||||
|
# --- Expert LR constituent arrays ---
|
||||||
|
n_const_exp_lr_vec = Int[]
|
||||||
|
const_rr_exp_lr_vec = Int[]
|
||||||
|
const_offset_exp_lr_vec = Int[]
|
||||||
|
|
||||||
|
offset = 1
|
||||||
|
for row in eachrow(expert_lr)
|
||||||
|
p_str = string(row.party)
|
||||||
|
p_int = tryparse(Int, p_str)
|
||||||
|
|
||||||
|
if p_int !== nothing && has_unions && haskey(union_to_constituents, p_int) && p_str in union_ids_in_data
|
||||||
|
constituent_rrs = Int[]
|
||||||
|
for cid in union_to_constituents[p_int]
|
||||||
|
ckey = (string(cid), row.Year)
|
||||||
|
if haskey(party_year_to_segment, ckey)
|
||||||
|
sid = party_year_to_segment[ckey]
|
||||||
|
rr_key = (sid, row.Year)
|
||||||
|
if haskey(seg_year_to_rr, rr_key)
|
||||||
|
push!(constituent_rrs, seg_year_to_rr[rr_key])
|
||||||
|
end
|
||||||
|
end
|
||||||
|
end
|
||||||
|
if isempty(constituent_rrs)
|
||||||
|
push!(constituent_rrs, row.rr_exp_lr)
|
||||||
|
end
|
||||||
|
push!(n_const_exp_lr_vec, length(constituent_rrs))
|
||||||
|
push!(const_offset_exp_lr_vec, offset)
|
||||||
|
append!(const_rr_exp_lr_vec, constituent_rrs)
|
||||||
|
offset += length(constituent_rrs)
|
||||||
|
else
|
||||||
|
push!(n_const_exp_lr_vec, 1)
|
||||||
|
push!(const_offset_exp_lr_vec, offset)
|
||||||
|
push!(const_rr_exp_lr_vec, row.rr_exp_lr)
|
||||||
|
offset += 1
|
||||||
|
end
|
||||||
|
end
|
||||||
|
|
||||||
|
N_const_exp_lr_total = length(const_rr_exp_lr_vec)
|
||||||
|
n_union_exp_lr = count(x -> x > 1, n_const_exp_lr_vec)
|
||||||
|
println(" Expert LR: $(nrow(expert_lr)) obs, $n_union_exp_lr union obs, $N_const_exp_lr_total total constituent entries")
|
||||||
|
|
||||||
|
# =========================================================================
|
||||||
|
# STEP 7: Filter manifesto items with sufficient observations
|
||||||
|
# =========================================================================
|
||||||
|
var_man_counts_df = combine(groupby(manifesto, :var), :var => length => :n_obs)
|
||||||
|
var_man_counts_df = var_man_counts_df[var_man_counts_df.n_obs .>= 2, :]
|
||||||
|
var_man_counts = var_man_counts_df.var
|
||||||
|
|
||||||
|
var_exp_dim_counts_df = combine(groupby(expert_dim, :var), :var => length => :n_obs)
|
||||||
|
var_exp_dim_counts_df = var_exp_dim_counts_df[var_exp_dim_counts_df.n_obs .>= 2, :]
|
||||||
|
var_exp_dim_counts = var_exp_dim_counts_df.var
|
||||||
|
|
||||||
|
var_exp_lr_counts_df = combine(groupby(expert_lr, :var), :var => length => :n_obs)
|
||||||
|
var_exp_lr_counts_df = var_exp_lr_counts_df[var_exp_lr_counts_df.n_obs .>= 2, :]
|
||||||
|
var_exp_lr_counts = var_exp_lr_counts_df.var
|
||||||
|
|
||||||
|
manifesto = manifesto[in.(manifesto.var, Ref(var_man_counts)), :]
|
||||||
|
manifesto.var_man = levelcode.(categorical(manifesto.var, levels=unique(var_man_counts)))
|
||||||
|
|
||||||
|
expert_dim = expert_dim[in.(expert_dim.var, Ref(var_exp_dim_counts)), :]
|
||||||
|
expert_dim.var_exp_dim = levelcode.(categorical(expert_dim.var, levels=unique(var_exp_dim_counts)))
|
||||||
|
|
||||||
|
expert_lr = expert_lr[in.(expert_lr.var, Ref(var_exp_lr_counts)), :]
|
||||||
|
expert_lr.var_exp_lr = levelcode.(categorical(expert_lr.var, levels=unique(var_exp_lr_counts)))
|
||||||
|
|
||||||
|
# =========================================================================
|
||||||
|
# STEP 8: Country (group) indexing
|
||||||
|
# =========================================================================
|
||||||
|
all_groups = unique(vcat(levels(manifesto.country), levels(expert_dim.country), levels(expert_lr.country)))
|
||||||
|
P = length(all_groups)
|
||||||
|
println("Total number of unique countries (P): $P")
|
||||||
|
|
||||||
|
group_to_index = Dict(all_groups .=> 1:P)
|
||||||
|
manifesto.pp_man = [group_to_index[c] for c in manifesto.country]
|
||||||
|
expert_dim.pp_exp_dim = [group_to_index[c] for c in expert_dim.country]
|
||||||
|
expert_lr.pp_exp_lr = [group_to_index[c] for c in expert_lr.country]
|
||||||
|
|
||||||
|
@assert all(1 .<= manifesto.pp_man .<= P)
|
||||||
|
@assert all(1 .<= expert_dim.pp_exp_dim .<= P)
|
||||||
|
@assert all(1 .<= expert_lr.pp_exp_lr .<= P)
|
||||||
|
|
||||||
|
# =========================================================================
|
||||||
|
# STEP 9: Country-item-year combinations for zero-inflation model
|
||||||
|
# =========================================================================
|
||||||
|
manifesto.ciy_key = string.(manifesto.country, "_", manifesto.var_man, "_", manifesto.Year)
|
||||||
|
ciy_keys = unique(manifesto.ciy_key)
|
||||||
|
N_ciy = length(ciy_keys)
|
||||||
|
println("Total unique country-item-year combinations (N_ciy): $N_ciy")
|
||||||
|
|
||||||
|
ciy_key_to_index = Dict(ciy_keys .=> 1:N_ciy)
|
||||||
|
manifesto.ciy_idx = [ciy_key_to_index[k] for k in manifesto.ciy_key]
|
||||||
|
@assert all(1 .<= manifesto.ciy_idx .<= N_ciy)
|
||||||
|
|
||||||
|
# =========================================================================
|
||||||
|
# STEP 10: Segment-country mapping (each segment inherits from party)
|
||||||
|
# For constituents: inherit country from their union's manifesto data
|
||||||
|
# =========================================================================
|
||||||
|
party_country_dict = Dict{String, Int}()
|
||||||
|
|
||||||
|
# From data directly
|
||||||
|
for row in eachrow(manifesto)
|
||||||
|
p_str = string(row.party)
|
||||||
|
p_int = tryparse(Int, p_str)
|
||||||
|
c_idx = group_to_index[string(row.country)]
|
||||||
|
party_country_dict[p_str] = c_idx
|
||||||
|
|
||||||
|
# If union, also assign country to all constituents
|
||||||
|
if p_int !== nothing && has_unions && haskey(union_to_constituents, p_int)
|
||||||
|
for cid in union_to_constituents[p_int]
|
||||||
|
party_country_dict[string(cid)] = c_idx
|
||||||
|
end
|
||||||
|
end
|
||||||
|
end
|
||||||
|
for row in eachrow(expert_dim)
|
||||||
|
party_country_dict[string(row.party)] = group_to_index[string(row.country)]
|
||||||
|
end
|
||||||
|
for row in eachrow(expert_lr)
|
||||||
|
party_country_dict[string(row.party)] = group_to_index[string(row.country)]
|
||||||
|
end
|
||||||
|
|
||||||
|
# Map each segment to its party's country
|
||||||
|
segment_country_idx = Int[]
|
||||||
|
for row in eachrow(segment_info)
|
||||||
|
pid = string(row.party_id)
|
||||||
|
if haskey(party_country_dict, pid)
|
||||||
|
push!(segment_country_idx, party_country_dict[pid])
|
||||||
|
else
|
||||||
|
# Fallback: try to find via union mapping
|
||||||
|
pf_int = tryparse(Int, pid)
|
||||||
|
if pf_int !== nothing && haskey(constituent_to_union, pf_int)
|
||||||
|
uid = constituent_to_union[pf_int]
|
||||||
|
if haskey(party_country_dict, string(uid))
|
||||||
|
push!(segment_country_idx, party_country_dict[string(uid)])
|
||||||
|
else
|
||||||
|
error("Cannot find country for constituent $pid (union $uid)")
|
||||||
|
end
|
||||||
|
else
|
||||||
|
error("Cannot find country for party $pid")
|
||||||
|
end
|
||||||
|
end
|
||||||
|
end
|
||||||
|
@assert all(1 .<= segment_country_idx .<= P)
|
||||||
|
@assert length(segment_country_idx) == S
|
||||||
|
|
||||||
|
# Persist canonical country on segment mapping tables
|
||||||
|
segment_country = [all_groups[idx] for idx in segment_country_idx]
|
||||||
|
segment_info.country = segment_country
|
||||||
|
|
||||||
|
segment_country_by_id = Dict(row.segment_id => segment_country[i] for (i, row) in enumerate(eachrow(segment_info)))
|
||||||
|
segment_year.country = [segment_country_by_id[sid] for sid in segment_year.segment_id]
|
||||||
|
|
||||||
|
# =========================================================================
|
||||||
|
# STEP 11: Load party family data and map to segments
|
||||||
|
# =========================================================================
|
||||||
|
println("\nLoading party family data...")
|
||||||
|
party_families_file = joinpath("data", "party_families.csv")
|
||||||
|
if !isfile(party_families_file)
|
||||||
|
error("party_families.csv not found at $party_families_file")
|
||||||
|
end
|
||||||
|
party_families_df = CSV.read(party_families_file, DataFrame)
|
||||||
|
|
||||||
|
pf_to_family = Dict(row.partyfacts_id => row.family for row in eachrow(party_families_df))
|
||||||
|
|
||||||
|
all_family_names = unique(party_families_df.family)
|
||||||
|
family_to_idx = Dict(f => i for (i, f) in enumerate(all_family_names))
|
||||||
|
F = length(all_family_names)
|
||||||
|
println("Total number of party families (F): $F")
|
||||||
|
println("Family categories: ", join(all_family_names, ", "))
|
||||||
|
|
||||||
|
# Map each segment to its party's family
|
||||||
|
# For constituents: try own family first, fall back to union's family
|
||||||
|
segment_family_idx = Int[]
|
||||||
|
unmatched_segments = String[]
|
||||||
|
default_family_idx = haskey(family_to_idx, "other") ? family_to_idx["other"] : 1
|
||||||
|
|
||||||
|
for row in eachrow(segment_info)
|
||||||
|
pf_id = tryparse(Int, string(row.party_id))
|
||||||
|
if !isnothing(pf_id) && haskey(pf_to_family, pf_id)
|
||||||
|
family_name = pf_to_family[pf_id]
|
||||||
|
push!(segment_family_idx, family_to_idx[family_name])
|
||||||
|
elseif !isnothing(pf_id) && haskey(constituent_to_union, pf_id) && haskey(pf_to_family, constituent_to_union[pf_id])
|
||||||
|
# Fall back to union's family
|
||||||
|
family_name = pf_to_family[constituent_to_union[pf_id]]
|
||||||
|
push!(segment_family_idx, family_to_idx[family_name])
|
||||||
|
else
|
||||||
|
push!(segment_family_idx, default_family_idx)
|
||||||
|
push!(unmatched_segments, "$(row.party_id)_seg$(row.segment_num)")
|
||||||
|
end
|
||||||
|
end
|
||||||
|
|
||||||
|
if !isempty(unmatched_segments)
|
||||||
|
println(" Warning: $(length(unmatched_segments)) segments not matched to families (assigned to 'other')")
|
||||||
|
if length(unmatched_segments) <= 10
|
||||||
|
println(" Unmatched: ", join(unmatched_segments, ", "))
|
||||||
|
end
|
||||||
|
end
|
||||||
|
|
||||||
|
@assert all(1 .<= segment_family_idx .<= F)
|
||||||
|
@assert length(segment_family_idx) == S
|
||||||
|
|
||||||
|
# =========================================================================
|
||||||
|
# STEP 12: Find anchor segment
|
||||||
|
# With unions: use CDU (1375) as anchor (individual constituent)
|
||||||
|
# Without unions: use CDU/CSU (211) as anchor (union ID)
|
||||||
|
# =========================================================================
|
||||||
|
anchor_party_id = has_unions ? 1375 : 211
|
||||||
|
anchor_label = has_unions ? "CDU" : "CDU/CSU"
|
||||||
|
anchor_segments = filter(row -> tryparse(Int, string(row.party_id)) == anchor_party_id, segment_info)
|
||||||
|
|
||||||
|
if nrow(anchor_segments) > 0
|
||||||
|
# Pick segment with most observations
|
||||||
|
anchor_segment_idx = anchor_segments[argmax(anchor_segments.n_obs), :segment_id]
|
||||||
|
# Convert to ss index (1:S)
|
||||||
|
anchor_segment_ss = segment_to_ss[anchor_segment_idx]
|
||||||
|
println(" Anchor segment ($anchor_label, ID $anchor_party_id): segment $anchor_segment_ss ($(anchor_segments[argmax(anchor_segments.n_obs), :n_obs]) obs)")
|
||||||
|
else
|
||||||
|
println(" Warning: Anchor party $anchor_label (ID $anchor_party_id) not found, using segment 1")
|
||||||
|
anchor_segment_ss = 1
|
||||||
|
end
|
||||||
|
|
||||||
|
# =========================================================================
|
||||||
|
# STEP 13: Validation - check no long gaps remain within segments
|
||||||
|
# =========================================================================
|
||||||
|
println("\nValidating segment structure...")
|
||||||
|
rr_to_segment_year = Dict(row.rr => (segment_id=row.segment_id, year=row.Year) for row in eachrow(segment_year))
|
||||||
|
seg_obs_years = Dict(row.segment_id => Set{Int}() for row in eachrow(segment_info))
|
||||||
|
|
||||||
|
for row in eachrow(manifesto)
|
||||||
|
push!(seg_obs_years[row.segment_id], row.Year)
|
||||||
|
end
|
||||||
|
for row in eachrow(expert_dim)
|
||||||
|
push!(seg_obs_years[row.segment_id], row.Year)
|
||||||
|
end
|
||||||
|
for row in eachrow(expert_lr)
|
||||||
|
push!(seg_obs_years[row.segment_id], row.Year)
|
||||||
|
end
|
||||||
|
|
||||||
|
# Union manifesto rows contribute to every constituent through const_rr_man_vec,
|
||||||
|
# even though row.segment_id stores only a representative segment for indexing.
|
||||||
|
# Include these constituent rr values so validation matches the actual Stan data.
|
||||||
|
for rr in const_rr_man_vec
|
||||||
|
if haskey(rr_to_segment_year, rr)
|
||||||
|
sy = rr_to_segment_year[rr]
|
||||||
|
push!(seg_obs_years[sy.segment_id], sy.year)
|
||||||
|
end
|
||||||
|
end
|
||||||
|
|
||||||
|
max_internal_gap = 0
|
||||||
|
for (i, row) in enumerate(eachrow(segment_info))
|
||||||
|
seg_obs = collect(seg_obs_years[row.segment_id])
|
||||||
|
if length(seg_obs) > 1
|
||||||
|
gaps = diff(sort(unique(seg_obs)))
|
||||||
|
if !isempty(gaps)
|
||||||
|
max_gap_in_seg = maximum(gaps)
|
||||||
|
max_internal_gap = max(max_internal_gap, max_gap_in_seg)
|
||||||
|
if max_gap_in_seg > MAX_GAP
|
||||||
|
@warn "Segment $i (party $(row.party_id)) has internal gap of $max_gap_in_seg years!"
|
||||||
|
end
|
||||||
|
end
|
||||||
|
end
|
||||||
|
end
|
||||||
|
println(" Maximum internal gap within segments: $max_internal_gap years (limit: $MAX_GAP)")
|
||||||
|
|
||||||
|
# Validate all segments have >= MIN_OBS
|
||||||
|
for row in eachrow(segment_info)
|
||||||
|
@assert row.n_obs >= MIN_OBS "Segment $(row.segment_id) has only $(row.n_obs) obs (min: $MIN_OBS)"
|
||||||
|
end
|
||||||
|
println(" All segments have >= $MIN_OBS observations: PASS")
|
||||||
|
|
||||||
|
println("\nSegment-based indexing created successfully")
|
||||||
|
println(" S (segments): $S")
|
||||||
|
println(" R (segment-years): $R")
|
||||||
|
println(" J (parties with valid segments): $J")
|
||||||
|
|
||||||
|
return (manifesto=manifesto, expert_dim=expert_dim, expert_lr=expert_lr,
|
||||||
|
segment_year=segment_year, segment_info=segment_info,
|
||||||
|
all_parties=all_parties, all_groups=all_groups,
|
||||||
|
S=S, J=J, P=P, R=R, N_ciy=N_ciy, len_theta_ts=len_theta_ts,
|
||||||
|
segment_country_idx=segment_country_idx, group_to_index=group_to_index,
|
||||||
|
F=F, segment_family_idx=segment_family_idx, anchor_segment_idx=anchor_segment_ss,
|
||||||
|
# Constituent arrays for mean-constituent model
|
||||||
|
N_const_man_total=N_const_man_total,
|
||||||
|
n_const_man=n_const_man_vec,
|
||||||
|
const_offset_man=const_offset_man_vec,
|
||||||
|
const_rr_man=const_rr_man_vec,
|
||||||
|
N_const_exp_dim_total=N_const_exp_dim_total,
|
||||||
|
n_const_exp_dim=n_const_exp_dim_vec,
|
||||||
|
const_offset_exp_dim=const_offset_exp_dim_vec,
|
||||||
|
const_rr_exp_dim=const_rr_exp_dim_vec,
|
||||||
|
N_const_exp_lr_total=N_const_exp_lr_total,
|
||||||
|
n_const_exp_lr=n_const_exp_lr_vec,
|
||||||
|
const_offset_exp_lr=const_offset_exp_lr_vec,
|
||||||
|
const_rr_exp_lr=const_rr_exp_lr_vec,
|
||||||
|
union_to_constituents=union_to_constituents,
|
||||||
|
constituent_to_union=constituent_to_union)
|
||||||
|
end
|
||||||
|
|
||||||
|
function finalize_4dim_stan_data(manifesto, expert_dim, expert_lr, segment_year, segment_info,
|
||||||
|
all_parties, all_groups, group_to_index, year0, S, J, P, R, N_ciy,
|
||||||
|
len_theta_ts, segment_country_idx, F, segment_family_idx, anchor_segment_idx;
|
||||||
|
N_const_man_total=0, n_const_man=Int[], const_offset_man=Int[], const_rr_man=Int[],
|
||||||
|
N_const_exp_dim_total=0, n_const_exp_dim=Int[], const_offset_exp_dim=Int[], const_rr_exp_dim=Int[],
|
||||||
|
N_const_exp_lr_total=0, n_const_exp_lr=Int[], const_offset_exp_lr=Int[], const_rr_exp_lr=Int[])
|
||||||
|
println("Finalizing 4D Stan data structure (V10: segment-based)...")
|
||||||
|
|
||||||
|
# Map years for temporal indexing
|
||||||
|
years_all = sort(unique(vcat(manifesto.Year, expert_dim.Year, expert_lr.Year)))
|
||||||
|
T_year = length(years_all)
|
||||||
|
year_map = DataFrame(year_rel=years_all, year_ix=1:T_year)
|
||||||
|
|
||||||
|
# Apply year mapping to all datasets
|
||||||
|
manifesto = leftjoin(manifesto, year_map, on=[:Year => :year_rel])
|
||||||
|
rename!(manifesto, :year_ix => :year_for_man)
|
||||||
|
|
||||||
|
expert_dim = leftjoin(expert_dim, year_map, on=[:Year => :year_rel])
|
||||||
|
rename!(expert_dim, :year_ix => :year_for_exp_dim)
|
||||||
|
|
||||||
|
expert_lr = leftjoin(expert_lr, year_map, on=[:Year => :year_rel])
|
||||||
|
rename!(expert_lr, :year_ix => :year_for_exp_lr)
|
||||||
|
|
||||||
|
# Validate year assignments
|
||||||
|
@assert all(.!ismissing.(manifesto.year_for_man))
|
||||||
|
@assert all(.!ismissing.(expert_dim.year_for_exp_dim))
|
||||||
|
@assert all(.!ismissing.(expert_lr.year_for_exp_lr))
|
||||||
|
@assert all(1 .<= manifesto.year_for_man .<= T_year)
|
||||||
|
@assert all(1 .<= expert_dim.year_for_exp_dim .<= T_year)
|
||||||
|
@assert all(1 .<= expert_lr.year_for_exp_lr .<= T_year)
|
||||||
|
|
||||||
|
# Add small epsilon to prevent exact zeros and ones
|
||||||
|
epsilon = 1e-6
|
||||||
|
expert_dim.val = clamp.(expert_dim.val, epsilon, 1.0 - epsilon)
|
||||||
|
expert_lr.val = clamp.(expert_lr.val, epsilon, 1.0 - epsilon)
|
||||||
|
|
||||||
|
# Calculate prior means
|
||||||
|
man_positive_sample = manifesto.positive[manifesto.sample .> 0] ./ manifesto.sample[manifesto.sample .> 0]
|
||||||
|
mn_resp_log_man = StatsFuns.logit(mean(man_positive_sample))
|
||||||
|
mn_resp_log_exp_dim = StatsFuns.logit(mean(expert_dim.val))
|
||||||
|
mn_resp_log_exp_lr = StatsFuns.logit(mean(expert_lr.val))
|
||||||
|
|
||||||
|
println("Prior means calculated:")
|
||||||
|
println(" Manifesto: $(round(mn_resp_log_man, digits=3))")
|
||||||
|
println(" Expert dimension-specific: $(round(mn_resp_log_exp_dim, digits=3))")
|
||||||
|
println(" Expert general L-R: $(round(mn_resp_log_exp_lr, digits=3))")
|
||||||
|
|
||||||
|
# V6: Decade indexing for hierarchical L-R weights
|
||||||
|
expert_lr_actual_years = expert_lr.Year .+ year0
|
||||||
|
expert_lr_decade_raw = div.(expert_lr_actual_years, 10)
|
||||||
|
all_lr_decades = sort(unique(expert_lr_decade_raw))
|
||||||
|
lr_decade_to_index = Dict(all_lr_decades .=> 1:length(all_lr_decades))
|
||||||
|
dd_exp_lr = [lr_decade_to_index[d] for d in expert_lr_decade_raw]
|
||||||
|
D_lr = length(all_lr_decades)
|
||||||
|
println(" Decade indexing (V6): $D_lr decades, range $(minimum(all_lr_decades)*10)s-$(maximum(all_lr_decades)*10)s")
|
||||||
|
|
||||||
|
# Create Stan data dictionary - V10 uses S (segments) instead of J (parties)
|
||||||
|
dat_4dim = Dict(
|
||||||
|
# Common data - V10: S = number of segments
|
||||||
|
"S" => S, # NEW: Number of segments (was J)
|
||||||
|
"J" => J, # Keep J for reference (parties with valid segments)
|
||||||
|
"P" => P,
|
||||||
|
"R" => R,
|
||||||
|
"T_year" => T_year,
|
||||||
|
"len_theta_ts" => Int.(len_theta_ts),
|
||||||
|
|
||||||
|
# Segment-country mapping (V10: segments inherit country from party)
|
||||||
|
"segment_country" => segment_country_idx,
|
||||||
|
|
||||||
|
# Segment family data (V10: segments inherit family from party)
|
||||||
|
"F" => F,
|
||||||
|
"segment_family" => segment_family_idx,
|
||||||
|
|
||||||
|
# Anchor segment for identification (CDU/CSU segment)
|
||||||
|
"anchor_segment" => anchor_segment_idx,
|
||||||
|
|
||||||
|
# Manifesto data - use ss_man (segment index) instead of jj_man (party index)
|
||||||
|
"N_man" => nrow(manifesto),
|
||||||
|
"K_man" => length(unique(manifesto.var_man)),
|
||||||
|
"kk_man" => manifesto.var_man,
|
||||||
|
"ss_man" => manifesto.ss_man, # V10: segment index (was jj_man)
|
||||||
|
"rr_man" => manifesto.rr_man,
|
||||||
|
"pp_man" => manifesto.pp_man,
|
||||||
|
"positive" => manifesto.positive,
|
||||||
|
"sample" => manifesto.sample,
|
||||||
|
"year_for_man" => manifesto.year_for_man,
|
||||||
|
"type_high_idx_man" => manifesto.type_high_idx,
|
||||||
|
"type_low_idx_man" => manifesto.type_low_idx,
|
||||||
|
|
||||||
|
# V1 (2D model): dimension index and direction for text data
|
||||||
|
"dim_idx_man" => manifesto.dim_idx_man,
|
||||||
|
"direction_man" => manifesto.direction_man,
|
||||||
|
|
||||||
|
# Country-item-year data for zero-inflation
|
||||||
|
"N_ciy" => N_ciy,
|
||||||
|
"ciy_idx" => manifesto.ciy_idx,
|
||||||
|
|
||||||
|
# Expert dimension-specific data
|
||||||
|
"N_exp_dim" => nrow(expert_dim),
|
||||||
|
"K_exp_dim" => length(unique(expert_dim.var_exp_dim)),
|
||||||
|
"kk_exp_dim" => expert_dim.var_exp_dim,
|
||||||
|
"ss_exp_dim" => expert_dim.ss_exp_dim, # V10: segment index
|
||||||
|
"rr_exp_dim" => expert_dim.rr_exp_dim,
|
||||||
|
"pp_exp_dim" => expert_dim.pp_exp_dim,
|
||||||
|
# V5 K-scaling: use rounded sum (mean × K × n_scale) and total trials (K × n_scale)
|
||||||
|
"val_dim_int" => Int.(clamp.(round.(expert_dim.val .* expert_dim.n_scale .* expert_dim.n_experts), 0, expert_dim.n_scale .* expert_dim.n_experts)),
|
||||||
|
"n_total_exp_dim" => expert_dim.n_scale .* expert_dim.n_experts,
|
||||||
|
"n_experts_exp_dim" => expert_dim.n_experts,
|
||||||
|
"type_high_idx" => expert_dim.type_high_idx,
|
||||||
|
"type_low_idx" => expert_dim.type_low_idx,
|
||||||
|
|
||||||
|
# V1 (2D model): dimension index for expert dimension data
|
||||||
|
"dim_idx_exp" => expert_dim.dim_idx_exp,
|
||||||
|
|
||||||
|
# Expert general L-R data
|
||||||
|
"N_exp_lr" => nrow(expert_lr),
|
||||||
|
"K_exp_lr" => length(unique(expert_lr.var_exp_lr)),
|
||||||
|
"kk_exp_lr" => expert_lr.var_exp_lr,
|
||||||
|
"ss_exp_lr" => expert_lr.ss_exp_lr, # V10: segment index
|
||||||
|
"rr_exp_lr" => expert_lr.rr_exp_lr,
|
||||||
|
"pp_exp_lr" => expert_lr.pp_exp_lr,
|
||||||
|
# V5 K-scaling: use rounded sum (mean × K × n_scale) and total trials (K × n_scale)
|
||||||
|
"val_lr_int" => Int.(clamp.(round.(expert_lr.val .* expert_lr.n_scale .* expert_lr.n_experts), 0, expert_lr.n_scale .* expert_lr.n_experts)),
|
||||||
|
"n_total_exp_lr" => expert_lr.n_scale .* expert_lr.n_experts,
|
||||||
|
"n_experts_exp_lr" => expert_lr.n_experts,
|
||||||
|
|
||||||
|
# V6: Decade indexing for hierarchical L-R weights
|
||||||
|
"D_lr" => D_lr,
|
||||||
|
"dd_exp_lr" => dd_exp_lr,
|
||||||
|
|
||||||
|
# Prior information
|
||||||
|
"mn_resp_log_man" => mn_resp_log_man,
|
||||||
|
"mn_resp_log_exp_dim" => mn_resp_log_exp_dim,
|
||||||
|
"mn_resp_log_exp_lr" => mn_resp_log_exp_lr,
|
||||||
|
|
||||||
|
# Constituent arrays for mean-constituent model (V4)
|
||||||
|
"N_const_man_total" => max(1, N_const_man_total),
|
||||||
|
"n_const_man" => isempty(n_const_man) ? ones(Int, nrow(manifesto)) : n_const_man,
|
||||||
|
"const_offset_man" => isempty(const_offset_man) ? collect(1:nrow(manifesto)) : const_offset_man,
|
||||||
|
"const_rr_man" => isempty(const_rr_man) ? manifesto.rr_man : const_rr_man,
|
||||||
|
|
||||||
|
"N_const_exp_dim_total" => max(1, N_const_exp_dim_total),
|
||||||
|
"n_const_exp_dim" => isempty(n_const_exp_dim) ? ones(Int, nrow(expert_dim)) : n_const_exp_dim,
|
||||||
|
"const_offset_exp_dim" => isempty(const_offset_exp_dim) ? collect(1:nrow(expert_dim)) : const_offset_exp_dim,
|
||||||
|
"const_rr_exp_dim" => isempty(const_rr_exp_dim) ? expert_dim.rr_exp_dim : const_rr_exp_dim,
|
||||||
|
|
||||||
|
"N_const_exp_lr_total" => max(1, N_const_exp_lr_total),
|
||||||
|
"n_const_exp_lr" => isempty(n_const_exp_lr) ? ones(Int, nrow(expert_lr)) : n_const_exp_lr,
|
||||||
|
"const_offset_exp_lr" => isempty(const_offset_exp_lr) ? collect(1:nrow(expert_lr)) : const_offset_exp_lr,
|
||||||
|
"const_rr_exp_lr" => isempty(const_rr_exp_lr) ? expert_lr.rr_exp_lr : const_rr_exp_lr
|
||||||
|
)
|
||||||
|
|
||||||
|
println("4D Stan data dictionary created with $(length(dat_4dim)) elements")
|
||||||
|
|
||||||
|
# Print summary statistics
|
||||||
|
println("\nData summary (V10: Segment-based):")
|
||||||
|
println(" Segments: $(dat_4dim["S"])")
|
||||||
|
println(" Parties with valid segments: $(dat_4dim["J"])")
|
||||||
|
println(" Segment-year combinations: $(dat_4dim["R"])")
|
||||||
|
println(" Manifesto observations: $(dat_4dim["N_man"])")
|
||||||
|
println(" Expert dimension-specific observations: $(dat_4dim["N_exp_dim"])")
|
||||||
|
println(" Expert general L-R observations: $(dat_4dim["N_exp_lr"])")
|
||||||
|
println(" Unique manifesto items: $(dat_4dim["K_man"])")
|
||||||
|
println(" Unique expert dimension-specific items: $(dat_4dim["K_exp_dim"])")
|
||||||
|
println(" Unique expert general L-R items: $(dat_4dim["K_exp_lr"])")
|
||||||
|
println(" Years: $(dat_4dim["T_year"])")
|
||||||
|
|
||||||
|
return (dat_4dim=dat_4dim, manifesto=manifesto, expert_dim=expert_dim,
|
||||||
|
expert_lr=expert_lr, T_year=T_year, segment_year=segment_year,
|
||||||
|
segment_info=segment_info)
|
||||||
|
end
|
||||||
|
|
||||||
|
# Execute if run directly
|
||||||
|
if abspath(PROGRAM_FILE) == @__FILE__
|
||||||
|
println("Run from main script to execute the full 4D pipeline")
|
||||||
|
end
|
||||||
File diff suppressed because it is too large
Load Diff
@@ -0,0 +1,174 @@
|
|||||||
|
#!/usr/bin/env julia
|
||||||
|
#############################################################################
|
||||||
|
## 05_results_processing.jl
|
||||||
|
## Extract and process 4D model results with diagnostics
|
||||||
|
## Extract and process model results without election effects
|
||||||
|
#############################################################################
|
||||||
|
|
||||||
|
using StanSample, DataFrames, Statistics
|
||||||
|
|
||||||
|
function extract_model_results_4dim(stanmodel)
|
||||||
|
"""
|
||||||
|
Extract model results for the party-position model
|
||||||
|
Simplified version - no election effects (pure latent traits)
|
||||||
|
"""
|
||||||
|
println("Extracting 4D model results...")
|
||||||
|
|
||||||
|
try
|
||||||
|
println("Model completed successfully - extracting results")
|
||||||
|
|
||||||
|
# Save the full stanmodel object for downstream processing
|
||||||
|
# Post-estimation will extract specific parameters later
|
||||||
|
|
||||||
|
return (
|
||||||
|
samples = stanmodel, # Save the full stanmodel with all MCMC samples
|
||||||
|
extraction_status = "success"
|
||||||
|
)
|
||||||
|
|
||||||
|
catch e
|
||||||
|
println("Error in result extraction: $e")
|
||||||
|
return (
|
||||||
|
samples = "error",
|
||||||
|
extraction_status = "error",
|
||||||
|
error_message = string(e)
|
||||||
|
)
|
||||||
|
end
|
||||||
|
end
|
||||||
|
|
||||||
|
function compute_model_diagnostics(stanmodel_result)
|
||||||
|
"""
|
||||||
|
Compute convergence diagnostics from Stan model
|
||||||
|
Returns R-hat, ESS statistics, and overall convergence assessment
|
||||||
|
|
||||||
|
stanmodel_result can be either:
|
||||||
|
- A SampleModel object directly
|
||||||
|
- A named tuple from run_4dim_stan_model containing .stanmodel
|
||||||
|
"""
|
||||||
|
println("Computing model diagnostics...")
|
||||||
|
|
||||||
|
try
|
||||||
|
# Handle both direct SampleModel and named tuple from run_4dim_stan_model
|
||||||
|
stanmodel = if hasproperty(stanmodel_result, :stanmodel)
|
||||||
|
stanmodel_result.stanmodel
|
||||||
|
else
|
||||||
|
stanmodel_result
|
||||||
|
end
|
||||||
|
|
||||||
|
# Get REAL diagnostics using StanSample.jl
|
||||||
|
diagnostics_summary = read_summary(stanmodel)
|
||||||
|
|
||||||
|
# Extract real Rhat and ESS values. Stan summary column names differ
|
||||||
|
# across CmdStan/StanSample versions, so resolve aliases explicitly.
|
||||||
|
summary_names = names(diagnostics_summary)
|
||||||
|
rhat_col = if "r_hat" in summary_names
|
||||||
|
"r_hat"
|
||||||
|
elseif "R_hat" in summary_names
|
||||||
|
"R_hat"
|
||||||
|
elseif "RHat" in summary_names
|
||||||
|
"RHat"
|
||||||
|
else
|
||||||
|
error("No R-hat column found in summary. Columns: $(join(summary_names, ", "))")
|
||||||
|
end
|
||||||
|
ess_col = if "ess_bulk" in summary_names
|
||||||
|
"ess_bulk"
|
||||||
|
elseif "ess" in summary_names
|
||||||
|
"ess"
|
||||||
|
elseif "ESS_bulk" in summary_names
|
||||||
|
"ESS_bulk"
|
||||||
|
elseif "n_eff" in summary_names
|
||||||
|
"n_eff"
|
||||||
|
else
|
||||||
|
error("No ESS column found in summary. Columns: $(join(summary_names, ", "))")
|
||||||
|
end
|
||||||
|
|
||||||
|
rhat_vals = diagnostics_summary[!, rhat_col]
|
||||||
|
ess_bulk_vals = diagnostics_summary[!, ess_col]
|
||||||
|
|
||||||
|
# Compute real statistics (handle NaN values properly)
|
||||||
|
# Use isfinite to exclude both missing and NaN values
|
||||||
|
valid_rhat = filter(isfinite, rhat_vals)
|
||||||
|
valid_ess = filter(isfinite, ess_bulk_vals)
|
||||||
|
|
||||||
|
mean_rhat = length(valid_rhat) > 0 ? mean(valid_rhat) : NaN
|
||||||
|
max_rhat = length(valid_rhat) > 0 ? maximum(valid_rhat) : NaN
|
||||||
|
mean_ess = length(valid_ess) > 0 ? mean(valid_ess) : NaN
|
||||||
|
min_ess = length(valid_ess) > 0 ? minimum(valid_ess) : NaN
|
||||||
|
|
||||||
|
# Count problematic parameters (use isfinite for consistent counting)
|
||||||
|
high_rhat_count = count(x -> isfinite(x) && x > 1.1, rhat_vals)
|
||||||
|
moderate_rhat_count = count(x -> isfinite(x) && x > 1.05, rhat_vals)
|
||||||
|
low_ess_count = count(x -> isfinite(x) && x < 400, ess_bulk_vals)
|
||||||
|
very_low_ess_count = count(x -> isfinite(x) && x < 100, ess_bulk_vals)
|
||||||
|
|
||||||
|
# Total parameter count
|
||||||
|
total_params = length(valid_rhat)
|
||||||
|
|
||||||
|
# Overall assessment (handle NaN values)
|
||||||
|
if isnan(max_rhat) || total_params == 0
|
||||||
|
convergence_status = "insufficient_data"
|
||||||
|
else
|
||||||
|
excellent_convergence = max_rhat < 1.05 && high_rhat_count == 0 && very_low_ess_count == 0
|
||||||
|
good_convergence = max_rhat < 1.1 && high_rhat_count < 5 && very_low_ess_count < total_params * 0.1
|
||||||
|
acceptable_convergence = max_rhat < 1.2 && high_rhat_count < total_params * 0.1
|
||||||
|
|
||||||
|
if excellent_convergence
|
||||||
|
convergence_status = "excellent"
|
||||||
|
elseif good_convergence
|
||||||
|
convergence_status = "good"
|
||||||
|
elseif acceptable_convergence
|
||||||
|
convergence_status = "acceptable"
|
||||||
|
else
|
||||||
|
convergence_status = "poor"
|
||||||
|
end
|
||||||
|
end
|
||||||
|
|
||||||
|
println("\nDiagnostics computed:")
|
||||||
|
println(" Total parameters: $total_params")
|
||||||
|
println(" Mean R-hat: $(round(mean_rhat, digits=4))")
|
||||||
|
println(" Max R-hat: $(round(max_rhat, digits=4))")
|
||||||
|
println(" High R-hat count (>1.1): $high_rhat_count")
|
||||||
|
println(" Mean ESS: $(round(mean_ess, digits=0))")
|
||||||
|
println(" Min ESS: $(round(min_ess, digits=0))")
|
||||||
|
println(" Very low ESS count (<100): $very_low_ess_count")
|
||||||
|
println(" Convergence status: $convergence_status")
|
||||||
|
|
||||||
|
return (
|
||||||
|
diagnostics_summary = diagnostics_summary,
|
||||||
|
convergence_status = convergence_status,
|
||||||
|
mean_rhat = mean_rhat,
|
||||||
|
max_rhat = max_rhat,
|
||||||
|
mean_ess = mean_ess,
|
||||||
|
min_ess = min_ess,
|
||||||
|
high_rhat_count = high_rhat_count,
|
||||||
|
moderate_rhat_count = moderate_rhat_count,
|
||||||
|
low_ess_count = low_ess_count,
|
||||||
|
very_low_ess_count = very_low_ess_count,
|
||||||
|
total_params = total_params
|
||||||
|
)
|
||||||
|
|
||||||
|
catch e
|
||||||
|
println("Error in diagnostics computation: $e")
|
||||||
|
println("Stack trace:")
|
||||||
|
showerror(stdout, e, catch_backtrace())
|
||||||
|
|
||||||
|
return (
|
||||||
|
diagnostics_summary = "error",
|
||||||
|
convergence_status = "error",
|
||||||
|
mean_rhat = 999.0,
|
||||||
|
max_rhat = 999.0,
|
||||||
|
mean_ess = 0.0,
|
||||||
|
min_ess = 0.0,
|
||||||
|
high_rhat_count = 999,
|
||||||
|
moderate_rhat_count = 999,
|
||||||
|
low_ess_count = 999,
|
||||||
|
very_low_ess_count = 999,
|
||||||
|
total_params = 0,
|
||||||
|
error_message = string(e)
|
||||||
|
)
|
||||||
|
end
|
||||||
|
end
|
||||||
|
|
||||||
|
# Execute if run directly
|
||||||
|
if abspath(PROGRAM_FILE) == @__FILE__
|
||||||
|
println("Run from main run_model.jl to execute the full pipeline")
|
||||||
|
end
|
||||||
@@ -0,0 +1,428 @@
|
|||||||
|
#!/usr/bin/env julia
|
||||||
|
#############################################################################
|
||||||
|
## 06_save_model.jl
|
||||||
|
## CSV-First Save Architecture
|
||||||
|
## Robust, portable, no serialization issues
|
||||||
|
#############################################################################
|
||||||
|
|
||||||
|
module RobustSave
|
||||||
|
|
||||||
|
using Dates, Printf, CSV, DataFrames, JSON
|
||||||
|
|
||||||
|
export robust_save_model_csv, robust_save_model
|
||||||
|
|
||||||
|
"""
|
||||||
|
CSV-First Save Architecture
|
||||||
|
|
||||||
|
When chains_already_saved=true (BULLETPROOF MODE):
|
||||||
|
- Chains already saved to run_dir/chains/ by model execution
|
||||||
|
- Just verify chains and add metadata/data files
|
||||||
|
- Even if this function crashes, chains are SAFE
|
||||||
|
|
||||||
|
When chains_already_saved=false (legacy mode):
|
||||||
|
- Copy chains from temp directory to new run directory
|
||||||
|
- Add metadata/data files
|
||||||
|
|
||||||
|
Saves model results as:
|
||||||
|
1. CSV chain files (source of truth - never fail)
|
||||||
|
2. Data CSVs (original inputs for reproducibility)
|
||||||
|
3. Simple metadata.json (no complex types)
|
||||||
|
4. Human-readable README.txt
|
||||||
|
|
||||||
|
No JLD2, no serialization issues, fully portable and reproducible.
|
||||||
|
"""
|
||||||
|
function robust_save_model_csv(
|
||||||
|
run_dir_or_temp::String,
|
||||||
|
data_dict::Dict,
|
||||||
|
original_data::Dict,
|
||||||
|
metadata::Dict;
|
||||||
|
chains_already_saved::Bool=false
|
||||||
|
)
|
||||||
|
println("\n" * "=" ^ 70)
|
||||||
|
if chains_already_saved
|
||||||
|
println("ADDING METADATA TO EXISTING RUN (chains already secured)")
|
||||||
|
else
|
||||||
|
println("CSV-FIRST MODEL SAVE")
|
||||||
|
end
|
||||||
|
println("=" ^ 70)
|
||||||
|
|
||||||
|
# Determine directories based on mode
|
||||||
|
if chains_already_saved
|
||||||
|
# Chains already saved - run_dir_or_temp IS the run directory
|
||||||
|
run_dir = run_dir_or_temp
|
||||||
|
chains_dir = joinpath(run_dir, "chains")
|
||||||
|
data_dir = joinpath(run_dir, "data")
|
||||||
|
run_id = basename(run_dir)
|
||||||
|
timestamp = replace(run_id, "run_" => "")
|
||||||
|
|
||||||
|
println("Run directory: $run_dir")
|
||||||
|
println("Mode: Chains already secured, adding metadata")
|
||||||
|
|
||||||
|
# Verify chains directory exists
|
||||||
|
if !isdir(chains_dir)
|
||||||
|
error("CRITICAL: Chains directory not found: $chains_dir")
|
||||||
|
end
|
||||||
|
|
||||||
|
# Count existing chain files
|
||||||
|
chain_files = filter(f -> endswith(f, ".csv") && contains(f, "chain"), readdir(chains_dir))
|
||||||
|
if isempty(chain_files)
|
||||||
|
error("CRITICAL: No chain CSV files found in $chains_dir")
|
||||||
|
end
|
||||||
|
|
||||||
|
println("Found $(length(chain_files)) chain files already saved")
|
||||||
|
|
||||||
|
# Calculate total size
|
||||||
|
total_size_gb = 0.0
|
||||||
|
for chain_file in chain_files
|
||||||
|
chain_path = joinpath(chains_dir, chain_file)
|
||||||
|
total_size_gb += filesize(chain_path) / (1024^3)
|
||||||
|
end
|
||||||
|
|
||||||
|
println("✓ Chains verified ($(round(total_size_gb, digits=2)) GB total)")
|
||||||
|
|
||||||
|
else
|
||||||
|
# Legacy mode - create new run directory and copy chains
|
||||||
|
temp_csv_dir = run_dir_or_temp
|
||||||
|
|
||||||
|
timestamp = Dates.format(Dates.now(), "yyyy-mm-dd_HH-MM-SS")
|
||||||
|
run_id = "run_$(timestamp)"
|
||||||
|
run_dir = joinpath("outputs", "model_outputs", "latest", run_id)
|
||||||
|
chains_dir = joinpath(run_dir, "chains")
|
||||||
|
data_dir = joinpath(run_dir, "data")
|
||||||
|
|
||||||
|
println("Run ID: $run_id")
|
||||||
|
println("Output directory: $run_dir")
|
||||||
|
|
||||||
|
# Create directory structure
|
||||||
|
mkpath(chains_dir)
|
||||||
|
|
||||||
|
# STEP 1: Copy CSV chain files
|
||||||
|
println("\n" * "=" ^ 70)
|
||||||
|
println("STEP 1: Copying MCMC chain CSV files")
|
||||||
|
println("=" ^ 70)
|
||||||
|
println("Source: $temp_csv_dir")
|
||||||
|
println("Destination: $chains_dir")
|
||||||
|
|
||||||
|
csv_files = filter(f -> endswith(f, ".csv"), readdir(temp_csv_dir))
|
||||||
|
chain_files = filter(f -> contains(f, "chain"), csv_files)
|
||||||
|
|
||||||
|
if isempty(chain_files)
|
||||||
|
error("No chain CSV files found in $temp_csv_dir")
|
||||||
|
end
|
||||||
|
|
||||||
|
println("Found $(length(chain_files)) chain files")
|
||||||
|
|
||||||
|
total_size_gb = 0.0
|
||||||
|
for (i, csv_file) in enumerate(sort(chain_files))
|
||||||
|
src_path = joinpath(temp_csv_dir, csv_file)
|
||||||
|
|
||||||
|
# Rename to standard format: chain_1.csv, chain_2.csv, etc.
|
||||||
|
dest_filename = "chain_$i.csv"
|
||||||
|
dest_path = joinpath(chains_dir, dest_filename)
|
||||||
|
|
||||||
|
src_size = filesize(src_path)
|
||||||
|
size_gb = src_size / (1024^3)
|
||||||
|
total_size_gb += size_gb
|
||||||
|
|
||||||
|
println(" Copying $csv_file → $dest_filename ($(round(size_gb, digits=2)) GB)")
|
||||||
|
cp(src_path, dest_path, force=true)
|
||||||
|
|
||||||
|
# Verify copy with size check
|
||||||
|
dest_size = filesize(dest_path)
|
||||||
|
if dest_size != src_size
|
||||||
|
error("CRITICAL: Size mismatch for $dest_filename! Source: $src_size, Dest: $dest_size")
|
||||||
|
end
|
||||||
|
end
|
||||||
|
|
||||||
|
println("✓ All chains copied and verified ($(round(total_size_gb, digits=2)) GB total)")
|
||||||
|
end
|
||||||
|
|
||||||
|
# Create data directory
|
||||||
|
mkpath(data_dir)
|
||||||
|
|
||||||
|
# STEP 2: Save data CSVs
|
||||||
|
println("\n" * "=" ^ 70)
|
||||||
|
println("STEP 2: Saving original data CSVs")
|
||||||
|
println("=" ^ 70)
|
||||||
|
|
||||||
|
for (name, df) in original_data
|
||||||
|
if isa(df, DataFrame)
|
||||||
|
csv_path = joinpath(data_dir, "$(name).csv")
|
||||||
|
println(" Saving $(name).csv ($(nrow(df)) rows)")
|
||||||
|
CSV.write(csv_path, df)
|
||||||
|
end
|
||||||
|
end
|
||||||
|
|
||||||
|
println("✓ Data CSVs saved")
|
||||||
|
|
||||||
|
# STEP 3: Save Stan data dictionary as JSON
|
||||||
|
println("\n" * "=" ^ 70)
|
||||||
|
println("STEP 3: Saving Stan data dictionary")
|
||||||
|
println("=" ^ 70)
|
||||||
|
|
||||||
|
# Convert data_dict to JSON-serializable format
|
||||||
|
stan_data_json = Dict{String, Any}()
|
||||||
|
for (k, v) in data_dict
|
||||||
|
try
|
||||||
|
# Only save simple types (numbers, arrays of numbers)
|
||||||
|
if isa(v, Number) || isa(v, AbstractArray{<:Number})
|
||||||
|
stan_data_json[k] = v
|
||||||
|
elseif isa(v, AbstractArray)
|
||||||
|
# Try to convert, skip if fails
|
||||||
|
try
|
||||||
|
stan_data_json[k] = collect(v)
|
||||||
|
catch
|
||||||
|
println(" Skipping $k (complex type)")
|
||||||
|
end
|
||||||
|
end
|
||||||
|
catch e
|
||||||
|
println(" Warning: Could not serialize $k: $e")
|
||||||
|
end
|
||||||
|
end
|
||||||
|
|
||||||
|
stan_data_path = joinpath(data_dir, "stan_data.json")
|
||||||
|
open(stan_data_path, "w") do f
|
||||||
|
JSON.print(f, stan_data_json, 2)
|
||||||
|
end
|
||||||
|
println("✓ Stan data saved to stan_data.json")
|
||||||
|
|
||||||
|
# Count chain files for metadata
|
||||||
|
chain_files_final = filter(f -> endswith(f, ".csv") && contains(f, "chain"), readdir(chains_dir))
|
||||||
|
num_chains = length(chain_files_final)
|
||||||
|
|
||||||
|
# Recalculate total_size_gb if in chains_already_saved mode
|
||||||
|
if chains_already_saved
|
||||||
|
total_size_gb = 0.0
|
||||||
|
for chain_file in chain_files_final
|
||||||
|
chain_path = joinpath(chains_dir, chain_file)
|
||||||
|
total_size_gb += filesize(chain_path) / (1024^3)
|
||||||
|
end
|
||||||
|
end
|
||||||
|
|
||||||
|
# STEP 4: Save metadata
|
||||||
|
println("\n" * "=" ^ 70)
|
||||||
|
println("STEP 4: Saving metadata")
|
||||||
|
println("=" ^ 70)
|
||||||
|
|
||||||
|
# Add run info to metadata
|
||||||
|
metadata["run_id"] = run_id
|
||||||
|
metadata["timestamp"] = timestamp
|
||||||
|
metadata["files"] = Dict(
|
||||||
|
"chains" => ["chains/chain_$i.csv" for i in 1:num_chains],
|
||||||
|
"data" => readdir(data_dir),
|
||||||
|
"chain_size_gb" => num_chains > 0 ? round(total_size_gb / num_chains, digits=2) : 0.0,
|
||||||
|
"total_size_gb" => round(total_size_gb, digits=2)
|
||||||
|
)
|
||||||
|
|
||||||
|
metadata_path = joinpath(run_dir, "metadata.json")
|
||||||
|
open(metadata_path, "w") do f
|
||||||
|
JSON.print(f, metadata, 2)
|
||||||
|
end
|
||||||
|
println("✓ Metadata saved to metadata.json")
|
||||||
|
|
||||||
|
# STEP 5: Generate README
|
||||||
|
println("\n" * "=" ^ 70)
|
||||||
|
println("STEP 5: Generating README")
|
||||||
|
println("=" ^ 70)
|
||||||
|
|
||||||
|
readme_path = joinpath(run_dir, "README.txt")
|
||||||
|
generate_readme(readme_path, run_id, metadata, num_chains, total_size_gb)
|
||||||
|
println("✓ README generated")
|
||||||
|
|
||||||
|
# STEP 6: Final verification
|
||||||
|
println("\n" * "=" ^ 70)
|
||||||
|
println("STEP 6: Verification")
|
||||||
|
println("=" ^ 70)
|
||||||
|
|
||||||
|
# Verify all chain files exist and are readable
|
||||||
|
all_good = true
|
||||||
|
verified_files = Dict{String, Dict{String, Any}}()
|
||||||
|
|
||||||
|
for i in 1:num_chains
|
||||||
|
chain_path = joinpath(chains_dir, "chain_$i.csv")
|
||||||
|
if !isfile(chain_path)
|
||||||
|
println(" ✗ Missing: chain_$i.csv")
|
||||||
|
all_good = false
|
||||||
|
else
|
||||||
|
# Quick read test and size check
|
||||||
|
try
|
||||||
|
CSV.File(chain_path; limit=1)
|
||||||
|
file_size_gb = filesize(chain_path) / (1024^3)
|
||||||
|
verified_files["chain_$i.csv"] = Dict(
|
||||||
|
"path" => chain_path,
|
||||||
|
"size_gb" => file_size_gb,
|
||||||
|
"verified" => true
|
||||||
|
)
|
||||||
|
println(" ✓ chain_$i.csv verified ($(round(file_size_gb, digits=2)) GB)")
|
||||||
|
catch e
|
||||||
|
println(" ✗ Cannot read chain_$i.csv: $e")
|
||||||
|
all_good = false
|
||||||
|
end
|
||||||
|
end
|
||||||
|
end
|
||||||
|
|
||||||
|
if !all_good
|
||||||
|
error("Verification failed - some files are missing or corrupted")
|
||||||
|
end
|
||||||
|
|
||||||
|
println("\n" * "=" ^ 70)
|
||||||
|
println("✓ MODEL SAVED SUCCESSFULLY")
|
||||||
|
println("=" ^ 70)
|
||||||
|
println("Run directory: $run_dir")
|
||||||
|
println("Total size: $(round(total_size_gb, digits=2)) GB")
|
||||||
|
println("Status: All files verified and ready")
|
||||||
|
println("=" ^ 70)
|
||||||
|
|
||||||
|
# Return verification details for cleanup
|
||||||
|
return (
|
||||||
|
run_dir = run_dir,
|
||||||
|
verified_files = verified_files,
|
||||||
|
total_size_gb = total_size_gb,
|
||||||
|
verification_passed = all_good,
|
||||||
|
num_chains = num_chains
|
||||||
|
)
|
||||||
|
end
|
||||||
|
|
||||||
|
function generate_readme(
|
||||||
|
filepath::String,
|
||||||
|
run_id::String,
|
||||||
|
metadata::Dict,
|
||||||
|
num_chains::Int,
|
||||||
|
total_size_gb::Float64
|
||||||
|
)
|
||||||
|
"""Generate human-readable README file"""
|
||||||
|
|
||||||
|
open(filepath, "w") do f
|
||||||
|
write(f, "=" ^ 78 * "\n")
|
||||||
|
write(f, "PARTY-POSITION MODEL - MODEL RUN RESULTS\n")
|
||||||
|
write(f, "=" ^ 78 * "\n\n")
|
||||||
|
|
||||||
|
write(f, "Run ID: $run_id\n")
|
||||||
|
write(f, "Model: $(get(metadata, "model_file", "unknown"))\n")
|
||||||
|
write(f, "Date: $(Dates.format(Dates.now(), "yyyy-mm-dd HH:MM:SS"))\n")
|
||||||
|
write(f, "Status: $(get(metadata, "convergence_status", "unknown"))\n\n")
|
||||||
|
|
||||||
|
write(f, "=" ^ 78 * "\n")
|
||||||
|
write(f, "DIRECTORY CONTENTS\n")
|
||||||
|
write(f, "=" ^ 78 * "\n\n")
|
||||||
|
|
||||||
|
write(f, "chains/\n")
|
||||||
|
for i in 1:num_chains
|
||||||
|
write(f, " ├── chain_$i.csv\n")
|
||||||
|
end
|
||||||
|
write(f, " Total: $(get(metadata, "num_chains", num_chains)) chains × " *
|
||||||
|
"$(get(metadata, "num_samples", "?")) samples\n")
|
||||||
|
write(f, " Size: $(round(total_size_gb, digits=2)) GB\n\n")
|
||||||
|
|
||||||
|
write(f, "data/\n")
|
||||||
|
write(f, " ├── text_data.csv\n")
|
||||||
|
write(f, " ├── expert_dim.csv\n")
|
||||||
|
write(f, " ├── expert_lr.csv\n")
|
||||||
|
write(f, " ├── segment_year_map.csv (V10)\n")
|
||||||
|
write(f, " ├── segment_info.csv (V10)\n")
|
||||||
|
write(f, " └── stan_data.json\n\n")
|
||||||
|
|
||||||
|
write(f, "=" ^ 78 * "\n")
|
||||||
|
write(f, "MODEL CONFIGURATION\n")
|
||||||
|
write(f, "=" ^ 78 * "\n\n")
|
||||||
|
|
||||||
|
write(f, "Chains: $(get(metadata, "num_chains", "?"))\n")
|
||||||
|
write(f, "Warmup: $(get(metadata, "num_warmup", "?"))\n")
|
||||||
|
write(f, "Samples: $(get(metadata, "num_samples", "?"))\n")
|
||||||
|
write(f, "Adapt delta: $(get(metadata, "adapt_delta", "?"))\n")
|
||||||
|
write(f, "Max depth: $(get(metadata, "max_depth", "?"))\n\n")
|
||||||
|
|
||||||
|
write(f, "Dimensions: $(join(get(metadata, "dimensions", ["?"]), ", "))\n\n")
|
||||||
|
|
||||||
|
write(f, "=" ^ 78 * "\n")
|
||||||
|
write(f, "HOW TO USE THESE RESULTS\n")
|
||||||
|
write(f, "=" ^ 78 * "\n\n")
|
||||||
|
|
||||||
|
write(f, "To extract party positions:\n\n")
|
||||||
|
write(f, " julia 02_post_estimation.jl\n\n")
|
||||||
|
|
||||||
|
write(f, "This will read the CSV files and generate party_positions_[timestamp].csv\n")
|
||||||
|
write(f, "with uncertainty estimates (SE, credible intervals).\n\n")
|
||||||
|
|
||||||
|
write(f, "=" ^ 78 * "\n")
|
||||||
|
write(f, "Generated: $(Dates.format(Dates.now(), "yyyy-mm-dd HH:MM:SS"))\n")
|
||||||
|
write(f, "=" ^ 78 * "\n")
|
||||||
|
end
|
||||||
|
end
|
||||||
|
|
||||||
|
"""
|
||||||
|
Wrapper for robust_save_model_csv that handles the stanmodel tuple from run_4dim_stan_model.
|
||||||
|
|
||||||
|
The model execution now saves chains BEFORE returning, so this function:
|
||||||
|
1. Verifies chains are already saved in final_run_dir
|
||||||
|
2. Adds metadata and data files
|
||||||
|
3. Returns the output path
|
||||||
|
|
||||||
|
Arguments:
|
||||||
|
- stanmodel_tuple: Named tuple from run_4dim_stan_model (contains .stanmodel, .final_run_dir, etc.)
|
||||||
|
- model_data: Dict with data_dict, original data, and metadata
|
||||||
|
- output_dir: Base output directory (ignored - uses stanmodel_tuple.final_run_dir)
|
||||||
|
- compress: Ignored (CSV-first architecture)
|
||||||
|
- keep_local_backups: Ignored (chains already saved)
|
||||||
|
"""
|
||||||
|
function robust_save_model(
|
||||||
|
stanmodel_tuple,
|
||||||
|
model_data::Dict,
|
||||||
|
output_dir::String;
|
||||||
|
compress::Bool=true,
|
||||||
|
keep_local_backups::Int=2
|
||||||
|
)
|
||||||
|
# Extract the final run directory from the stanmodel tuple
|
||||||
|
if !hasproperty(stanmodel_tuple, :final_run_dir)
|
||||||
|
error("stanmodel_tuple missing :final_run_dir - chains may not be saved!")
|
||||||
|
end
|
||||||
|
|
||||||
|
final_run_dir = stanmodel_tuple.final_run_dir
|
||||||
|
|
||||||
|
# Prepare original data for saving
|
||||||
|
original_data = Dict{String, Any}()
|
||||||
|
if haskey(model_data, "manifesto")
|
||||||
|
original_data["text_data"] = model_data["manifesto"]
|
||||||
|
end
|
||||||
|
if haskey(model_data, "expert_dim")
|
||||||
|
original_data["expert_dim"] = model_data["expert_dim"]
|
||||||
|
end
|
||||||
|
if haskey(model_data, "expert_lr")
|
||||||
|
original_data["expert_lr"] = model_data["expert_lr"]
|
||||||
|
end
|
||||||
|
# V10: Save segment_year_map and segment_info for post-estimation
|
||||||
|
if haskey(model_data, "segment_year")
|
||||||
|
original_data["segment_year_map"] = model_data["segment_year"]
|
||||||
|
end
|
||||||
|
if haskey(model_data, "segment_info")
|
||||||
|
original_data["segment_info"] = model_data["segment_info"]
|
||||||
|
end
|
||||||
|
# V9 fallback: Save party_year_map for post-estimation (includes interpolated years)
|
||||||
|
if haskey(model_data, "party_year")
|
||||||
|
original_data["party_year_map"] = model_data["party_year"]
|
||||||
|
end
|
||||||
|
|
||||||
|
# Prepare metadata
|
||||||
|
metadata = Dict{String, Any}()
|
||||||
|
if haskey(model_data, "model_info")
|
||||||
|
for (k, v) in model_data["model_info"]
|
||||||
|
metadata[string(k)] = v
|
||||||
|
end
|
||||||
|
end
|
||||||
|
|
||||||
|
# Get data_dict
|
||||||
|
data_dict = get(model_data, "data_dict", Dict{String, Any}())
|
||||||
|
|
||||||
|
# Call the CSV-first save with chains_already_saved=true
|
||||||
|
result = robust_save_model_csv(
|
||||||
|
final_run_dir,
|
||||||
|
data_dict,
|
||||||
|
original_data,
|
||||||
|
metadata;
|
||||||
|
chains_already_saved=true
|
||||||
|
)
|
||||||
|
|
||||||
|
return result.run_dir
|
||||||
|
end
|
||||||
|
|
||||||
|
end # module
|
||||||
@@ -0,0 +1,150 @@
|
|||||||
|
#!/usr/bin/env julia
|
||||||
|
#############################################################################
|
||||||
|
## performance_monitoring.jl
|
||||||
|
## Post-run performance diagnostics for Stan sampling jobs
|
||||||
|
#############################################################################
|
||||||
|
|
||||||
|
using StanSample
|
||||||
|
using Statistics: mean, median
|
||||||
|
using Dates
|
||||||
|
|
||||||
|
function _safe_column(df, candidates::Vector{String})
|
||||||
|
for candidate in candidates
|
||||||
|
sym = Symbol(candidate)
|
||||||
|
if sym in names(df)
|
||||||
|
return df[!, sym]
|
||||||
|
elseif candidate in names(df)
|
||||||
|
return df[!, candidate]
|
||||||
|
end
|
||||||
|
end
|
||||||
|
return nothing
|
||||||
|
end
|
||||||
|
|
||||||
|
function _clean_values(vec)
|
||||||
|
cleaned = Float64[]
|
||||||
|
for v in vec
|
||||||
|
if v isa Missing || v === nothing
|
||||||
|
continue
|
||||||
|
end
|
||||||
|
try
|
||||||
|
value = Float64(v)
|
||||||
|
isfinite(value) && push!(cleaned, value)
|
||||||
|
catch
|
||||||
|
continue
|
||||||
|
end
|
||||||
|
end
|
||||||
|
return cleaned
|
||||||
|
end
|
||||||
|
|
||||||
|
function _summarize_vector(vec)
|
||||||
|
if vec === nothing
|
||||||
|
return Dict{String,Any}("available" => false)
|
||||||
|
end
|
||||||
|
cleaned = _clean_values(vec)
|
||||||
|
if isempty(cleaned)
|
||||||
|
return Dict{String,Any}("available" => false)
|
||||||
|
end
|
||||||
|
return Dict{String,Any}(
|
||||||
|
"available" => true,
|
||||||
|
"count" => length(cleaned),
|
||||||
|
"mean" => mean(cleaned),
|
||||||
|
"median" => median(cleaned),
|
||||||
|
"min" => minimum(cleaned),
|
||||||
|
"max" => maximum(cleaned)
|
||||||
|
)
|
||||||
|
end
|
||||||
|
|
||||||
|
function monitor_sampling_performance!(stanmodel;
|
||||||
|
run_metrics::Union{Nothing,Dict{String,Any}}=nothing,
|
||||||
|
metrics_path::Union{Nothing,String}=nothing,
|
||||||
|
csv_paths::Union{Nothing,Vector{String}}=nothing,
|
||||||
|
aggregate_metrics::Union{Nothing,Dict{String,Any}}=nothing,
|
||||||
|
max_depth::Union{Nothing,Int}=nothing)
|
||||||
|
|
||||||
|
csv_paths === nothing && (csv_paths = discover_stan_csvs([stanmodel.tmpdir]))
|
||||||
|
aggregate_metrics === nothing && begin
|
||||||
|
_, aggregate_metrics = collect_run_metrics(csv_paths; max_depth=max_depth)
|
||||||
|
end
|
||||||
|
|
||||||
|
summary_df = nothing
|
||||||
|
try
|
||||||
|
summary_df = read_summary(stanmodel)
|
||||||
|
catch e
|
||||||
|
println("Warning: could not read Stan summary: $e")
|
||||||
|
end
|
||||||
|
|
||||||
|
performance = Dict{String,Any}(
|
||||||
|
"generated_at" => Dates.format(Dates.now(), "yyyy-mm-ddTHH:MM:SS"),
|
||||||
|
"csv_paths" => csv_paths
|
||||||
|
)
|
||||||
|
|
||||||
|
if summary_df !== nothing
|
||||||
|
performance["ess_bulk"] = _summarize_vector(_safe_column(summary_df, ["ess_bulk", "ess"]))
|
||||||
|
performance["ess_tail"] = _summarize_vector(_safe_column(summary_df, ["ess_tail"]))
|
||||||
|
performance["ess_per_sec"] = _summarize_vector(_safe_column(summary_df, ["ess_per_sec", "n_eff/s"]))
|
||||||
|
performance["r_hat"] = _summarize_vector(_safe_column(summary_df, ["r_hat"]))
|
||||||
|
|
||||||
|
if haskey(performance["r_hat"], "available") && performance["r_hat"]["available"]
|
||||||
|
performance["r_hat"]["max"] = maximum(_clean_values(_safe_column(summary_df, ["r_hat"])))
|
||||||
|
end
|
||||||
|
|
||||||
|
performance["parameters_considered"] = size(summary_df, 1)
|
||||||
|
else
|
||||||
|
performance["ess_bulk"] = Dict{String,Any}("available" => false)
|
||||||
|
performance["ess_tail"] = Dict{String,Any}("available" => false)
|
||||||
|
performance["ess_per_sec"] = Dict{String,Any}("available" => false)
|
||||||
|
performance["r_hat"] = Dict{String,Any}("available" => false)
|
||||||
|
performance["parameters_considered"] = 0
|
||||||
|
end
|
||||||
|
|
||||||
|
divergences = get(aggregate_metrics, "divergences", 0)
|
||||||
|
total_samples = stanmodel.num_samples * stanmodel.num_chains
|
||||||
|
divergence_rate = total_samples > 0 ? divergences / total_samples : nothing
|
||||||
|
|
||||||
|
performance["divergences"] = Dict{String,Any}(
|
||||||
|
"count" => divergences,
|
||||||
|
"rate" => divergence_rate,
|
||||||
|
"total_draws" => total_samples
|
||||||
|
)
|
||||||
|
|
||||||
|
performance["leapfrog"] = Dict{String,Any}(
|
||||||
|
"mean" => get(aggregate_metrics, "mean_leapfrog", nothing)
|
||||||
|
)
|
||||||
|
|
||||||
|
performance["step_size"] = Dict{String,Any}(
|
||||||
|
"mean" => get(aggregate_metrics, "mean_step_size", nothing)
|
||||||
|
)
|
||||||
|
|
||||||
|
sampling_seconds = get(aggregate_metrics, "sampling_seconds", nothing)
|
||||||
|
if sampling_seconds !== nothing && sampling_seconds > 0
|
||||||
|
performance["throughput"] = Dict{String,Any}(
|
||||||
|
"samples_per_second" => (total_samples / sampling_seconds),
|
||||||
|
"seconds_sampling" => sampling_seconds
|
||||||
|
)
|
||||||
|
else
|
||||||
|
performance["throughput"] = Dict{String,Any}(
|
||||||
|
"samples_per_second" => nothing,
|
||||||
|
"seconds_sampling" => sampling_seconds
|
||||||
|
)
|
||||||
|
end
|
||||||
|
|
||||||
|
if run_metrics !== nothing
|
||||||
|
run_metrics["performance"] = performance
|
||||||
|
if metrics_path !== nothing
|
||||||
|
safe_write_json(metrics_path, run_metrics)
|
||||||
|
end
|
||||||
|
elseif metrics_path !== nothing
|
||||||
|
temp_metrics = Dict{String,Any}("performance" => performance)
|
||||||
|
safe_write_json(metrics_path, temp_metrics)
|
||||||
|
end
|
||||||
|
|
||||||
|
println("\nPERFORMANCE SUMMARY")
|
||||||
|
println(" ESS bulk (mean): $(get(performance["ess_bulk"], "mean", "n/a"))")
|
||||||
|
println(" ESS/sec (mean): $(get(performance["ess_per_sec"], "mean", "n/a"))")
|
||||||
|
println(" Divergences: $(divergences)")
|
||||||
|
println(" Divergence rate: $(divergence_rate === nothing ? "n/a" : round(divergence_rate, digits=6))")
|
||||||
|
println(" Mean leapfrog steps: $(get(performance["leapfrog"], "mean", "n/a"))")
|
||||||
|
|
||||||
|
return performance
|
||||||
|
end
|
||||||
|
|
||||||
@@ -0,0 +1,450 @@
|
|||||||
|
#!/usr/bin/env julia
|
||||||
|
#############################################################################
|
||||||
|
## validate_construct.jl
|
||||||
|
## Construct validity: Party family ordering and temporal stability
|
||||||
|
##
|
||||||
|
## Following Claassen (2019), this script validates:
|
||||||
|
## 1. Party family ordering: Do family means follow theoretically expected orderings?
|
||||||
|
## 2. Temporal stability: Flag parties with implausible position changes
|
||||||
|
##
|
||||||
|
## Uses ParlGov party family classifications (Döring & Manow 2024) via PartyFacts IDs.
|
||||||
|
#############################################################################
|
||||||
|
|
||||||
|
using CSV, DataFrames, Statistics, StatsBase, Dates, Printf
|
||||||
|
|
||||||
|
# Family code → display name mapping
|
||||||
|
const FAMILY_DISPLAY_NAMES = Dict(
|
||||||
|
"com" => "Communist/Far Left",
|
||||||
|
"eco" => "Green/Ecological",
|
||||||
|
"soc" => "Social Democratic",
|
||||||
|
"lib" => "Liberal",
|
||||||
|
"chr" => "Christian Democratic",
|
||||||
|
"con" => "Conservative",
|
||||||
|
"right" => "Radical Right"
|
||||||
|
)
|
||||||
|
|
||||||
|
# Substantive families (drop Specialist, Other, Agrarian — heterogeneous or ambiguous)
|
||||||
|
const SUBSTANTIVE_FAMILIES = Set(["com", "eco", "soc", "lib", "chr", "con", "right"])
|
||||||
|
|
||||||
|
# Expected orderings (theoretically motivated)
|
||||||
|
# Economic: Communist < Social Democratic < Green < Christian Democratic < Conservative
|
||||||
|
# (5-family core — Liberal position is ambiguous cross-nationally)
|
||||||
|
const EXPECTED_ECONOMIC_ORDER = ["com", "soc", "eco", "chr", "con"]
|
||||||
|
|
||||||
|
# Cultural: Green < Liberal < Social Democratic < Christian Democratic < Conservative < Radical Right
|
||||||
|
const EXPECTED_GALTAN_ORDER = ["eco", "lib", "soc", "chr", "con", "right"]
|
||||||
|
|
||||||
|
function load_model_output(base_dir::String=".")
|
||||||
|
"""Load the most recent 2D model party positions output"""
|
||||||
|
|
||||||
|
position_files = filter(f -> startswith(f, "party_positions_") && endswith(f, ".csv") &&
|
||||||
|
!endswith(f, "_metadata.txt") && !endswith(f, "_tables.tex"), readdir(base_dir))
|
||||||
|
legacy_files = filter(f -> startswith(f, "party_positions_v1_") && endswith(f, ".csv"), readdir(base_dir))
|
||||||
|
append!(position_files, legacy_files)
|
||||||
|
|
||||||
|
if !isempty(position_files)
|
||||||
|
latest = sort(position_files)[end]
|
||||||
|
println("Loading model output: $latest")
|
||||||
|
return CSV.read(joinpath(base_dir, latest), DataFrame), latest
|
||||||
|
end
|
||||||
|
|
||||||
|
# Check output estimations directory
|
||||||
|
est_dir = joinpath(base_dir, "outputs", "estimations", "latest")
|
||||||
|
if isdir(est_dir)
|
||||||
|
est_files = filter(f -> startswith(f, "party_positions_") && endswith(f, ".csv") &&
|
||||||
|
!endswith(f, "_metadata.txt") && !endswith(f, "_tables.tex"), readdir(est_dir))
|
||||||
|
if !isempty(est_files)
|
||||||
|
latest = sort(est_files)[end]
|
||||||
|
println("Loading model output: outputs/estimations/latest/$latest")
|
||||||
|
return CSV.read(joinpath(est_dir, latest), DataFrame), latest
|
||||||
|
end
|
||||||
|
end
|
||||||
|
|
||||||
|
error("No party_positions_*.csv found. Run 02_post_estimation.jl first.")
|
||||||
|
end
|
||||||
|
|
||||||
|
function validate_party_families(model::DataFrame)
|
||||||
|
"""Check whether party family means follow theoretically expected orderings"""
|
||||||
|
|
||||||
|
println("\n" * "="^60)
|
||||||
|
println("PARTY FAMILY ORDERING VALIDATION")
|
||||||
|
println("="^60)
|
||||||
|
println("\nUsing ParlGov family classifications (Döring & Manow 2024)")
|
||||||
|
println()
|
||||||
|
|
||||||
|
party_col = hasproperty(model, :party_id) ? :party_id : :party
|
||||||
|
|
||||||
|
# Load party families
|
||||||
|
families_df = CSV.read("data/party_families.csv", DataFrame)
|
||||||
|
|
||||||
|
# Join to model output
|
||||||
|
model_with_families = innerjoin(model, families_df, on=party_col => :partyfacts_id)
|
||||||
|
|
||||||
|
# Filter to substantive families
|
||||||
|
filter!(r -> r.family in SUBSTANTIVE_FAMILIES, model_with_families)
|
||||||
|
|
||||||
|
n_parties = length(unique(model_with_families[!, party_col]))
|
||||||
|
n_obs = nrow(model_with_families)
|
||||||
|
println(" Matched $n_parties parties ($n_obs party-years) across $(length(SUBSTANTIVE_FAMILIES)) families")
|
||||||
|
println()
|
||||||
|
|
||||||
|
# Compute family means
|
||||||
|
family_stats = combine(groupby(model_with_families, :family)) do df
|
||||||
|
DataFrame(
|
||||||
|
n_parties = length(unique(df[!, party_col])),
|
||||||
|
n_obs = nrow(df),
|
||||||
|
mean_economic = mean(df.economic_lr),
|
||||||
|
sd_economic = std(df.economic_lr),
|
||||||
|
mean_galtan = mean(df.galtan),
|
||||||
|
sd_galtan = std(df.galtan)
|
||||||
|
)
|
||||||
|
end
|
||||||
|
|
||||||
|
# Add display names
|
||||||
|
family_stats.family_name = [get(FAMILY_DISPLAY_NAMES, f, f) for f in family_stats.family]
|
||||||
|
|
||||||
|
# Sort by economic mean for display
|
||||||
|
sort!(family_stats, :mean_economic)
|
||||||
|
|
||||||
|
# Print table
|
||||||
|
@printf(" %-22s %7s %7s %10s %10s %10s %10s\n",
|
||||||
|
"Family", "Parties", "Obs", "Econ Mean", "Econ SD", "Cult Mean", "Cult SD")
|
||||||
|
println(" " * "-"^76)
|
||||||
|
|
||||||
|
for row in eachrow(family_stats)
|
||||||
|
@printf(" %-22s %7d %7d %10.3f %10.3f %10.3f %10.3f\n",
|
||||||
|
row.family_name, row.n_parties, row.n_obs,
|
||||||
|
row.mean_economic, row.sd_economic, row.mean_galtan, row.sd_galtan)
|
||||||
|
end
|
||||||
|
|
||||||
|
# Compute Spearman rank correlations for expected orderings
|
||||||
|
println()
|
||||||
|
|
||||||
|
# Economic ordering
|
||||||
|
econ_lookup = Dict(row.family => row.mean_economic for row in eachrow(family_stats))
|
||||||
|
econ_observed = [econ_lookup[f] for f in EXPECTED_ECONOMIC_ORDER if haskey(econ_lookup, f)]
|
||||||
|
econ_expected_ranks = collect(1:length(econ_observed))
|
||||||
|
econ_observed_ranks = ordinalrank(econ_observed)
|
||||||
|
rho_econ = corspearman(Float64.(econ_expected_ranks), Float64.(econ_observed_ranks))
|
||||||
|
|
||||||
|
println(@sprintf(" Economic ordering (5-family core): Spearman ρ = %.3f", rho_econ))
|
||||||
|
econ_families_used = [f for f in EXPECTED_ECONOMIC_ORDER if haskey(econ_lookup, f)]
|
||||||
|
econ_names = [get(FAMILY_DISPLAY_NAMES, f, f) for f in econ_families_used]
|
||||||
|
println(" Expected: ", join(econ_names, " < "))
|
||||||
|
observed_econ_order = econ_families_used[sortperm(econ_observed)]
|
||||||
|
observed_econ_names = [get(FAMILY_DISPLAY_NAMES, f, f) for f in observed_econ_order]
|
||||||
|
println(" Observed: ", join(observed_econ_names, " < "))
|
||||||
|
|
||||||
|
# Cultural ordering
|
||||||
|
galtan_lookup = Dict(row.family => row.mean_galtan for row in eachrow(family_stats))
|
||||||
|
galtan_observed = [galtan_lookup[f] for f in EXPECTED_GALTAN_ORDER if haskey(galtan_lookup, f)]
|
||||||
|
galtan_expected_ranks = collect(1:length(galtan_observed))
|
||||||
|
galtan_observed_ranks = ordinalrank(galtan_observed)
|
||||||
|
rho_galtan = corspearman(Float64.(galtan_expected_ranks), Float64.(galtan_observed_ranks))
|
||||||
|
|
||||||
|
println(@sprintf(" Cultural ordering (6-family): Spearman ρ = %.3f", rho_galtan))
|
||||||
|
galtan_families_used = [f for f in EXPECTED_GALTAN_ORDER if haskey(galtan_lookup, f)]
|
||||||
|
galtan_names = [get(FAMILY_DISPLAY_NAMES, f, f) for f in galtan_families_used]
|
||||||
|
println(" Expected: ", join(galtan_names, " < "))
|
||||||
|
observed_galtan_order = galtan_families_used[sortperm(galtan_observed)]
|
||||||
|
observed_galtan_names = [get(FAMILY_DISPLAY_NAMES, f, f) for f in observed_galtan_order]
|
||||||
|
println(" Observed: ", join(observed_galtan_names, " < "))
|
||||||
|
|
||||||
|
println()
|
||||||
|
println("-"^60)
|
||||||
|
if rho_econ >= 0.9 && rho_galtan >= 0.8
|
||||||
|
println("EXCELLENT: Family means follow expected orderings on both dimensions")
|
||||||
|
elseif rho_econ >= 0.7 && rho_galtan >= 0.7
|
||||||
|
println("GOOD: Family means broadly follow expected orderings")
|
||||||
|
else
|
||||||
|
println("CONCERN: Inspect family ordering results")
|
||||||
|
end
|
||||||
|
|
||||||
|
return family_stats, rho_econ, rho_galtan
|
||||||
|
end
|
||||||
|
|
||||||
|
function validate_temporal_stability(model::DataFrame)
|
||||||
|
"""Check for implausible year-to-year position changes"""
|
||||||
|
|
||||||
|
println("\n" * "="^60)
|
||||||
|
println("TEMPORAL STABILITY VALIDATION")
|
||||||
|
println("="^60)
|
||||||
|
println("\nFlagging parties with >0.10 change per year")
|
||||||
|
println()
|
||||||
|
|
||||||
|
party_col = hasproperty(model, :party_id) ? :party_id : :party
|
||||||
|
|
||||||
|
# Compute year-to-year changes within each party
|
||||||
|
sort!(model, [party_col, :year])
|
||||||
|
|
||||||
|
unstable_parties = []
|
||||||
|
|
||||||
|
for party_df in groupby(model, party_col)
|
||||||
|
if nrow(party_df) < 2
|
||||||
|
continue
|
||||||
|
end
|
||||||
|
|
||||||
|
party_id = party_df[1, party_col]
|
||||||
|
country = party_df[1, :country]
|
||||||
|
|
||||||
|
# Compute differences
|
||||||
|
for dim in [:economic_lr, :galtan]
|
||||||
|
vals = party_df[!, dim]
|
||||||
|
years = party_df.year
|
||||||
|
|
||||||
|
for i in 2:length(vals)
|
||||||
|
diff = abs(vals[i] - vals[i-1])
|
||||||
|
year_gap = years[i] - years[i-1]
|
||||||
|
|
||||||
|
# Normalize by year gap (handle multi-year gaps)
|
||||||
|
annual_change = diff / max(year_gap, 1)
|
||||||
|
|
||||||
|
if annual_change > 0.10
|
||||||
|
push!(unstable_parties, (
|
||||||
|
party_id = party_id,
|
||||||
|
country = country,
|
||||||
|
dimension = string(dim),
|
||||||
|
year_from = years[i-1],
|
||||||
|
year_to = years[i],
|
||||||
|
val_from = vals[i-1],
|
||||||
|
val_to = vals[i],
|
||||||
|
change = diff,
|
||||||
|
annual_change = annual_change
|
||||||
|
))
|
||||||
|
end
|
||||||
|
end
|
||||||
|
end
|
||||||
|
end
|
||||||
|
|
||||||
|
if isempty(unstable_parties)
|
||||||
|
println(" No parties with >0.10 annual change found")
|
||||||
|
println(" EXCELLENT: Positions are temporally stable")
|
||||||
|
return DataFrame()
|
||||||
|
end
|
||||||
|
|
||||||
|
unstable_df = DataFrame(unstable_parties)
|
||||||
|
sort!(unstable_df, :annual_change, rev=true)
|
||||||
|
|
||||||
|
println(" Found $(nrow(unstable_df)) instances of rapid change:")
|
||||||
|
println()
|
||||||
|
@printf(" %-8s %-8s %-12s %-10s %-10s %8s\n",
|
||||||
|
"Party", "Country", "Dimension", "Years", "Change", "Annual")
|
||||||
|
println(" " * "-"^60)
|
||||||
|
|
||||||
|
for row in eachrow(unstable_df[1:min(20, nrow(unstable_df)), :])
|
||||||
|
@printf(" %-8d %-8s %-12s %d->%d %8.3f %8.3f\n",
|
||||||
|
row.party_id, row.country, row.dimension,
|
||||||
|
row.year_from, row.year_to, row.change, row.annual_change)
|
||||||
|
end
|
||||||
|
|
||||||
|
if nrow(unstable_df) > 20
|
||||||
|
println(" ... and $(nrow(unstable_df) - 20) more")
|
||||||
|
end
|
||||||
|
|
||||||
|
println()
|
||||||
|
println("-"^60)
|
||||||
|
n_parties = length(unique(unstable_df.party_id))
|
||||||
|
n_total = length(unique(model[!, party_col]))
|
||||||
|
println(@sprintf("Unstable parties: %d/%d (%.1f%%)", n_parties, n_total, 100*n_parties/n_total))
|
||||||
|
|
||||||
|
return unstable_df
|
||||||
|
end
|
||||||
|
|
||||||
|
function validate_position_distributions(model::DataFrame)
|
||||||
|
"""Check overall distribution of positions makes sense"""
|
||||||
|
|
||||||
|
println("\n" * "="^60)
|
||||||
|
println("POSITION DISTRIBUTION VALIDATION")
|
||||||
|
println("="^60)
|
||||||
|
println("\nSummary statistics for model estimates")
|
||||||
|
println()
|
||||||
|
|
||||||
|
for dim in [:economic_lr, :galtan]
|
||||||
|
if !hasproperty(model, dim)
|
||||||
|
continue
|
||||||
|
end
|
||||||
|
|
||||||
|
vals = model[!, dim]
|
||||||
|
println("$dim:")
|
||||||
|
println(@sprintf(" Mean: %.3f (should be ~0.50)", mean(vals)))
|
||||||
|
println(@sprintf(" Median: %.3f (should be ~0.50)", median(vals)))
|
||||||
|
println(@sprintf(" SD: %.3f (should be ~0.15)", std(vals)))
|
||||||
|
println(@sprintf(" Min: %.3f", minimum(vals)))
|
||||||
|
println(@sprintf(" Max: %.3f", maximum(vals)))
|
||||||
|
println(@sprintf(" Q25: %.3f", quantile(vals, 0.25)))
|
||||||
|
println(@sprintf(" Q75: %.3f", quantile(vals, 0.75)))
|
||||||
|
println()
|
||||||
|
end
|
||||||
|
|
||||||
|
# Check for extreme values
|
||||||
|
println("Extreme positions (< 0.10 or > 0.90):")
|
||||||
|
|
||||||
|
party_col = hasproperty(model, :party_id) ? :party_id : :party
|
||||||
|
|
||||||
|
for dim in [:economic_lr, :galtan]
|
||||||
|
if !hasproperty(model, dim)
|
||||||
|
continue
|
||||||
|
end
|
||||||
|
|
||||||
|
extreme = filter(row -> row[dim] < 0.10 || row[dim] > 0.90, model)
|
||||||
|
n_extreme = nrow(extreme)
|
||||||
|
pct_extreme = 100 * n_extreme / nrow(model)
|
||||||
|
|
||||||
|
println(@sprintf(" %s: %d (%.1f%%)", dim, n_extreme, pct_extreme))
|
||||||
|
|
||||||
|
if n_extreme > 0 && n_extreme <= 10
|
||||||
|
for row in eachrow(extreme[1:min(5, nrow(extreme)), :])
|
||||||
|
println(@sprintf(" Party %d (%s) %d: %.3f",
|
||||||
|
row[party_col], row.country, row.year, row[dim]))
|
||||||
|
end
|
||||||
|
end
|
||||||
|
end
|
||||||
|
end
|
||||||
|
|
||||||
|
function validate_country_patterns(model::DataFrame)
|
||||||
|
"""Check country-level patterns make sense"""
|
||||||
|
|
||||||
|
println("\n" * "="^60)
|
||||||
|
println("COUNTRY-LEVEL VALIDATION")
|
||||||
|
println("="^60)
|
||||||
|
println("\nMean positions by country (should vary but not wildly)")
|
||||||
|
println()
|
||||||
|
|
||||||
|
country_stats = combine(groupby(model, :country)) do df
|
||||||
|
DataFrame(
|
||||||
|
n_parties = length(unique(hasproperty(df, :party_id) ? df.party_id : df.party)),
|
||||||
|
n_obs = nrow(df),
|
||||||
|
mean_econ = mean(df.economic_lr),
|
||||||
|
mean_galtan = mean(df.galtan),
|
||||||
|
sd_econ = std(df.economic_lr),
|
||||||
|
sd_galtan = std(df.galtan)
|
||||||
|
)
|
||||||
|
end
|
||||||
|
|
||||||
|
sort!(country_stats, :n_obs, rev=true)
|
||||||
|
|
||||||
|
@printf("%-4s %6s %6s %8s %8s %8s %8s\n",
|
||||||
|
"CC", "Parties", "N", "Econ", "SD", "Cult", "SD")
|
||||||
|
println("-"^60)
|
||||||
|
|
||||||
|
for row in eachrow(country_stats[1:min(20, nrow(country_stats)), :])
|
||||||
|
@printf("%-4s %6d %6d %8.3f %8.3f %8.3f %8.3f\n",
|
||||||
|
row.country, row.n_parties, row.n_obs,
|
||||||
|
row.mean_econ, row.sd_econ, row.mean_galtan, row.sd_galtan)
|
||||||
|
end
|
||||||
|
|
||||||
|
# Flag countries with unusual patterns
|
||||||
|
println("\nCountries with unusual patterns:")
|
||||||
|
unusual = filter(row -> row.mean_econ < 0.35 || row.mean_econ > 0.65 ||
|
||||||
|
row.mean_galtan < 0.35 || row.mean_galtan > 0.65, country_stats)
|
||||||
|
|
||||||
|
if nrow(unusual) == 0
|
||||||
|
println(" None - all countries have balanced party systems")
|
||||||
|
else
|
||||||
|
for row in eachrow(unusual)
|
||||||
|
issues = String[]
|
||||||
|
if row.mean_econ < 0.35
|
||||||
|
push!(issues, "left-skewed economy")
|
||||||
|
elseif row.mean_econ > 0.65
|
||||||
|
push!(issues, "right-skewed economy")
|
||||||
|
end
|
||||||
|
if row.mean_galtan < 0.35
|
||||||
|
push!(issues, "cosmopolitan-skewed")
|
||||||
|
elseif row.mean_galtan > 0.65
|
||||||
|
push!(issues, "traditionalist-skewed")
|
||||||
|
end
|
||||||
|
println(" $(row.country): $(join(issues, ", "))")
|
||||||
|
end
|
||||||
|
end
|
||||||
|
|
||||||
|
return country_stats
|
||||||
|
end
|
||||||
|
|
||||||
|
function save_construct_results(families::DataFrame, unstable::DataFrame,
|
||||||
|
countries::DataFrame, output_dir::String="outputs/checks")
|
||||||
|
"""Save construct validation results"""
|
||||||
|
|
||||||
|
if !isdir(output_dir)
|
||||||
|
mkpath(output_dir)
|
||||||
|
end
|
||||||
|
|
||||||
|
timestamp = Dates.format(now(), "yyyy-mm-dd_HH-MM-SS")
|
||||||
|
|
||||||
|
if nrow(families) > 0
|
||||||
|
families_file = joinpath(output_dir, "construct_families_$timestamp.csv")
|
||||||
|
CSV.write(families_file, families)
|
||||||
|
println("\nSaved: $families_file")
|
||||||
|
end
|
||||||
|
|
||||||
|
if nrow(unstable) > 0
|
||||||
|
unstable_file = joinpath(output_dir, "construct_unstable_$timestamp.csv")
|
||||||
|
CSV.write(unstable_file, unstable)
|
||||||
|
println("Saved: $unstable_file")
|
||||||
|
end
|
||||||
|
|
||||||
|
if nrow(countries) > 0
|
||||||
|
country_file = joinpath(output_dir, "construct_countries_$timestamp.csv")
|
||||||
|
CSV.write(country_file, countries)
|
||||||
|
println("Saved: $country_file")
|
||||||
|
end
|
||||||
|
end
|
||||||
|
|
||||||
|
|
||||||
|
# Main execution
|
||||||
|
function main()
|
||||||
|
println("="^60)
|
||||||
|
println("CONSTRUCT VALIDITY: Face Validity Checks")
|
||||||
|
println("="^60)
|
||||||
|
println("Checking if model estimates match expectations")
|
||||||
|
println()
|
||||||
|
|
||||||
|
# Load data
|
||||||
|
model, model_file = load_model_output()
|
||||||
|
|
||||||
|
println("\nModel output:")
|
||||||
|
println(" Rows: $(nrow(model))")
|
||||||
|
println(" Columns: $(names(model))")
|
||||||
|
|
||||||
|
# Run validations
|
||||||
|
families, rho_econ, rho_galtan = validate_party_families(model)
|
||||||
|
unstable = validate_temporal_stability(model)
|
||||||
|
validate_position_distributions(model)
|
||||||
|
countries = validate_country_patterns(model)
|
||||||
|
|
||||||
|
# Save results
|
||||||
|
save_construct_results(families, unstable, countries)
|
||||||
|
|
||||||
|
# Summary
|
||||||
|
println("\n" * "="^60)
|
||||||
|
println("CONSTRUCT VALIDITY SUMMARY")
|
||||||
|
println("="^60)
|
||||||
|
|
||||||
|
println(@sprintf(" Party family ordering:"))
|
||||||
|
println(@sprintf(" Economic (5-family): Spearman ρ = %.3f", rho_econ))
|
||||||
|
println(@sprintf(" Cultural (6-family): Spearman ρ = %.3f", rho_galtan))
|
||||||
|
|
||||||
|
n_unstable = nrow(unstable) > 0 ? length(unique(unstable.party_id)) : 0
|
||||||
|
party_col = hasproperty(model, :party_id) ? :party_id : :party
|
||||||
|
n_total = length(unique(model[!, party_col]))
|
||||||
|
println(@sprintf(" Temporal stability: %d/%d parties stable (>0.10/yr threshold)",
|
||||||
|
n_total - n_unstable, n_total))
|
||||||
|
|
||||||
|
if rho_econ >= 0.9 && rho_galtan >= 0.8 && n_unstable < 0.1 * n_total
|
||||||
|
println("\n EXCELLENT: Model has good construct validity")
|
||||||
|
elseif rho_econ >= 0.7 && rho_galtan >= 0.7
|
||||||
|
println("\n GOOD: Model has reasonable construct validity")
|
||||||
|
else
|
||||||
|
println("\n CONCERN: Inspect family ordering results")
|
||||||
|
end
|
||||||
|
|
||||||
|
|
||||||
|
println("\n" * "="^60)
|
||||||
|
println("VALIDATION COMPLETE")
|
||||||
|
println("="^60)
|
||||||
|
|
||||||
|
return (families=families, unstable=unstable, countries=countries)
|
||||||
|
end
|
||||||
|
|
||||||
|
if abspath(PROGRAM_FILE) == @__FILE__
|
||||||
|
main()
|
||||||
|
end
|
||||||
@@ -0,0 +1,533 @@
|
|||||||
|
#!/usr/bin/env julia
|
||||||
|
#############################################################################
|
||||||
|
## validate_convergent.jl
|
||||||
|
## Convergent validity: Compare model estimates to external expert surveys
|
||||||
|
##
|
||||||
|
## Following Claassen (2019), this script computes:
|
||||||
|
## - Pearson/Spearman correlations between model and expert estimates
|
||||||
|
## - Fisher z-transformation for correlation confidence intervals
|
||||||
|
## - Mean Absolute Error (MAE) and Root Mean Square Error (RMSE)
|
||||||
|
## - Breakdown by survey project and decade
|
||||||
|
##
|
||||||
|
## Target: r > 0.8 with CHES (Claassen achieved 0.50-0.57)
|
||||||
|
#############################################################################
|
||||||
|
|
||||||
|
using CSV, DataFrames, Statistics, StatsBase, Dates, Printf, JSON
|
||||||
|
|
||||||
|
# Fisher z-transformation for correlation confidence intervals
|
||||||
|
fisher_z(r) = 0.5 * log((1 + r) / (1 - r))
|
||||||
|
fisher_z_inv(z) = (exp(2z) - 1) / (exp(2z) + 1)
|
||||||
|
|
||||||
|
function correlation_ci(r, n; alpha=0.05)
|
||||||
|
"""Calculate confidence interval for correlation using Fisher z-transformation"""
|
||||||
|
if n < 4
|
||||||
|
return (lower=NaN, upper=NaN)
|
||||||
|
end
|
||||||
|
z = fisher_z(r)
|
||||||
|
se = 1 / sqrt(n - 3)
|
||||||
|
z_crit = 1.96 # For 95% CI
|
||||||
|
z_lower = z - z_crit * se
|
||||||
|
z_upper = z + z_crit * se
|
||||||
|
return (lower=fisher_z_inv(z_lower), upper=fisher_z_inv(z_upper))
|
||||||
|
end
|
||||||
|
|
||||||
|
function load_model_output(base_dir::String=".")
|
||||||
|
"""Load the most recent 2D model party positions output"""
|
||||||
|
|
||||||
|
# First check for post_estimation output in root (current name, with legacy fallback)
|
||||||
|
position_files = filter(f -> startswith(f, "party_positions_") && endswith(f, ".csv") &&
|
||||||
|
!endswith(f, "_metadata.txt") && !endswith(f, "_tables.tex"), readdir(base_dir))
|
||||||
|
legacy_files = filter(f -> startswith(f, "party_positions_v1_") && endswith(f, ".csv"), readdir(base_dir))
|
||||||
|
append!(position_files, legacy_files)
|
||||||
|
|
||||||
|
if !isempty(position_files)
|
||||||
|
latest = sort(position_files)[end]
|
||||||
|
println("Loading model output: $latest")
|
||||||
|
return CSV.read(joinpath(base_dir, latest), DataFrame), latest
|
||||||
|
end
|
||||||
|
|
||||||
|
# Check current pipeline output directory, with legacy estimations/ fallback
|
||||||
|
for (label, est_dir) in [
|
||||||
|
("outputs/estimations/latest", joinpath(base_dir, "outputs", "estimations", "latest")),
|
||||||
|
("estimations", joinpath(base_dir, "estimations")),
|
||||||
|
]
|
||||||
|
if isdir(est_dir)
|
||||||
|
est_files = filter(f -> startswith(f, "party_positions_") && endswith(f, ".csv") &&
|
||||||
|
!endswith(f, "_metadata.txt") && !endswith(f, "_tables.tex"), readdir(est_dir))
|
||||||
|
if !isempty(est_files)
|
||||||
|
latest = sort(est_files)[end]
|
||||||
|
println("Loading model output: $label/$latest")
|
||||||
|
return CSV.read(joinpath(est_dir, latest), DataFrame), latest
|
||||||
|
end
|
||||||
|
end
|
||||||
|
end
|
||||||
|
|
||||||
|
error("No party_positions_*.csv found. Run 02_post_estimation.jl first.")
|
||||||
|
end
|
||||||
|
|
||||||
|
function load_expert_data(base_dir::String=".")
|
||||||
|
"""Load expert survey data files"""
|
||||||
|
|
||||||
|
data_dir = isfile(joinpath(base_dir, "expert.csv")) ? base_dir : joinpath(base_dir, "data")
|
||||||
|
|
||||||
|
# Load dimension-specific expert data
|
||||||
|
expert_file = joinpath(data_dir, "expert.csv")
|
||||||
|
if !isfile(expert_file)
|
||||||
|
error("expert.csv not found in $base_dir or $(joinpath(base_dir, "data"))")
|
||||||
|
end
|
||||||
|
|
||||||
|
println("Loading expert.csv...")
|
||||||
|
expert = CSV.read(expert_file, DataFrame)
|
||||||
|
println(" Rows: $(nrow(expert))")
|
||||||
|
println(" Variables: $(unique(expert.var))")
|
||||||
|
|
||||||
|
# Load L-R data
|
||||||
|
lr_file = joinpath(data_dir, "lr_data.csv")
|
||||||
|
if !isfile(lr_file)
|
||||||
|
error("lr_data.csv not found in $base_dir or $(joinpath(base_dir, "data"))")
|
||||||
|
end
|
||||||
|
|
||||||
|
println("Loading lr_data.csv...")
|
||||||
|
lr_data = CSV.read(lr_file, DataFrame)
|
||||||
|
println(" Rows: $(nrow(lr_data))")
|
||||||
|
println(" Variables: $(unique(lr_data.var))")
|
||||||
|
|
||||||
|
return expert, lr_data
|
||||||
|
end
|
||||||
|
|
||||||
|
function validate_economic_lr(model::DataFrame, expert::DataFrame)
|
||||||
|
"""Validate economic_lr against CHES/V-Party/POPPA/GPS lrecon"""
|
||||||
|
|
||||||
|
println("\n" * "="^60)
|
||||||
|
println("CONVERGENT VALIDITY: economic_lr")
|
||||||
|
println("="^60)
|
||||||
|
|
||||||
|
# Filter expert data for economic dimension
|
||||||
|
econ_vars = filter(v -> startswith(v, "lrecon_"), unique(expert.var))
|
||||||
|
econ_expert = filter(row -> row.var in econ_vars, expert)
|
||||||
|
|
||||||
|
println("\nExpert data variables: $(econ_vars)")
|
||||||
|
println("Expert observations: $(nrow(econ_expert))")
|
||||||
|
|
||||||
|
# Merge with model output
|
||||||
|
# Model has party_id column (from segment-based), expert has party column
|
||||||
|
if hasproperty(model, :party_id)
|
||||||
|
model_merge = select(model, :party_id => :party, :year, :economic_lr, :economic_lr_se)
|
||||||
|
else
|
||||||
|
model_merge = select(model, :party, :year, :economic_lr, :economic_lr_se)
|
||||||
|
end
|
||||||
|
|
||||||
|
merged = innerjoin(econ_expert, model_merge, on=[:party, :year])
|
||||||
|
|
||||||
|
println("Merged observations: $(nrow(merged))")
|
||||||
|
|
||||||
|
if nrow(merged) < 10
|
||||||
|
println("WARNING: Too few observations for meaningful validation")
|
||||||
|
return nothing
|
||||||
|
end
|
||||||
|
|
||||||
|
# Compute overall correlation
|
||||||
|
r_pearson = cor(merged.val, merged.economic_lr)
|
||||||
|
r_spearman = corspearman(merged.val, merged.economic_lr)
|
||||||
|
mae = mean(abs.(merged.val .- merged.economic_lr))
|
||||||
|
rmse = sqrt(mean((merged.val .- merged.economic_lr).^2))
|
||||||
|
|
||||||
|
ci = correlation_ci(r_pearson, nrow(merged))
|
||||||
|
|
||||||
|
println("\n--- Overall Statistics ---")
|
||||||
|
println(@sprintf(" Pearson r: %.4f [%.4f, %.4f]", r_pearson, ci.lower, ci.upper))
|
||||||
|
println(@sprintf(" Spearman r: %.4f", r_spearman))
|
||||||
|
println(@sprintf(" MAE: %.4f", mae))
|
||||||
|
println(@sprintf(" RMSE: %.4f", rmse))
|
||||||
|
println(@sprintf(" N: %d", nrow(merged)))
|
||||||
|
|
||||||
|
# Breakdown by project
|
||||||
|
println("\n--- By Project ---")
|
||||||
|
by_project = combine(groupby(merged, :project)) do df
|
||||||
|
n = nrow(df)
|
||||||
|
if n < 3
|
||||||
|
return DataFrame(n=n, r_pearson=NaN, mae=NaN)
|
||||||
|
end
|
||||||
|
DataFrame(
|
||||||
|
n = n,
|
||||||
|
r_pearson = cor(df.val, df.economic_lr),
|
||||||
|
r_spearman = corspearman(df.val, df.economic_lr),
|
||||||
|
mae = mean(abs.(df.val .- df.economic_lr)),
|
||||||
|
rmse = sqrt(mean((df.val .- df.economic_lr).^2))
|
||||||
|
)
|
||||||
|
end
|
||||||
|
|
||||||
|
for row in eachrow(sort(by_project, :n, rev=true))
|
||||||
|
if !isnan(row.r_pearson)
|
||||||
|
println(@sprintf(" %-10s: r=%.3f, MAE=%.3f, n=%d",
|
||||||
|
row.project, row.r_pearson, row.mae, row.n))
|
||||||
|
end
|
||||||
|
end
|
||||||
|
|
||||||
|
# Breakdown by decade
|
||||||
|
println("\n--- By Decade ---")
|
||||||
|
merged.decade = div.(merged.year, 10) .* 10
|
||||||
|
by_decade = combine(groupby(merged, :decade)) do df
|
||||||
|
n = nrow(df)
|
||||||
|
if n < 3
|
||||||
|
return DataFrame(n=n, r_pearson=NaN, mae=NaN)
|
||||||
|
end
|
||||||
|
DataFrame(
|
||||||
|
n = n,
|
||||||
|
r_pearson = cor(df.val, df.economic_lr),
|
||||||
|
mae = mean(abs.(df.val .- df.economic_lr))
|
||||||
|
)
|
||||||
|
end
|
||||||
|
|
||||||
|
for row in eachrow(sort(by_decade, :decade))
|
||||||
|
if !isnan(row.r_pearson)
|
||||||
|
println(@sprintf(" %ds: r=%.3f, MAE=%.3f, n=%d",
|
||||||
|
row.decade, row.r_pearson, row.mae, row.n))
|
||||||
|
end
|
||||||
|
end
|
||||||
|
|
||||||
|
return (
|
||||||
|
dimension = "economic_lr",
|
||||||
|
r_pearson = r_pearson,
|
||||||
|
r_spearman = r_spearman,
|
||||||
|
ci_lower = ci.lower,
|
||||||
|
ci_upper = ci.upper,
|
||||||
|
mae = mae,
|
||||||
|
rmse = rmse,
|
||||||
|
n = nrow(merged),
|
||||||
|
by_project = by_project,
|
||||||
|
by_decade = by_decade
|
||||||
|
)
|
||||||
|
end
|
||||||
|
|
||||||
|
function validate_galtan(model::DataFrame, expert::DataFrame)
|
||||||
|
"""Validate cultural cosmopolitan--traditionalist estimates against CHES and V-Party/GPS cultural measures"""
|
||||||
|
|
||||||
|
println("\n" * "="^60)
|
||||||
|
println("CONVERGENT VALIDITY: cultural cosmopolitan--traditionalist")
|
||||||
|
println("="^60)
|
||||||
|
|
||||||
|
# Filter expert data for cultural cosmopolitan--traditionalist dimension
|
||||||
|
galtan_vars = filter(v -> occursin("galtan", v) || occursin("libcon", v), unique(expert.var))
|
||||||
|
galtan_expert = filter(row -> row.var in galtan_vars, expert)
|
||||||
|
|
||||||
|
println("\nExpert data variables: $(galtan_vars)")
|
||||||
|
println("Expert observations: $(nrow(galtan_expert))")
|
||||||
|
|
||||||
|
# Merge with model output
|
||||||
|
if hasproperty(model, :party_id)
|
||||||
|
model_merge = select(model, :party_id => :party, :year, :galtan, :galtan_se)
|
||||||
|
else
|
||||||
|
model_merge = select(model, :party, :year, :galtan, :galtan_se)
|
||||||
|
end
|
||||||
|
|
||||||
|
merged = innerjoin(galtan_expert, model_merge, on=[:party, :year])
|
||||||
|
|
||||||
|
println("Merged observations: $(nrow(merged))")
|
||||||
|
|
||||||
|
if nrow(merged) < 10
|
||||||
|
println("WARNING: Too few observations for meaningful validation")
|
||||||
|
return nothing
|
||||||
|
end
|
||||||
|
|
||||||
|
# Compute overall correlation
|
||||||
|
r_pearson = cor(merged.val, merged.galtan)
|
||||||
|
r_spearman = corspearman(merged.val, merged.galtan)
|
||||||
|
mae = mean(abs.(merged.val .- merged.galtan))
|
||||||
|
rmse = sqrt(mean((merged.val .- merged.galtan).^2))
|
||||||
|
|
||||||
|
ci = correlation_ci(r_pearson, nrow(merged))
|
||||||
|
|
||||||
|
println("\n--- Overall Statistics ---")
|
||||||
|
println(@sprintf(" Pearson r: %.4f [%.4f, %.4f]", r_pearson, ci.lower, ci.upper))
|
||||||
|
println(@sprintf(" Spearman r: %.4f", r_spearman))
|
||||||
|
println(@sprintf(" MAE: %.4f", mae))
|
||||||
|
println(@sprintf(" RMSE: %.4f", rmse))
|
||||||
|
println(@sprintf(" N: %d", nrow(merged)))
|
||||||
|
|
||||||
|
# Breakdown by project
|
||||||
|
println("\n--- By Project ---")
|
||||||
|
by_project = combine(groupby(merged, :project)) do df
|
||||||
|
n = nrow(df)
|
||||||
|
if n < 3
|
||||||
|
return DataFrame(n=n, r_pearson=NaN, mae=NaN)
|
||||||
|
end
|
||||||
|
DataFrame(
|
||||||
|
n = n,
|
||||||
|
r_pearson = cor(df.val, df.galtan),
|
||||||
|
r_spearman = corspearman(df.val, df.galtan),
|
||||||
|
mae = mean(abs.(df.val .- df.galtan)),
|
||||||
|
rmse = sqrt(mean((df.val .- df.galtan).^2))
|
||||||
|
)
|
||||||
|
end
|
||||||
|
|
||||||
|
for row in eachrow(sort(by_project, :n, rev=true))
|
||||||
|
if !isnan(row.r_pearson)
|
||||||
|
println(@sprintf(" %-10s: r=%.3f, MAE=%.3f, n=%d",
|
||||||
|
row.project, row.r_pearson, row.mae, row.n))
|
||||||
|
end
|
||||||
|
end
|
||||||
|
|
||||||
|
# Breakdown by decade
|
||||||
|
println("\n--- By Decade ---")
|
||||||
|
merged.decade = div.(merged.year, 10) .* 10
|
||||||
|
by_decade = combine(groupby(merged, :decade)) do df
|
||||||
|
n = nrow(df)
|
||||||
|
if n < 3
|
||||||
|
return DataFrame(n=n, r_pearson=NaN, mae=NaN)
|
||||||
|
end
|
||||||
|
DataFrame(
|
||||||
|
n = n,
|
||||||
|
r_pearson = cor(df.val, df.galtan),
|
||||||
|
mae = mean(abs.(df.val .- df.galtan))
|
||||||
|
)
|
||||||
|
end
|
||||||
|
|
||||||
|
for row in eachrow(sort(by_decade, :decade))
|
||||||
|
if !isnan(row.r_pearson)
|
||||||
|
println(@sprintf(" %ds: r=%.3f, MAE=%.3f, n=%d",
|
||||||
|
row.decade, row.r_pearson, row.mae, row.n))
|
||||||
|
end
|
||||||
|
end
|
||||||
|
|
||||||
|
return (
|
||||||
|
dimension = "galtan",
|
||||||
|
r_pearson = r_pearson,
|
||||||
|
r_spearman = r_spearman,
|
||||||
|
ci_lower = ci.lower,
|
||||||
|
ci_upper = ci.upper,
|
||||||
|
mae = mae,
|
||||||
|
rmse = rmse,
|
||||||
|
n = nrow(merged),
|
||||||
|
by_project = by_project,
|
||||||
|
by_decade = by_decade
|
||||||
|
)
|
||||||
|
end
|
||||||
|
|
||||||
|
function validate_discriminant(model::DataFrame, expert::DataFrame)
|
||||||
|
"""Compute cross-dimension correlations for discriminant validity (Campbell & Fiske 1959 MTMM)"""
|
||||||
|
|
||||||
|
println("\n" * "="^60)
|
||||||
|
println("DISCRIMINANT VALIDITY: Cross-dimension correlations")
|
||||||
|
println("="^60)
|
||||||
|
println("\nCampbell & Fiske (1959) MTMM framework:")
|
||||||
|
println(" Convergent: same dimension, different method → HIGH")
|
||||||
|
println(" Discriminant: different dimension, different method → LOW")
|
||||||
|
println()
|
||||||
|
|
||||||
|
# Get party column
|
||||||
|
if hasproperty(model, :party_id)
|
||||||
|
model_econ = select(model, :party_id => :party, :year, :economic_lr)
|
||||||
|
model_gal = select(model, :party_id => :party, :year, :galtan)
|
||||||
|
else
|
||||||
|
model_econ = select(model, :party, :year, :economic_lr)
|
||||||
|
model_gal = select(model, :party, :year, :galtan)
|
||||||
|
end
|
||||||
|
|
||||||
|
results = []
|
||||||
|
|
||||||
|
# 1. Expert economic vs Model economic (convergent - already computed, include for matrix)
|
||||||
|
econ_vars = filter(v -> startswith(v, "lrecon_"), unique(expert.var))
|
||||||
|
econ_expert = filter(row -> row.var in econ_vars, expert)
|
||||||
|
merged_ee = innerjoin(econ_expert, model_econ, on=[:party, :year])
|
||||||
|
if nrow(merged_ee) >= 10
|
||||||
|
r = cor(merged_ee.val, merged_ee.economic_lr)
|
||||||
|
push!(results, (model_dim="economic_lr", expert_dim="economic",
|
||||||
|
r_pearson=r, r_spearman=corspearman(merged_ee.val, merged_ee.economic_lr),
|
||||||
|
n=nrow(merged_ee), type="convergent"))
|
||||||
|
@printf(" Model Economic × Expert Economic: r = %.3f (convergent, n=%d)\n", r, nrow(merged_ee))
|
||||||
|
end
|
||||||
|
|
||||||
|
# 2. Expert economic vs model cultural dimension (discriminant)
|
||||||
|
merged_eg = innerjoin(econ_expert, model_gal, on=[:party, :year])
|
||||||
|
if nrow(merged_eg) >= 10
|
||||||
|
r = cor(merged_eg.val, merged_eg.galtan)
|
||||||
|
push!(results, (model_dim="galtan", expert_dim="economic",
|
||||||
|
r_pearson=r, r_spearman=corspearman(merged_eg.val, merged_eg.galtan),
|
||||||
|
n=nrow(merged_eg), type="discriminant"))
|
||||||
|
@printf(" Model Cultural × Expert Economic: r = %.3f (discriminant, n=%d)\n", r, nrow(merged_eg))
|
||||||
|
end
|
||||||
|
|
||||||
|
# 3. Expert cultural vs model cultural dimension (convergent - already computed, include for matrix)
|
||||||
|
galtan_vars = filter(v -> occursin("galtan", v) || occursin("libcon", v), unique(expert.var))
|
||||||
|
galtan_expert = filter(row -> row.var in galtan_vars, expert)
|
||||||
|
merged_gg = innerjoin(galtan_expert, model_gal, on=[:party, :year])
|
||||||
|
if nrow(merged_gg) >= 10
|
||||||
|
r = cor(merged_gg.val, merged_gg.galtan)
|
||||||
|
push!(results, (model_dim="galtan", expert_dim="galtan",
|
||||||
|
r_pearson=r, r_spearman=corspearman(merged_gg.val, merged_gg.galtan),
|
||||||
|
n=nrow(merged_gg), type="convergent"))
|
||||||
|
@printf(" Model Cultural × Expert Cultural: r = %.3f (convergent, n=%d)\n", r, nrow(merged_gg))
|
||||||
|
end
|
||||||
|
|
||||||
|
# 4. Expert cultural vs model economic dimension (discriminant)
|
||||||
|
merged_ge = innerjoin(galtan_expert, model_econ, on=[:party, :year])
|
||||||
|
if nrow(merged_ge) >= 10
|
||||||
|
r = cor(merged_ge.val, merged_ge.economic_lr)
|
||||||
|
push!(results, (model_dim="economic_lr", expert_dim="galtan",
|
||||||
|
r_pearson=r, r_spearman=corspearman(merged_ge.val, merged_ge.economic_lr),
|
||||||
|
n=nrow(merged_ge), type="discriminant"))
|
||||||
|
@printf(" Model Economic × Expert Cultural: r = %.3f (discriminant, n=%d)\n", r, nrow(merged_ge))
|
||||||
|
end
|
||||||
|
|
||||||
|
println()
|
||||||
|
println("MTMM Matrix:")
|
||||||
|
println(" Expert Economic Expert Cultural")
|
||||||
|
for r in results
|
||||||
|
if r.model_dim == "economic_lr" && r.expert_dim == "economic"
|
||||||
|
@printf(" Model Economic: %.3f ", r.r_pearson)
|
||||||
|
end
|
||||||
|
end
|
||||||
|
for r in results
|
||||||
|
if r.model_dim == "economic_lr" && r.expert_dim == "galtan"
|
||||||
|
@printf("%.3f\n", r.r_pearson)
|
||||||
|
end
|
||||||
|
end
|
||||||
|
for r in results
|
||||||
|
if r.model_dim == "galtan" && r.expert_dim == "economic"
|
||||||
|
@printf(" Model Cultural: %.3f ", r.r_pearson)
|
||||||
|
end
|
||||||
|
end
|
||||||
|
for r in results
|
||||||
|
if r.model_dim == "galtan" && r.expert_dim == "galtan"
|
||||||
|
@printf("%.3f\n", r.r_pearson)
|
||||||
|
end
|
||||||
|
end
|
||||||
|
|
||||||
|
return DataFrame(results)
|
||||||
|
end
|
||||||
|
|
||||||
|
function save_validation_results(results::Vector, output_dir::String="validation";
|
||||||
|
discriminant::Union{DataFrame, Nothing}=nothing)
|
||||||
|
"""Save validation results to CSV files"""
|
||||||
|
|
||||||
|
if !isdir(output_dir)
|
||||||
|
mkpath(output_dir)
|
||||||
|
end
|
||||||
|
|
||||||
|
timestamp = Dates.format(now(), "yyyy-mm-dd_HH-MM-SS")
|
||||||
|
|
||||||
|
# Summary table
|
||||||
|
summary_rows = []
|
||||||
|
for r in results
|
||||||
|
if r !== nothing
|
||||||
|
push!(summary_rows, (
|
||||||
|
dimension = r.dimension,
|
||||||
|
r_pearson = r.r_pearson,
|
||||||
|
r_spearman = r.r_spearman,
|
||||||
|
ci_lower = r.ci_lower,
|
||||||
|
ci_upper = r.ci_upper,
|
||||||
|
mae = r.mae,
|
||||||
|
rmse = r.rmse,
|
||||||
|
n = r.n
|
||||||
|
))
|
||||||
|
end
|
||||||
|
end
|
||||||
|
|
||||||
|
if !isempty(summary_rows)
|
||||||
|
summary_df = DataFrame(summary_rows)
|
||||||
|
summary_file = joinpath(output_dir, "convergent_summary_$timestamp.csv")
|
||||||
|
CSV.write(summary_file, summary_df)
|
||||||
|
println("\nSaved: $summary_file")
|
||||||
|
end
|
||||||
|
|
||||||
|
# By-project tables
|
||||||
|
for r in results
|
||||||
|
if r !== nothing && hasproperty(r, :by_project) && r.by_project !== nothing
|
||||||
|
project_file = joinpath(output_dir, "convergent_$(r.dimension)_by_project_$timestamp.csv")
|
||||||
|
CSV.write(project_file, r.by_project)
|
||||||
|
println("Saved: $project_file")
|
||||||
|
end
|
||||||
|
end
|
||||||
|
|
||||||
|
# By-decade tables
|
||||||
|
for r in results
|
||||||
|
if r !== nothing && hasproperty(r, :by_decade) && r.by_decade !== nothing
|
||||||
|
decade_file = joinpath(output_dir, "convergent_$(r.dimension)_by_decade_$timestamp.csv")
|
||||||
|
CSV.write(decade_file, r.by_decade)
|
||||||
|
println("Saved: $decade_file")
|
||||||
|
end
|
||||||
|
end
|
||||||
|
|
||||||
|
# Discriminant validity table
|
||||||
|
if discriminant !== nothing && nrow(discriminant) > 0
|
||||||
|
disc_file = joinpath(output_dir, "discriminant_summary_$timestamp.csv")
|
||||||
|
CSV.write(disc_file, discriminant)
|
||||||
|
println("Saved: $disc_file")
|
||||||
|
end
|
||||||
|
|
||||||
|
return summary_rows
|
||||||
|
end
|
||||||
|
|
||||||
|
function print_claassen_comparison(results::Vector)
|
||||||
|
"""Print comparison with Claassen (2019) benchmarks"""
|
||||||
|
|
||||||
|
println("\n" * "="^60)
|
||||||
|
println("COMPARISON WITH CLAASSEN (2019) BENCHMARKS")
|
||||||
|
println("="^60)
|
||||||
|
|
||||||
|
println("\nClaassen's results (mood estimates vs survey data):")
|
||||||
|
println(" Pearson r: 0.50-0.57")
|
||||||
|
println(" MAE: ~0.06 (6 pp on 0-1 scale)")
|
||||||
|
println()
|
||||||
|
|
||||||
|
println("Our target (party positions, should be HIGHER than mood):")
|
||||||
|
println(" Pearson r > 0.80 with expert surveys")
|
||||||
|
println(" MAE < 0.15 (reasonable measurement error)")
|
||||||
|
println()
|
||||||
|
|
||||||
|
println("-"^60)
|
||||||
|
@printf("%-15s %8s %8s %8s %8s\n", "Dimension", "r", "Target", "MAE", "Status")
|
||||||
|
println("-"^60)
|
||||||
|
|
||||||
|
for r in results
|
||||||
|
if r !== nothing
|
||||||
|
status = r.r_pearson > 0.80 ? "PASS" : (r.r_pearson > 0.70 ? "OK" : "LOW")
|
||||||
|
@printf("%-15s %8.3f %8s %8.3f %8s\n",
|
||||||
|
r.dimension, r.r_pearson, "> 0.80", r.mae, status)
|
||||||
|
end
|
||||||
|
end
|
||||||
|
println("-"^60)
|
||||||
|
end
|
||||||
|
|
||||||
|
# Main execution
|
||||||
|
function main()
|
||||||
|
println("="^60)
|
||||||
|
println("CONVERGENT VALIDITY: Model vs Expert Surveys")
|
||||||
|
println("="^60)
|
||||||
|
println("Following Claassen (2019) validation framework")
|
||||||
|
println()
|
||||||
|
|
||||||
|
# Load data
|
||||||
|
model, model_file = load_model_output()
|
||||||
|
expert, _ = load_expert_data()
|
||||||
|
|
||||||
|
println("\nModel output:")
|
||||||
|
println(" Rows: $(nrow(model))")
|
||||||
|
println(" Columns: $(names(model))")
|
||||||
|
|
||||||
|
# Run validations
|
||||||
|
results = []
|
||||||
|
|
||||||
|
push!(results, validate_economic_lr(model, expert))
|
||||||
|
push!(results, validate_galtan(model, expert))
|
||||||
|
|
||||||
|
# Run discriminant validity
|
||||||
|
discriminant = validate_discriminant(model, expert)
|
||||||
|
|
||||||
|
# Save results
|
||||||
|
save_validation_results(results; discriminant=discriminant)
|
||||||
|
|
||||||
|
# Print Claassen comparison
|
||||||
|
print_claassen_comparison(results)
|
||||||
|
|
||||||
|
println("\n" * "="^60)
|
||||||
|
println("VALIDATION COMPLETE")
|
||||||
|
println("="^60)
|
||||||
|
|
||||||
|
return results
|
||||||
|
end
|
||||||
|
|
||||||
|
if abspath(PROGRAM_FILE) == @__FILE__
|
||||||
|
main()
|
||||||
|
end
|
||||||
@@ -0,0 +1,375 @@
|
|||||||
|
#!/usr/bin/env julia
|
||||||
|
#############################################################################
|
||||||
|
## validate_external.jl
|
||||||
|
## Out-of-sample validation via held-out expert observations
|
||||||
|
##
|
||||||
|
## Design:
|
||||||
|
## - Text data stays 100% intact (same parties, segments, indices)
|
||||||
|
## - 20% of expert/LR observations held out (stratified by source)
|
||||||
|
## - Expert-only parties (no text data) are never held out
|
||||||
|
## - Model trains on 80% expert + 100% text
|
||||||
|
## - Held-out expert ratings compared to model predictions
|
||||||
|
##
|
||||||
|
## Usage:
|
||||||
|
## julia scripts/validate_external.jl prepare
|
||||||
|
## julia 01_run_model.jl --data-dir validation/external_split/
|
||||||
|
## julia 02_post_estimation.jl # on training run
|
||||||
|
## julia scripts/validate_external.jl compute <model_positions.csv>
|
||||||
|
#############################################################################
|
||||||
|
|
||||||
|
using CSV, DataFrames, Statistics, Random, Dates, Printf
|
||||||
|
|
||||||
|
const HOLDOUT_FRAC = 0.20
|
||||||
|
const SEED = 42
|
||||||
|
|
||||||
|
# =========================================================================
|
||||||
|
# STEP 1: Prepare train/test split
|
||||||
|
# =========================================================================
|
||||||
|
|
||||||
|
function prepare_holdout_data(base_dir::String=".")
|
||||||
|
println("="^70)
|
||||||
|
println("PREPARING OUT-OF-SAMPLE VALIDATION SPLIT")
|
||||||
|
println("="^70)
|
||||||
|
println()
|
||||||
|
|
||||||
|
# Load data
|
||||||
|
text_data = CSV.read(joinpath(base_dir, "text_data.csv"), DataFrame)
|
||||||
|
expert = CSV.read(joinpath(base_dir, "expert.csv"), DataFrame)
|
||||||
|
lr_data = CSV.read(joinpath(base_dir, "lr_data.csv"), DataFrame)
|
||||||
|
|
||||||
|
println("Full dataset:")
|
||||||
|
println(" text_data: $(nrow(text_data)) rows, $(length(unique(text_data.party))) parties")
|
||||||
|
println(" expert: $(nrow(expert)) rows, $(length(unique(expert.party))) parties")
|
||||||
|
println(" lr_data: $(nrow(lr_data)) rows, $(length(unique(lr_data.party))) parties")
|
||||||
|
|
||||||
|
# Identify expert-only parties (no text data) — these are NEVER held out
|
||||||
|
text_parties = Set(unique(text_data.party))
|
||||||
|
expert_only_parties = Set(p for p in unique(vcat(expert.party, lr_data.party))
|
||||||
|
if !(p in text_parties))
|
||||||
|
|
||||||
|
println()
|
||||||
|
println("Expert-only parties (protected from holdout): $(length(expert_only_parties))")
|
||||||
|
|
||||||
|
# Split expert data: stratified by source variable
|
||||||
|
# Safeguard: ensure each party keeps at least one observation in training
|
||||||
|
Random.seed!(SEED)
|
||||||
|
expert.row_id = 1:nrow(expert)
|
||||||
|
expert.is_holdout = falses(nrow(expert))
|
||||||
|
|
||||||
|
for var_group in groupby(expert, :var)
|
||||||
|
var_name = first(var_group.var)
|
||||||
|
eligible = findall(row -> !(row.party in expert_only_parties), eachrow(var_group))
|
||||||
|
|
||||||
|
n_holdout = round(Int, length(eligible) * HOLDOUT_FRAC)
|
||||||
|
holdout_candidates = shuffle(eligible)
|
||||||
|
|
||||||
|
# Track per-party counts to ensure at least 1 stays in training
|
||||||
|
party_train_count = Dict{Int, Int}()
|
||||||
|
for idx in eligible
|
||||||
|
p = var_group.party[idx]
|
||||||
|
party_train_count[p] = get(party_train_count, p, 0) + 1
|
||||||
|
end
|
||||||
|
|
||||||
|
n_held = 0
|
||||||
|
for idx in holdout_candidates
|
||||||
|
n_held >= n_holdout && break
|
||||||
|
p = var_group.party[idx]
|
||||||
|
if party_train_count[p] > 1 # keep at least 1 in training
|
||||||
|
row_id = var_group.row_id[idx]
|
||||||
|
expert.is_holdout[row_id] = true
|
||||||
|
party_train_count[p] -= 1
|
||||||
|
n_held += 1
|
||||||
|
end
|
||||||
|
end
|
||||||
|
end
|
||||||
|
|
||||||
|
# Split LR data: stratified by source variable (same safeguard)
|
||||||
|
lr_data.row_id = 1:nrow(lr_data)
|
||||||
|
lr_data.is_holdout = falses(nrow(lr_data))
|
||||||
|
|
||||||
|
for var_group in groupby(lr_data, :var)
|
||||||
|
var_name = first(var_group.var)
|
||||||
|
eligible = findall(row -> !(row.party in expert_only_parties), eachrow(var_group))
|
||||||
|
|
||||||
|
n_holdout = round(Int, length(eligible) * HOLDOUT_FRAC)
|
||||||
|
holdout_candidates = shuffle(eligible)
|
||||||
|
|
||||||
|
party_train_count = Dict{Int, Int}()
|
||||||
|
for idx in eligible
|
||||||
|
p = var_group.party[idx]
|
||||||
|
party_train_count[p] = get(party_train_count, p, 0) + 1
|
||||||
|
end
|
||||||
|
|
||||||
|
n_held = 0
|
||||||
|
for idx in holdout_candidates
|
||||||
|
n_held >= n_holdout && break
|
||||||
|
p = var_group.party[idx]
|
||||||
|
if party_train_count[p] > 1
|
||||||
|
row_id = var_group.row_id[idx]
|
||||||
|
lr_data.is_holdout[row_id] = true
|
||||||
|
party_train_count[p] -= 1
|
||||||
|
n_held += 1
|
||||||
|
end
|
||||||
|
end
|
||||||
|
end
|
||||||
|
|
||||||
|
# Create train/test splits
|
||||||
|
expert_train = expert[.!expert.is_holdout, Not([:row_id, :is_holdout])]
|
||||||
|
expert_test = expert[expert.is_holdout, Not([:row_id, :is_holdout])]
|
||||||
|
lr_train = lr_data[.!lr_data.is_holdout, Not([:row_id, :is_holdout])]
|
||||||
|
lr_test = lr_data[lr_data.is_holdout, Not([:row_id, :is_holdout])]
|
||||||
|
|
||||||
|
# Report split
|
||||||
|
println()
|
||||||
|
println("Split summary ($(round(100*HOLDOUT_FRAC))% holdout):")
|
||||||
|
println(" expert train: $(nrow(expert_train)) rows ($(round(100*nrow(expert_train)/nrow(expert), digits=1))%)")
|
||||||
|
println(" expert test: $(nrow(expert_test)) rows ($(round(100*nrow(expert_test)/nrow(expert), digits=1))%)")
|
||||||
|
println(" lr train: $(nrow(lr_train)) rows ($(round(100*nrow(lr_train)/nrow(lr_data), digits=1))%)")
|
||||||
|
println(" lr test: $(nrow(lr_test)) rows ($(round(100*nrow(lr_test)/nrow(lr_data), digits=1))%)")
|
||||||
|
|
||||||
|
# Report per-source breakdown
|
||||||
|
println()
|
||||||
|
println("Per-source breakdown (expert):")
|
||||||
|
for var_name in sort(unique(expert.var))
|
||||||
|
n_full = count(expert.var .== var_name)
|
||||||
|
n_test = count(expert_test.var .== var_name)
|
||||||
|
println(" $var_name: $(n_full - n_test) train / $n_test test")
|
||||||
|
end
|
||||||
|
println()
|
||||||
|
println("Per-source breakdown (LR):")
|
||||||
|
for var_name in sort(unique(lr_data.var))
|
||||||
|
n_full = count(lr_data.var .== var_name)
|
||||||
|
n_test = count(lr_test.var .== var_name)
|
||||||
|
println(" $var_name: $(n_full - n_test) train / $n_test test")
|
||||||
|
end
|
||||||
|
|
||||||
|
# Verify: training set has same parties as full set
|
||||||
|
train_parties_expert = Set(unique(expert_train.party))
|
||||||
|
train_parties_lr = Set(unique(lr_train.party))
|
||||||
|
full_parties_expert = Set(unique(expert.party))
|
||||||
|
full_parties_lr = Set(unique(lr_data.party))
|
||||||
|
|
||||||
|
lost_expert = setdiff(full_parties_expert, train_parties_expert)
|
||||||
|
lost_lr = setdiff(full_parties_lr, train_parties_lr)
|
||||||
|
|
||||||
|
println()
|
||||||
|
if isempty(lost_expert) && isempty(lost_lr)
|
||||||
|
println("✓ No parties lost from training set")
|
||||||
|
else
|
||||||
|
println("⚠ Parties lost from expert training: $(length(lost_expert))")
|
||||||
|
println("⚠ Parties lost from LR training: $(length(lost_lr))")
|
||||||
|
end
|
||||||
|
|
||||||
|
# Save to output directory
|
||||||
|
# Files use standard names so 01_run_model.jl can load with --data-dir
|
||||||
|
output_dir = joinpath(base_dir, "validation", "external_split")
|
||||||
|
mkpath(output_dir)
|
||||||
|
|
||||||
|
# Training files (standard names for model loading)
|
||||||
|
CSV.write(joinpath(output_dir, "text_data.csv"), text_data) # UNCHANGED
|
||||||
|
CSV.write(joinpath(output_dir, "expert.csv"), expert_train)
|
||||||
|
CSV.write(joinpath(output_dir, "lr_data.csv"), lr_train)
|
||||||
|
|
||||||
|
# Test files (for compute step)
|
||||||
|
CSV.write(joinpath(output_dir, "expert_test.csv"), expert_test)
|
||||||
|
CSV.write(joinpath(output_dir, "lr_data_test.csv"), lr_test)
|
||||||
|
|
||||||
|
# Copy union mapping (needed by model)
|
||||||
|
if isdir(joinpath(base_dir, "data"))
|
||||||
|
mkpath(joinpath(output_dir, "data"))
|
||||||
|
cp(joinpath(base_dir, "data", "union_mapping.csv"),
|
||||||
|
joinpath(output_dir, "data", "union_mapping.csv"), force=true)
|
||||||
|
end
|
||||||
|
|
||||||
|
println()
|
||||||
|
println("Files saved to: $output_dir")
|
||||||
|
println(" text_data.csv — IDENTICAL to original ($(nrow(text_data)) rows)")
|
||||||
|
println(" expert.csv — training only ($(nrow(expert_train)) rows)")
|
||||||
|
println(" lr_data.csv — training only ($(nrow(lr_train)) rows)")
|
||||||
|
println(" expert_test.csv — held-out ($(nrow(expert_test)) rows)")
|
||||||
|
println(" lr_data_test.csv — held-out ($(nrow(lr_test)) rows)")
|
||||||
|
|
||||||
|
return output_dir
|
||||||
|
end
|
||||||
|
|
||||||
|
# =========================================================================
|
||||||
|
# STEP 2: Compute held-out validation metrics
|
||||||
|
# =========================================================================
|
||||||
|
|
||||||
|
function compute_holdout_metrics(model_file::String, test_dir::String)
|
||||||
|
println()
|
||||||
|
println("="^70)
|
||||||
|
println("COMPUTING HELD-OUT VALIDATION METRICS")
|
||||||
|
println("="^70)
|
||||||
|
|
||||||
|
# Load model output
|
||||||
|
model = CSV.read(model_file, DataFrame)
|
||||||
|
party_col = hasproperty(model, :party_id) ? :party_id : :party
|
||||||
|
println("Model output: $(nrow(model)) party-years")
|
||||||
|
|
||||||
|
# Load test data
|
||||||
|
expert_test = CSV.read(joinpath(test_dir, "expert_test.csv"), DataFrame)
|
||||||
|
lr_test = CSV.read(joinpath(test_dir, "lr_data_test.csv"), DataFrame)
|
||||||
|
println("Held-out expert: $(nrow(expert_test)) observations")
|
||||||
|
println("Held-out LR: $(nrow(lr_test)) observations")
|
||||||
|
|
||||||
|
# Build lookup: (party, year) → model estimates
|
||||||
|
model_lookup = Dict{Tuple{Int,Int}, NamedTuple}()
|
||||||
|
for row in eachrow(model)
|
||||||
|
key = (row[party_col], row.year)
|
||||||
|
model_lookup[key] = (
|
||||||
|
economic_lr = row.economic_lr,
|
||||||
|
galtan = row.galtan,
|
||||||
|
economic_lr_se = hasproperty(row, :economic_lr_se) ? row.economic_lr_se : missing,
|
||||||
|
galtan_se = hasproperty(row, :galtan_se) ? row.galtan_se : missing,
|
||||||
|
economic_lr_q025 = hasproperty(row, :economic_lr_q025) ? row.economic_lr_q025 : missing,
|
||||||
|
economic_lr_q975 = hasproperty(row, :economic_lr_q975) ? row.economic_lr_q975 : missing,
|
||||||
|
galtan_q025 = hasproperty(row, :galtan_q025) ? row.galtan_q025 : missing,
|
||||||
|
galtan_q975 = hasproperty(row, :galtan_q975) ? row.galtan_q975 : missing,
|
||||||
|
)
|
||||||
|
end
|
||||||
|
|
||||||
|
# Map expert variables to dimensions
|
||||||
|
econ_vars = Set(["lrecon_ches", "lrecon_poppa", "lrecon_gps", "lrecon_vparty", "welf_vparty"])
|
||||||
|
galtan_vars = Set(["galtan_ches", "libcon_gps", "immig_vparty", "lgbt_vparty",
|
||||||
|
"culsup_vparty", "relig_vparty", "gender_vparty"])
|
||||||
|
|
||||||
|
# Process expert test observations
|
||||||
|
results = NamedTuple[]
|
||||||
|
|
||||||
|
for row in eachrow(expert_test)
|
||||||
|
key = (row.party, row.year)
|
||||||
|
haskey(model_lookup, key) || continue
|
||||||
|
m = model_lookup[key]
|
||||||
|
|
||||||
|
if row.var in econ_vars
|
||||||
|
dim = "economic_lr"
|
||||||
|
model_val = m.economic_lr
|
||||||
|
model_q025 = m.economic_lr_q025
|
||||||
|
model_q975 = m.economic_lr_q975
|
||||||
|
elseif row.var in galtan_vars
|
||||||
|
dim = "galtan"
|
||||||
|
model_val = m.galtan
|
||||||
|
model_q025 = m.galtan_q025
|
||||||
|
model_q975 = m.galtan_q975
|
||||||
|
else
|
||||||
|
continue
|
||||||
|
end
|
||||||
|
|
||||||
|
covered = !ismissing(model_q025) && !ismissing(model_q975) &&
|
||||||
|
row.val >= model_q025 && row.val <= model_q975
|
||||||
|
|
||||||
|
push!(results, (
|
||||||
|
party = row.party,
|
||||||
|
country = row.country,
|
||||||
|
year = row.year,
|
||||||
|
var = row.var,
|
||||||
|
dimension = dim,
|
||||||
|
expert_val = row.val,
|
||||||
|
model_val = model_val,
|
||||||
|
error = row.val - model_val,
|
||||||
|
abs_error = abs(row.val - model_val),
|
||||||
|
covered_95 = ismissing(model_q025) ? missing : covered,
|
||||||
|
))
|
||||||
|
end
|
||||||
|
|
||||||
|
if isempty(results)
|
||||||
|
println("ERROR: No matching test observations found")
|
||||||
|
return nothing
|
||||||
|
end
|
||||||
|
|
||||||
|
results_df = DataFrame(results)
|
||||||
|
|
||||||
|
# Compute and report metrics
|
||||||
|
println()
|
||||||
|
println("-"^70)
|
||||||
|
println("HELD-OUT VALIDATION RESULTS")
|
||||||
|
println("-"^70)
|
||||||
|
|
||||||
|
# Overall
|
||||||
|
overall_r = cor(results_df.expert_val, results_df.model_val)
|
||||||
|
overall_mae = mean(results_df.abs_error)
|
||||||
|
overall_rmse = sqrt(mean(results_df.error .^ 2))
|
||||||
|
println()
|
||||||
|
println(@sprintf("Overall: r=%.4f, MAE=%.4f, RMSE=%.4f, n=%d",
|
||||||
|
overall_r, overall_mae, overall_rmse, nrow(results_df)))
|
||||||
|
|
||||||
|
# By dimension
|
||||||
|
println()
|
||||||
|
println("By dimension:")
|
||||||
|
for dim in sort(unique(results_df.dimension))
|
||||||
|
d = filter(r -> r.dimension == dim, results_df)
|
||||||
|
r_val = cor(d.expert_val, d.model_val)
|
||||||
|
mae = mean(d.abs_error)
|
||||||
|
rmse = sqrt(mean(d.error .^ 2))
|
||||||
|
cov = count(skipmissing(d.covered_95)) / count(!ismissing, d.covered_95)
|
||||||
|
println(@sprintf(" %-12s: r=%.4f, MAE=%.4f, RMSE=%.4f, CIC95=%.1f%%, n=%d",
|
||||||
|
dim, r_val, mae, rmse, 100*cov, nrow(d)))
|
||||||
|
end
|
||||||
|
|
||||||
|
# By source
|
||||||
|
println()
|
||||||
|
println("By source:")
|
||||||
|
for var in sort(unique(results_df.var))
|
||||||
|
v = filter(r -> r.var == var, results_df)
|
||||||
|
nrow(v) < 5 && continue
|
||||||
|
r_val = cor(v.expert_val, v.model_val)
|
||||||
|
mae = mean(v.abs_error)
|
||||||
|
println(@sprintf(" %-20s: r=%.4f, MAE=%.4f, n=%d", var, r_val, mae, nrow(v)))
|
||||||
|
end
|
||||||
|
|
||||||
|
return results_df
|
||||||
|
end
|
||||||
|
|
||||||
|
# =========================================================================
|
||||||
|
# Main
|
||||||
|
# =========================================================================
|
||||||
|
|
||||||
|
function main()
|
||||||
|
args = ARGS
|
||||||
|
|
||||||
|
if isempty(args) || args[1] == "prepare"
|
||||||
|
output_dir = prepare_holdout_data()
|
||||||
|
|
||||||
|
println()
|
||||||
|
println("="^70)
|
||||||
|
println("NEXT STEPS")
|
||||||
|
println("="^70)
|
||||||
|
println("""
|
||||||
|
1. Run model on training data:
|
||||||
|
julia 01_run_model.jl --data-dir $output_dir
|
||||||
|
|
||||||
|
2. Run post-estimation on training run output:
|
||||||
|
julia 02_post_estimation.jl
|
||||||
|
|
||||||
|
3. Compute held-out metrics:
|
||||||
|
julia scripts/validate_external.jl compute <party_positions_file.csv>
|
||||||
|
""")
|
||||||
|
|
||||||
|
elseif args[1] == "compute" && length(args) >= 2
|
||||||
|
model_file = args[2]
|
||||||
|
test_dir = joinpath("validation", "external_split")
|
||||||
|
|
||||||
|
isfile(model_file) || error("Model file not found: $model_file")
|
||||||
|
isdir(test_dir) || error("Test data not found. Run 'prepare' first.")
|
||||||
|
|
||||||
|
results_df = compute_holdout_metrics(model_file, test_dir)
|
||||||
|
|
||||||
|
if results_df !== nothing
|
||||||
|
timestamp = Dates.format(now(), "yyyy-mm-dd_HH-MM-SS")
|
||||||
|
output_file = joinpath("validation", "external_validation_$(timestamp).csv")
|
||||||
|
CSV.write(output_file, results_df)
|
||||||
|
println()
|
||||||
|
println("Detailed results saved: $output_file")
|
||||||
|
end
|
||||||
|
|
||||||
|
else
|
||||||
|
println("Usage:")
|
||||||
|
println(" julia scripts/validate_external.jl prepare")
|
||||||
|
println(" julia scripts/validate_external.jl compute <party_positions.csv>")
|
||||||
|
end
|
||||||
|
end
|
||||||
|
|
||||||
|
if abspath(PROGRAM_FILE) == @__FILE__
|
||||||
|
main()
|
||||||
|
end
|
||||||
@@ -0,0 +1,681 @@
|
|||||||
|
#!/usr/bin/env julia
|
||||||
|
#############################################################################
|
||||||
|
## validate_uncertainty.jl
|
||||||
|
## Uncertainty validation: Posterior Predictive Coverage (PPC)
|
||||||
|
##
|
||||||
|
## Following Claassen (2019), this script computes:
|
||||||
|
## - PPC: What % of expert values fall within 95% posterior predictive interval?
|
||||||
|
## - Also computes 80% PPC for direct Claassen comparison (he reports 60.3%)
|
||||||
|
## - Wilson score CI for coverage proportion
|
||||||
|
## - Breakdown by survey source and decade
|
||||||
|
##
|
||||||
|
## Unlike credible interval coverage (which checks θ-CIs), posterior predictive
|
||||||
|
## coverage simulates what a new expert observation would look like given the
|
||||||
|
## model's beta likelihood, accounting for both position uncertainty AND
|
||||||
|
## measurement noise. Well-calibrated models should yield ~95% PPC at 95%.
|
||||||
|
##
|
||||||
|
## No model re-run needed: reads θ, γ, and φ from existing chain CSV files.
|
||||||
|
#############################################################################
|
||||||
|
|
||||||
|
using CSV, DataFrames, Statistics, Dates, Printf, Random, JSON
|
||||||
|
|
||||||
|
# =============================================================================
|
||||||
|
# Utility: Wilson score CI for a proportion
|
||||||
|
# =============================================================================
|
||||||
|
|
||||||
|
function wilson_ci(p, n; alpha=0.05)
|
||||||
|
if n == 0
|
||||||
|
return (lower=NaN, upper=NaN, se=NaN)
|
||||||
|
end
|
||||||
|
z = 1.96 # For 95% CI
|
||||||
|
denominator = 1 + z^2/n
|
||||||
|
center = (p + z^2/(2n)) / denominator
|
||||||
|
margin = z * sqrt((p*(1-p) + z^2/(4n))/n) / denominator
|
||||||
|
se = sqrt(p * (1-p) / n)
|
||||||
|
return (lower=center - margin, upper=center + margin, se=se)
|
||||||
|
end
|
||||||
|
|
||||||
|
# =============================================================================
|
||||||
|
# STEP 0: Find latest model run
|
||||||
|
# =============================================================================
|
||||||
|
|
||||||
|
function find_latest_run(base_dir::String="model_outputs")
|
||||||
|
if !isdir(base_dir)
|
||||||
|
error("Model outputs directory not found: $base_dir")
|
||||||
|
end
|
||||||
|
runs = filter(d -> startswith(d, "run_") && isdir(joinpath(base_dir, d)), readdir(base_dir))
|
||||||
|
if isempty(runs)
|
||||||
|
error("No runs found in $base_dir")
|
||||||
|
end
|
||||||
|
sort!(runs, rev=true)
|
||||||
|
latest = joinpath(base_dir, runs[1])
|
||||||
|
println("Using latest run: $latest")
|
||||||
|
return latest
|
||||||
|
end
|
||||||
|
|
||||||
|
# =============================================================================
|
||||||
|
# STEP 1: Load expert_dim.csv from model run data
|
||||||
|
# =============================================================================
|
||||||
|
|
||||||
|
function load_expert_dim(run_dir::String)
|
||||||
|
expert_dim_file = joinpath(run_dir, "data", "expert_dim.csv")
|
||||||
|
if !isfile(expert_dim_file)
|
||||||
|
error("expert_dim.csv not found in $run_dir/data/")
|
||||||
|
end
|
||||||
|
expert_dim = CSV.read(expert_dim_file, DataFrame)
|
||||||
|
println("Loaded expert_dim.csv: $(nrow(expert_dim)) observations")
|
||||||
|
println(" Unique rr values: $(length(unique(expert_dim.rr_exp_dim)))")
|
||||||
|
println(" Item indices (var_exp_dim): $(sort(unique(expert_dim.var_exp_dim)))")
|
||||||
|
println(" Dimensions (dim_idx_exp): $(sort(unique(expert_dim.dim_idx_exp)))")
|
||||||
|
return expert_dim
|
||||||
|
end
|
||||||
|
|
||||||
|
# =============================================================================
|
||||||
|
# STEP 2: Selectively load chain columns
|
||||||
|
# =============================================================================
|
||||||
|
|
||||||
|
function load_chains_selective(run_dir::String, needed_rr::Set{Int}, K::Int)
|
||||||
|
"""Load only the chain columns we need for posterior predictive checks."""
|
||||||
|
chains_dir = joinpath(run_dir, "chains")
|
||||||
|
chain_files = sort(filter(f -> endswith(f, ".csv") && startswith(f, "chain_"), readdir(chains_dir)))
|
||||||
|
|
||||||
|
if isempty(chain_files)
|
||||||
|
error("No chain files found in $chains_dir")
|
||||||
|
end
|
||||||
|
println("\nLoading $(length(chain_files)) chain files (selective columns)...")
|
||||||
|
|
||||||
|
# Build the set of column names we need
|
||||||
|
needed_cols = Set{String}()
|
||||||
|
|
||||||
|
# theta columns: economic_lr.{rr} and galtan.{rr} (cultural cosmopolitan--traditionalist) for each unique rr
|
||||||
|
for rr in needed_rr
|
||||||
|
push!(needed_cols, "economic_lr.$rr")
|
||||||
|
push!(needed_cols, "galtan.$rr")
|
||||||
|
end
|
||||||
|
|
||||||
|
# Item parameters: gamma_exp_intercept.1-K, gamma_exp_slope.1-K
|
||||||
|
for k in 1:K
|
||||||
|
push!(needed_cols, "gamma_exp_intercept.$k")
|
||||||
|
push!(needed_cols, "gamma_exp_slope.$k")
|
||||||
|
end
|
||||||
|
|
||||||
|
# Precision parameter
|
||||||
|
push!(needed_cols, "phi_exp_dim")
|
||||||
|
|
||||||
|
println(" Need $(length(needed_cols)) columns ($(length(needed_rr)) rr × 2 dims + $(2*K) item params + 1 phi)")
|
||||||
|
|
||||||
|
# Read the header from first chain to identify column indices
|
||||||
|
first_chain_path = joinpath(chains_dir, chain_files[1])
|
||||||
|
header_line = ""
|
||||||
|
open(first_chain_path) do f
|
||||||
|
for line in eachline(f)
|
||||||
|
if !startswith(line, "#")
|
||||||
|
header_line = line
|
||||||
|
break
|
||||||
|
end
|
||||||
|
end
|
||||||
|
end
|
||||||
|
|
||||||
|
all_cols = split(header_line, ",")
|
||||||
|
col_indices = Int[]
|
||||||
|
col_names = String[]
|
||||||
|
|
||||||
|
for (i, col) in enumerate(all_cols)
|
||||||
|
if col in needed_cols
|
||||||
|
push!(col_indices, i)
|
||||||
|
push!(col_names, col)
|
||||||
|
end
|
||||||
|
end
|
||||||
|
|
||||||
|
println(" Found $(length(col_indices))/$(length(needed_cols)) columns in chains")
|
||||||
|
|
||||||
|
if length(col_indices) < length(needed_cols)
|
||||||
|
missing_cols = setdiff(needed_cols, Set(col_names))
|
||||||
|
n_missing = length(missing_cols)
|
||||||
|
sample = collect(missing_cols)[1:min(5, n_missing)]
|
||||||
|
println(" WARNING: Missing columns (showing $( min(5, n_missing))/$n_missing): $sample")
|
||||||
|
end
|
||||||
|
|
||||||
|
# Build a type specification for selective reading
|
||||||
|
# We'll use CSV.read with select parameter
|
||||||
|
select_symbols = Symbol.(col_names)
|
||||||
|
|
||||||
|
all_chains = DataFrame[]
|
||||||
|
for (i, cf) in enumerate(chain_files)
|
||||||
|
path = joinpath(chains_dir, cf)
|
||||||
|
print(" Loading chain $i: $(cf)... ")
|
||||||
|
t = @elapsed begin
|
||||||
|
chain = CSV.read(path, DataFrame; comment="#", select=select_symbols)
|
||||||
|
end
|
||||||
|
println("$(nrow(chain)) samples, $(round(t, digits=1))s")
|
||||||
|
push!(all_chains, chain)
|
||||||
|
end
|
||||||
|
|
||||||
|
combined = vcat(all_chains...)
|
||||||
|
println("Combined: $(nrow(combined)) total posterior draws")
|
||||||
|
return combined
|
||||||
|
end
|
||||||
|
|
||||||
|
# =============================================================================
|
||||||
|
# STEP 3: Compute posterior predictive coverage
|
||||||
|
# =============================================================================
|
||||||
|
|
||||||
|
function compute_posterior_predictive_cic(chains::DataFrame, expert_dim::DataFrame;
|
||||||
|
ci_level::Float64=0.95,
|
||||||
|
calibration_levels::Vector{Float64}=[0.50, 0.80, 0.90, 0.95],
|
||||||
|
seed::Int=42)
|
||||||
|
"""
|
||||||
|
Compute posterior predictive coverage for expert dimension observations.
|
||||||
|
|
||||||
|
For each expert observation n with observed value y_n:
|
||||||
|
1. For each posterior draw s:
|
||||||
|
- Get theta_s = theta[dim, rr] (on logit scale, but chains store inv_logit)
|
||||||
|
- Compute mu_s = invlogit(gamma_intercept[k] + gamma_slope[k] * logit(theta_s))
|
||||||
|
- V4 (Beta): Draw y_pred_s ~ Beta(phi * mu_s, phi * (1 - mu_s))
|
||||||
|
- V5 (Beta-Binomial): Draw y_pred_s ~ Beta(phi * K * mu_s, phi * K * (1 - mu_s))
|
||||||
|
where K = n_experts for that observation
|
||||||
|
2. Compute quantile interval of y_pred draws
|
||||||
|
3. Check if y_n falls within interval
|
||||||
|
|
||||||
|
Returns DataFrame with one row per observation plus coverage indicator.
|
||||||
|
"""
|
||||||
|
rng = MersenneTwister(seed)
|
||||||
|
|
||||||
|
alpha_lower = (1 - ci_level) / 2
|
||||||
|
alpha_upper = 1 - alpha_lower
|
||||||
|
|
||||||
|
N = nrow(expert_dim)
|
||||||
|
S = nrow(chains) # total posterior draws
|
||||||
|
|
||||||
|
# Detect V5 (Beta-Binomial with K-scaling) by presence of n_experts column
|
||||||
|
has_k_scaling = hasproperty(expert_dim, :n_experts)
|
||||||
|
if has_k_scaling
|
||||||
|
k_vec = expert_dim.n_experts
|
||||||
|
println("\nV5 detected: using Beta(phi*K*mu, phi*K*(1-mu)) with per-observation K")
|
||||||
|
else
|
||||||
|
println("\nV4 detected: using Beta(phi*mu, phi*(1-mu))")
|
||||||
|
end
|
||||||
|
|
||||||
|
println("Computing posterior predictive coverage ($(round(Int, 100*ci_level))% level)")
|
||||||
|
println(" Expert observations: $N")
|
||||||
|
println(" Posterior draws: $S")
|
||||||
|
|
||||||
|
# Pre-extract phi vector
|
||||||
|
phi_vec = chains[!, :phi_exp_dim]
|
||||||
|
|
||||||
|
# Pre-extract gamma vectors for each item k
|
||||||
|
K = maximum(expert_dim.var_exp_dim)
|
||||||
|
gamma_int = Dict{Int, Vector{Float64}}()
|
||||||
|
gamma_slope = Dict{Int, Vector{Float64}}()
|
||||||
|
for k in 1:K
|
||||||
|
col_int = Symbol("gamma_exp_intercept.$k")
|
||||||
|
col_slope = Symbol("gamma_exp_slope.$k")
|
||||||
|
if hasproperty(chains, col_int) && hasproperty(chains, col_slope)
|
||||||
|
gamma_int[k] = chains[!, col_int]
|
||||||
|
gamma_slope[k] = chains[!, col_slope]
|
||||||
|
end
|
||||||
|
end
|
||||||
|
|
||||||
|
# Allocate result columns
|
||||||
|
covered = BitVector(undef, N)
|
||||||
|
pred_lower = Vector{Float64}(undef, N)
|
||||||
|
pred_upper = Vector{Float64}(undef, N)
|
||||||
|
pred_median = Vector{Float64}(undef, N)
|
||||||
|
calibration_covered = Dict(level => BitVector(undef, N) for level in calibration_levels)
|
||||||
|
|
||||||
|
# Pre-allocate per-observation draw buffer
|
||||||
|
y_pred = Vector{Float64}(undef, S)
|
||||||
|
|
||||||
|
prog_interval = max(1, N ÷ 20)
|
||||||
|
|
||||||
|
for n in 1:N
|
||||||
|
if n % prog_interval == 0 || n == N
|
||||||
|
pct = round(100 * n / N, digits=1)
|
||||||
|
print("\r Progress: $pct% ($n / $N)")
|
||||||
|
end
|
||||||
|
|
||||||
|
rr = expert_dim.rr_exp_dim[n]
|
||||||
|
dim = expert_dim.dim_idx_exp[n]
|
||||||
|
k = expert_dim.var_exp_dim[n]
|
||||||
|
y_obs = expert_dim.val[n]
|
||||||
|
|
||||||
|
# Get theta column (chains store inv_logit(theta), i.e. on [0,1] scale)
|
||||||
|
theta_col = dim == 1 ? Symbol("economic_lr.$rr") : Symbol("galtan.$rr")
|
||||||
|
|
||||||
|
if !hasproperty(chains, theta_col) || !haskey(gamma_int, k)
|
||||||
|
# Missing chain data — mark as not covered
|
||||||
|
covered[n] = false
|
||||||
|
pred_lower[n] = NaN
|
||||||
|
pred_upper[n] = NaN
|
||||||
|
pred_median[n] = NaN
|
||||||
|
for level in calibration_levels
|
||||||
|
calibration_covered[level][n] = false
|
||||||
|
end
|
||||||
|
continue
|
||||||
|
end
|
||||||
|
|
||||||
|
theta_star_vec = chains[!, theta_col] # inv_logit(theta), i.e. on [0,1]
|
||||||
|
g_int = gamma_int[k]
|
||||||
|
g_slope = gamma_slope[k]
|
||||||
|
|
||||||
|
# Effective concentration: phi for V4, phi * n_experts for V5
|
||||||
|
k_mult = has_k_scaling ? Float64(k_vec[n]) : 1.0
|
||||||
|
|
||||||
|
# For each posterior draw, simulate a predictive observation
|
||||||
|
for s in 1:S
|
||||||
|
theta_star = theta_star_vec[s]
|
||||||
|
|
||||||
|
# Convert back to latent scale for linear predictor
|
||||||
|
# theta_star is inv_logit(theta), so theta = logit(theta_star)
|
||||||
|
# Clamp to avoid Inf
|
||||||
|
theta_star_clamped = clamp(theta_star, 1e-10, 1 - 1e-10)
|
||||||
|
theta_latent = log(theta_star_clamped / (1 - theta_star_clamped))
|
||||||
|
|
||||||
|
# Linear predictor
|
||||||
|
lin = g_int[s] + g_slope[s] * theta_latent
|
||||||
|
|
||||||
|
# Mean of beta
|
||||||
|
mu = 1 / (1 + exp(-lin))
|
||||||
|
mu = clamp(mu, 1e-6, 1 - 1e-6)
|
||||||
|
|
||||||
|
# Beta parameters: phi * K * mu for V5, phi * mu for V4
|
||||||
|
phi = phi_vec[s] * k_mult
|
||||||
|
a = phi * mu
|
||||||
|
b = phi * (1 - mu)
|
||||||
|
|
||||||
|
# Draw from Beta(a, b) via gamma method (no Distributions.jl needed)
|
||||||
|
y_pred[s] = _rand_beta(rng, a, b)
|
||||||
|
end
|
||||||
|
|
||||||
|
# Compute predictive interval
|
||||||
|
sort!(y_pred)
|
||||||
|
idx_lo = max(1, round(Int, alpha_lower * S))
|
||||||
|
idx_hi = min(S, round(Int, alpha_upper * S))
|
||||||
|
idx_med = round(Int, 0.5 * S)
|
||||||
|
|
||||||
|
pred_lower[n] = y_pred[idx_lo]
|
||||||
|
pred_upper[n] = y_pred[idx_hi]
|
||||||
|
pred_median[n] = y_pred[idx_med]
|
||||||
|
covered[n] = (y_obs >= pred_lower[n]) && (y_obs <= pred_upper[n])
|
||||||
|
|
||||||
|
# Compute calibration at several nominal levels from the same predictive
|
||||||
|
# draws. This avoids repeating the expensive simulation and chain read.
|
||||||
|
for level in calibration_levels
|
||||||
|
level_alpha = (1 - level) / 2
|
||||||
|
level_lo = max(1, round(Int, level_alpha * S))
|
||||||
|
level_hi = min(S, round(Int, (1 - level_alpha) * S))
|
||||||
|
calibration_covered[level][n] =
|
||||||
|
(y_obs >= y_pred[level_lo]) && (y_obs <= y_pred[level_hi])
|
||||||
|
end
|
||||||
|
end
|
||||||
|
println() # newline after progress
|
||||||
|
|
||||||
|
# Add results to a copy of expert_dim
|
||||||
|
result = DataFrame(
|
||||||
|
rr = expert_dim.rr_exp_dim,
|
||||||
|
dim_idx = expert_dim.dim_idx_exp,
|
||||||
|
var_idx = expert_dim.var_exp_dim,
|
||||||
|
val = expert_dim.val,
|
||||||
|
party = expert_dim.party,
|
||||||
|
country = expert_dim.country,
|
||||||
|
year = expert_dim.year,
|
||||||
|
project = expert_dim.project,
|
||||||
|
var = expert_dim.var,
|
||||||
|
pred_lower = pred_lower,
|
||||||
|
pred_upper = pred_upper,
|
||||||
|
pred_median = pred_median,
|
||||||
|
covered = covered
|
||||||
|
)
|
||||||
|
|
||||||
|
for level in calibration_levels
|
||||||
|
suffix = lpad(string(round(Int, 100 * level)), 2, '0')
|
||||||
|
result[!, Symbol("covered_$(suffix)")] = calibration_covered[level]
|
||||||
|
end
|
||||||
|
|
||||||
|
return result
|
||||||
|
end
|
||||||
|
|
||||||
|
function coverage_breakdown(result::DataFrame, group_cols::Vector{Symbol};
|
||||||
|
covered_col::Symbol=:covered_95)
|
||||||
|
total_misses = count(.!result[!, covered_col])
|
||||||
|
grouped = combine(groupby(result, group_cols)) do df
|
||||||
|
n = nrow(df)
|
||||||
|
n_covered = count(df[!, covered_col])
|
||||||
|
n_missed = n - n_covered
|
||||||
|
p = n_covered / n
|
||||||
|
ci = wilson_ci(p, n)
|
||||||
|
DataFrame(
|
||||||
|
n = n,
|
||||||
|
covered = n_covered,
|
||||||
|
missed = n_missed,
|
||||||
|
observed_coverage = p,
|
||||||
|
coverage_ci_lower = ci.lower,
|
||||||
|
coverage_ci_upper = ci.upper,
|
||||||
|
mean_interval_width = mean(df.pred_upper .- df.pred_lower),
|
||||||
|
share_of_all_misses = total_misses == 0 ? 0.0 : n_missed / total_misses
|
||||||
|
)
|
||||||
|
end
|
||||||
|
sort!(grouped, [:missed, :n], rev=true)
|
||||||
|
return grouped
|
||||||
|
end
|
||||||
|
|
||||||
|
function save_detailed_results(result::DataFrame, output_dir::String)
|
||||||
|
mkpath(output_dir)
|
||||||
|
result_with_decade = copy(result)
|
||||||
|
result_with_decade.decade = div.(result_with_decade.year, 10) .* 10
|
||||||
|
|
||||||
|
CSV.write(joinpath(output_dir, "posterior_predictive_observations.csv"), result_with_decade)
|
||||||
|
CSV.write(joinpath(output_dir, "posterior_predictive_by_dimension_source.csv"),
|
||||||
|
coverage_breakdown(result_with_decade, [:dim_idx, :project]))
|
||||||
|
CSV.write(joinpath(output_dir, "posterior_predictive_by_dimension_item.csv"),
|
||||||
|
coverage_breakdown(result_with_decade, [:dim_idx, :project, :var]))
|
||||||
|
CSV.write(joinpath(output_dir, "posterior_predictive_by_dimension_decade.csv"),
|
||||||
|
coverage_breakdown(result_with_decade, [:dim_idx, :decade]))
|
||||||
|
CSV.write(joinpath(output_dir, "posterior_predictive_by_dimension_country.csv"),
|
||||||
|
coverage_breakdown(result_with_decade, [:dim_idx, :country]))
|
||||||
|
|
||||||
|
calibration_rows = NamedTuple[]
|
||||||
|
for dim in sort(unique(result_with_decade.dim_idx))
|
||||||
|
subset = filter(:dim_idx => ==(dim), result_with_decade)
|
||||||
|
for level in (50, 80, 90, 95)
|
||||||
|
col = Symbol("covered_$(level)")
|
||||||
|
n = nrow(subset)
|
||||||
|
n_covered = count(subset[!, col])
|
||||||
|
push!(calibration_rows, (
|
||||||
|
dim_idx=dim,
|
||||||
|
nominal_level=level / 100,
|
||||||
|
observed_coverage=n_covered / n,
|
||||||
|
covered=n_covered,
|
||||||
|
n=n
|
||||||
|
))
|
||||||
|
end
|
||||||
|
end
|
||||||
|
CSV.write(joinpath(output_dir, "posterior_predictive_calibration_curve.csv"),
|
||||||
|
DataFrame(calibration_rows))
|
||||||
|
println("Saved detailed posterior-predictive results to: $output_dir")
|
||||||
|
end
|
||||||
|
|
||||||
|
# =============================================================================
|
||||||
|
# Beta random variate without Distributions.jl
|
||||||
|
# =============================================================================
|
||||||
|
|
||||||
|
"""
|
||||||
|
_rand_beta(rng, a, b)
|
||||||
|
|
||||||
|
Generate a Beta(a, b) random variate using the Gamma method:
|
||||||
|
Beta(a,b) = X/(X+Y) where X ~ Gamma(a), Y ~ Gamma(b).
|
||||||
|
Uses Marsaglia & Tsang (2000) for Gamma generation.
|
||||||
|
"""
|
||||||
|
function _rand_beta(rng::AbstractRNG, a::Float64, b::Float64)
|
||||||
|
x = _rand_gamma(rng, a)
|
||||||
|
y = _rand_gamma(rng, b)
|
||||||
|
return x / (x + y)
|
||||||
|
end
|
||||||
|
|
||||||
|
"""
|
||||||
|
_rand_gamma(rng, shape)
|
||||||
|
|
||||||
|
Generate Gamma(shape, 1) random variate using Marsaglia & Tsang (2000).
|
||||||
|
For shape < 1, uses the rejection method with shape+1 then scales.
|
||||||
|
"""
|
||||||
|
function _rand_gamma(rng::AbstractRNG, shape::Float64)
|
||||||
|
if shape < 1.0
|
||||||
|
# Gamma(a) = Gamma(a+1) * U^(1/a) where U ~ Uniform(0,1)
|
||||||
|
return _rand_gamma(rng, shape + 1.0) * rand(rng)^(1.0 / shape)
|
||||||
|
end
|
||||||
|
|
||||||
|
# Marsaglia & Tsang (2000) for shape >= 1
|
||||||
|
d = shape - 1.0/3.0
|
||||||
|
c = 1.0 / sqrt(9.0 * d)
|
||||||
|
|
||||||
|
while true
|
||||||
|
local x::Float64
|
||||||
|
local v::Float64
|
||||||
|
|
||||||
|
while true
|
||||||
|
x = randn(rng)
|
||||||
|
v = 1.0 + c * x
|
||||||
|
if v > 0.0
|
||||||
|
break
|
||||||
|
end
|
||||||
|
end
|
||||||
|
|
||||||
|
v = v * v * v
|
||||||
|
u = rand(rng)
|
||||||
|
|
||||||
|
if u < 1.0 - 0.0331 * x^2 * x^2
|
||||||
|
return d * v
|
||||||
|
end
|
||||||
|
|
||||||
|
if log(u) < 0.5 * x^2 + d * (1.0 - v + log(v))
|
||||||
|
return d * v
|
||||||
|
end
|
||||||
|
end
|
||||||
|
end
|
||||||
|
|
||||||
|
# =============================================================================
|
||||||
|
# STEP 4: Summarize and save results
|
||||||
|
# =============================================================================
|
||||||
|
|
||||||
|
function summarize_coverage(result::DataFrame, ci_level::Float64)
|
||||||
|
level_pct = round(Int, 100 * ci_level)
|
||||||
|
println("\n" * "="^60)
|
||||||
|
println("POSTERIOR PREDICTIVE COVERAGE ($level_pct%)")
|
||||||
|
println("="^60)
|
||||||
|
|
||||||
|
# Overall by dimension
|
||||||
|
dim_names = Dict(1 => "economic_lr", 2 => "galtan")
|
||||||
|
display_dim_names = Dict(1 => "economic left-right", 2 => "cultural cosmopolitan--traditionalist")
|
||||||
|
|
||||||
|
summary_rows = []
|
||||||
|
for dim in sort(unique(result.dim_idx))
|
||||||
|
subset = filter(r -> r.dim_idx == dim, result)
|
||||||
|
n = nrow(subset)
|
||||||
|
n_covered = sum(subset.covered)
|
||||||
|
ppc = n_covered / n
|
||||||
|
ci = wilson_ci(ppc, n)
|
||||||
|
dim_name = dim_names[dim]
|
||||||
|
|
||||||
|
println(@sprintf("\n %-40s: %.1f%% [%.1f%%, %.1f%%] (%d/%d)",
|
||||||
|
display_dim_names[dim], 100*ppc, 100*ci.lower, 100*ci.upper, n_covered, n))
|
||||||
|
|
||||||
|
push!(summary_rows, (
|
||||||
|
dimension = dim_name,
|
||||||
|
cic = ppc,
|
||||||
|
cic_pct = round(100 * ppc, digits=1),
|
||||||
|
ci_lower = ci.lower,
|
||||||
|
ci_upper = ci.upper,
|
||||||
|
n = n,
|
||||||
|
covered = n_covered
|
||||||
|
))
|
||||||
|
|
||||||
|
# By project
|
||||||
|
println("\n By survey source:")
|
||||||
|
by_project = combine(groupby(subset, :project)) do df
|
||||||
|
nc = sum(df.covered)
|
||||||
|
DataFrame(n = nrow(df), covered = nc, cic = nc / nrow(df))
|
||||||
|
end
|
||||||
|
sort!(by_project, :n, rev=true)
|
||||||
|
|
||||||
|
@printf(" %-12s %6s %8s\n", "Project", "N", "PPC")
|
||||||
|
for row in eachrow(by_project)
|
||||||
|
@printf(" %-12s %6d %7.1f%%\n", row.project, row.n, 100*row.cic)
|
||||||
|
end
|
||||||
|
|
||||||
|
# By decade
|
||||||
|
subset_with_decade = copy(subset)
|
||||||
|
subset_with_decade.decade = div.(subset_with_decade.year, 10) .* 10
|
||||||
|
println("\n By decade:")
|
||||||
|
by_decade = combine(groupby(subset_with_decade, :decade)) do df
|
||||||
|
nc = sum(df.covered)
|
||||||
|
DataFrame(n = nrow(df), covered = nc, cic = nc / nrow(df))
|
||||||
|
end
|
||||||
|
sort!(by_decade, :decade)
|
||||||
|
|
||||||
|
@printf(" %-8s %6s %8s\n", "Decade", "N", "PPC")
|
||||||
|
for row in eachrow(by_decade)
|
||||||
|
@printf(" %-8d %6d %7.1f%%\n", row.decade, row.n, 100*row.cic)
|
||||||
|
end
|
||||||
|
end
|
||||||
|
|
||||||
|
return summary_rows
|
||||||
|
end
|
||||||
|
|
||||||
|
function save_results(result_95::DataFrame, summary_95, summary_80,
|
||||||
|
by_project_95::Dict, output_dir::String="validation")
|
||||||
|
if !isdir(output_dir)
|
||||||
|
mkpath(output_dir)
|
||||||
|
end
|
||||||
|
|
||||||
|
timestamp = Dates.format(now(), "yyyy-mm-dd_HH-MM-SS")
|
||||||
|
|
||||||
|
# Summary table (95%)
|
||||||
|
if !isempty(summary_95)
|
||||||
|
summary_df = DataFrame(summary_95)
|
||||||
|
summary_file = joinpath(output_dir, "uncertainty_cic_summary_$timestamp.csv")
|
||||||
|
CSV.write(summary_file, summary_df)
|
||||||
|
println("\nSaved: $summary_file")
|
||||||
|
end
|
||||||
|
|
||||||
|
# Also save 80% summary
|
||||||
|
if !isempty(summary_80)
|
||||||
|
summary80_df = DataFrame(summary_80)
|
||||||
|
summary80_file = joinpath(output_dir, "uncertainty_cic_80pct_summary_$timestamp.csv")
|
||||||
|
CSV.write(summary80_file, summary80_df)
|
||||||
|
println("Saved: $summary80_file")
|
||||||
|
end
|
||||||
|
|
||||||
|
# By-project tables (95%)
|
||||||
|
dim_names = Dict(1 => "economic_lr", 2 => "galtan")
|
||||||
|
for (dim, bp) in by_project_95
|
||||||
|
project_file = joinpath(output_dir, "uncertainty_$(dim_names[dim])_by_project_$timestamp.csv")
|
||||||
|
CSV.write(project_file, bp)
|
||||||
|
println("Saved: $project_file")
|
||||||
|
end
|
||||||
|
|
||||||
|
return summary_95
|
||||||
|
end
|
||||||
|
|
||||||
|
function print_claassen_comparison(summary_95, summary_80)
|
||||||
|
println("\n" * "="^60)
|
||||||
|
println("COMPARISON WITH CLAASSEN (2019) BENCHMARKS")
|
||||||
|
println("="^60)
|
||||||
|
|
||||||
|
println("\nClaassen's result:")
|
||||||
|
println(" CIC (80% CI): 60.3%")
|
||||||
|
println(" (Using credible intervals for θ, not posterior predictive)")
|
||||||
|
println()
|
||||||
|
println("Our results (posterior predictive):")
|
||||||
|
println()
|
||||||
|
|
||||||
|
println("-"^60)
|
||||||
|
@printf("%-15s %10s %10s %8s\n", "Dimension", "PPC 95%", "PPC 80%", "Status")
|
||||||
|
println("-"^60)
|
||||||
|
|
||||||
|
dim_map_80 = Dict(r.dimension => r for r in summary_80)
|
||||||
|
|
||||||
|
for r in summary_95
|
||||||
|
ppc80 = haskey(dim_map_80, r.dimension) ? dim_map_80[r.dimension].cic : NaN
|
||||||
|
# Well-calibrated: 95% PPC should be near 95%
|
||||||
|
status = r.cic >= 0.90 ? "GOOD" : (r.cic >= 0.80 ? "OK" : "LOW")
|
||||||
|
@printf("%-15s %9.1f%% %9.1f%% %8s\n",
|
||||||
|
r.dimension, 100*r.cic, 100*ppc80, status)
|
||||||
|
end
|
||||||
|
println("-"^60)
|
||||||
|
println()
|
||||||
|
println("Interpretation:")
|
||||||
|
println(" 95% PPC ~95% = well-calibrated uncertainty")
|
||||||
|
println(" 80% PPC > 60% = exceeds Claassen (2019) benchmark")
|
||||||
|
end
|
||||||
|
|
||||||
|
# =============================================================================
|
||||||
|
# MAIN
|
||||||
|
# =============================================================================
|
||||||
|
|
||||||
|
function main()
|
||||||
|
println("="^60)
|
||||||
|
println("UNCERTAINTY VALIDATION: Posterior Predictive Coverage")
|
||||||
|
println("="^60)
|
||||||
|
println("Following Claassen (2019) validation framework")
|
||||||
|
println("Posterior predictive intervals account for both position")
|
||||||
|
println("uncertainty AND observation-level measurement noise.")
|
||||||
|
println()
|
||||||
|
|
||||||
|
# Step 0: Parse options and find run directory
|
||||||
|
run_dir = nothing
|
||||||
|
output_dir = "validation/revision"
|
||||||
|
quick_mode = get(ENV, "QUICK_VALIDATION", "0") == "1"
|
||||||
|
for (i, arg) in enumerate(ARGS)
|
||||||
|
if arg == "--run-dir" && i < length(ARGS)
|
||||||
|
run_dir = ARGS[i + 1]
|
||||||
|
elseif startswith(arg, "--run-dir=")
|
||||||
|
run_dir = split(arg, "=", limit=2)[2]
|
||||||
|
elseif arg == "--output-dir" && i < length(ARGS)
|
||||||
|
output_dir = ARGS[i + 1]
|
||||||
|
elseif startswith(arg, "--output-dir=")
|
||||||
|
output_dir = split(arg, "=", limit=2)[2]
|
||||||
|
elseif arg == "--quick"
|
||||||
|
quick_mode = true
|
||||||
|
end
|
||||||
|
end
|
||||||
|
if run_dir === nothing
|
||||||
|
run_dir = find_latest_run()
|
||||||
|
else
|
||||||
|
println("Using specified run directory: $run_dir")
|
||||||
|
end
|
||||||
|
|
||||||
|
# Step 1: Load expert_dim.csv
|
||||||
|
expert_dim = load_expert_dim(run_dir)
|
||||||
|
|
||||||
|
# Step 2: Selectively load chains
|
||||||
|
needed_rr = Set(expert_dim.rr_exp_dim)
|
||||||
|
K = maximum(expert_dim.var_exp_dim)
|
||||||
|
chains = load_chains_selective(run_dir, needed_rr, K)
|
||||||
|
|
||||||
|
# Step 3a: Compute 95% posterior predictive coverage
|
||||||
|
result_95 = compute_posterior_predictive_cic(chains, expert_dim; ci_level=0.95)
|
||||||
|
summary_95 = summarize_coverage(result_95, 0.95)
|
||||||
|
save_detailed_results(result_95, output_dir)
|
||||||
|
|
||||||
|
# Step 3b: Compute 80% posterior predictive coverage (Claassen benchmark)
|
||||||
|
# Recompute coverage from the same predictive draws but with 80% quantiles
|
||||||
|
if quick_mode
|
||||||
|
println("\nQUICK MODE: skipping 80% PPC recomputation")
|
||||||
|
summary_80 = [(dimension=r.dimension, n=r.n, covered=r.covered, cic=NaN, ci_level=0.80) for r in summary_95]
|
||||||
|
else
|
||||||
|
println("\n" * "="^60)
|
||||||
|
println("RECOMPUTING WITH 80% LEVEL (Claassen comparison)")
|
||||||
|
println("="^60)
|
||||||
|
result_80 = compute_posterior_predictive_cic(chains, expert_dim; ci_level=0.80, seed=42)
|
||||||
|
summary_80 = summarize_coverage(result_80, 0.80)
|
||||||
|
end
|
||||||
|
|
||||||
|
# Build by-project tables for 95%
|
||||||
|
dim_names = Dict(1 => "economic_lr", 2 => "galtan")
|
||||||
|
by_project_95 = Dict{Int, DataFrame}()
|
||||||
|
for dim in sort(unique(result_95.dim_idx))
|
||||||
|
subset = filter(r -> r.dim_idx == dim, result_95)
|
||||||
|
bp = combine(groupby(subset, :project)) do df
|
||||||
|
nc = sum(df.covered)
|
||||||
|
DataFrame(n = nrow(df), covered = nc, cic = nc / nrow(df))
|
||||||
|
end
|
||||||
|
sort!(bp, :n, rev=true)
|
||||||
|
by_project_95[dim] = bp
|
||||||
|
end
|
||||||
|
|
||||||
|
# Step 4: Save results
|
||||||
|
save_results(result_95, summary_95, summary_80, by_project_95, output_dir)
|
||||||
|
|
||||||
|
# Step 5: Print Claassen comparison
|
||||||
|
print_claassen_comparison(summary_95, summary_80)
|
||||||
|
|
||||||
|
println("\n" * "="^60)
|
||||||
|
println("VALIDATION COMPLETE")
|
||||||
|
println("="^60)
|
||||||
|
|
||||||
|
return (summary_95=summary_95, summary_80=summary_80)
|
||||||
|
end
|
||||||
|
|
||||||
|
if abspath(PROGRAM_FILE) == @__FILE__
|
||||||
|
main()
|
||||||
|
end
|
||||||
Executable
+54
@@ -0,0 +1,54 @@
|
|||||||
|
#!/usr/bin/env bash
|
||||||
|
set -euo pipefail
|
||||||
|
|
||||||
|
repo_root="$(cd "$(dirname "${BASH_SOURCE[0]}")/../.." && pwd -P)"
|
||||||
|
project_root="$(cd "$repo_root/.." && pwd -P)"
|
||||||
|
raw_data_dir="${PARTY2D_RAW_DATA_DIR:-$project_root/_local/raw}"
|
||||||
|
|
||||||
|
required_files=(
|
||||||
|
"poldem/poldem-election_all.csv"
|
||||||
|
)
|
||||||
|
|
||||||
|
optional_files=(
|
||||||
|
"manifesto/MPDataset_MPDS2025a.csv"
|
||||||
|
)
|
||||||
|
|
||||||
|
echo "Raw data directory: $raw_data_dir"
|
||||||
|
echo
|
||||||
|
echo "Required raw inputs for regeneration:"
|
||||||
|
|
||||||
|
missing=0
|
||||||
|
for rel in "${required_files[@]}"; do
|
||||||
|
path="$raw_data_dir/$rel"
|
||||||
|
if [ -s "$path" ]; then
|
||||||
|
bytes="$(wc -c < "$path")"
|
||||||
|
read -r sha _ < <(sha256sum "$path")
|
||||||
|
echo " OK $rel ($bytes bytes, sha256=$sha)"
|
||||||
|
else
|
||||||
|
echo " MISSING $rel"
|
||||||
|
missing=1
|
||||||
|
fi
|
||||||
|
done
|
||||||
|
|
||||||
|
echo
|
||||||
|
echo "Optional raw inputs used only if cached processed files are regenerated:"
|
||||||
|
for rel in "${optional_files[@]}"; do
|
||||||
|
path="$raw_data_dir/$rel"
|
||||||
|
if [ -s "$path" ]; then
|
||||||
|
bytes="$(wc -c < "$path")"
|
||||||
|
read -r sha _ < <(sha256sum "$path")
|
||||||
|
echo " OK $rel ($bytes bytes, sha256=$sha)"
|
||||||
|
else
|
||||||
|
echo " MISSING $rel"
|
||||||
|
fi
|
||||||
|
done
|
||||||
|
|
||||||
|
if [ "$missing" -ne 0 ]; then
|
||||||
|
echo
|
||||||
|
echo "At least one required raw input is missing." >&2
|
||||||
|
echo "See docs/RAW_DATA_SOURCES.md for download/local-placement instructions." >&2
|
||||||
|
exit 1
|
||||||
|
fi
|
||||||
|
|
||||||
|
echo
|
||||||
|
echo "Required raw data preflight passed."
|
||||||
Executable
+48
@@ -0,0 +1,48 @@
|
|||||||
|
#!/usr/bin/env bash
|
||||||
|
set -euo pipefail
|
||||||
|
|
||||||
|
repo_root="$(cd "$(dirname "${BASH_SOURCE[0]}")/../.." && pwd -P)"
|
||||||
|
logs_dir="$repo_root/outputs/logs"
|
||||||
|
model_dir="$repo_root/outputs/model_outputs/latest"
|
||||||
|
|
||||||
|
latest_log="$(ls "$logs_dir"/full_run_*.log 2>/dev/null | sort | tail -n 1 || true)"
|
||||||
|
latest_run="$(ls -d "$model_dir"/run_* 2>/dev/null | sort | tail -n 1 || true)"
|
||||||
|
|
||||||
|
echo "party2d full-run progress"
|
||||||
|
echo "repo: $repo_root"
|
||||||
|
echo
|
||||||
|
|
||||||
|
if [[ -n "$latest_log" ]]; then
|
||||||
|
echo "Latest durable log: $latest_log"
|
||||||
|
echo "--- last 40 log lines ---"
|
||||||
|
tail -n 40 "$latest_log"
|
||||||
|
else
|
||||||
|
echo "No durable full_run_*.log found under $logs_dir"
|
||||||
|
fi
|
||||||
|
|
||||||
|
echo
|
||||||
|
if [[ -n "$latest_run" ]]; then
|
||||||
|
echo "Latest model run: $latest_run"
|
||||||
|
metrics="$latest_run/diagnostics/run_metrics.json"
|
||||||
|
if [[ -f "$metrics" ]]; then
|
||||||
|
echo "Metrics: $metrics"
|
||||||
|
echo "Open the JSON metrics file above for run-summary fields."
|
||||||
|
else
|
||||||
|
echo "No archived metrics yet. If the run is still sampling, watch the durable log with:"
|
||||||
|
if [[ -n "$latest_log" ]]; then
|
||||||
|
echo " tail -f \"$latest_log\""
|
||||||
|
else
|
||||||
|
echo " tail -f outputs/logs/full_run_<timestamp>.log"
|
||||||
|
fi
|
||||||
|
fi
|
||||||
|
else
|
||||||
|
echo "No completed model run found under $model_dir"
|
||||||
|
fi
|
||||||
|
|
||||||
|
echo
|
||||||
|
echo "To follow a running job live:"
|
||||||
|
if [[ -n "$latest_log" ]]; then
|
||||||
|
echo " tail -f \"$latest_log\""
|
||||||
|
else
|
||||||
|
echo " tail -f outputs/logs/full_run_<timestamp>.log"
|
||||||
|
fi
|
||||||
@@ -0,0 +1,37 @@
|
|||||||
|
# Predeclared case-selection rules
|
||||||
|
|
||||||
|
These rules were recorded before inspecting the selected parties' position estimates.
|
||||||
|
|
||||||
|
## Trajectory figure
|
||||||
|
|
||||||
|
The figure will use parties selected for cross-national recognition, family diversity, long election coverage, and relevance to interpreting movement on the two scales. Selection is not based on the magnitude of the estimated movement.
|
||||||
|
|
||||||
|
Preselected parties:
|
||||||
|
|
||||||
|
- Germany: Social Democratic Party of Germany (PartyFacts 383)
|
||||||
|
- Germany: Christian Democratic Union (PartyFacts 1375)
|
||||||
|
- United Kingdom: Labour Party (PartyFacts 1516)
|
||||||
|
- United Kingdom: Conservative Party (PartyFacts 1567)
|
||||||
|
- Denmark: Social Democratic Party, short name SD (PartyFacts 379)
|
||||||
|
- Sweden: Sweden Democrats (PartyFacts 409)
|
||||||
|
- United States: Democratic Party (PartyFacts 432)
|
||||||
|
- United States: Republican Party (PartyFacts 809)
|
||||||
|
|
||||||
|
The final display may use a subset only to preserve legibility, but exclusions must be based on panel layout or insufficient coverage rather than unattractive results. Both posterior means and 95% latent-position credible intervals will be shown. Text will not infer a cause of movement from the estimates alone.
|
||||||
|
|
||||||
|
## Two-dimensional landmarks
|
||||||
|
|
||||||
|
Landmarks will be selected from recognizable party-election cases, with representation of economically left/right, culturally cosmopolitan/traditionalist, and moderate combinations. Target years are declared before checking exact positions; if the requested year is absent, the nearest election within three years will be used and disclosed.
|
||||||
|
|
||||||
|
- German Social Democratic Party: 1972 and 2021
|
||||||
|
- German Christian Democratic Union: 1983 and 2021
|
||||||
|
- UK Labour Party: 1983 and 1997
|
||||||
|
- UK Conservative Party: 1979 and 2019
|
||||||
|
- Swedish Social Democratic Labour Party (PartyFacts 487): 1994 and 2022
|
||||||
|
- Sweden Democrats: 2010 and 2022
|
||||||
|
- French National Front (PartyFacts 433): 1988 and 2022
|
||||||
|
- German The Left (PartyFacts 1545): 2021
|
||||||
|
- US Democratic Party: 2020
|
||||||
|
- US Republican Party: 2020
|
||||||
|
|
||||||
|
Labels may be pruned only to avoid overlap. The plotting-data table will retain all declared cases, including missing cases and substitutions.
|
||||||
@@ -0,0 +1,13 @@
|
|||||||
|
id,topic,input,output,status
|
||||||
|
1,country coverage,release v0 election-year panel,metadata/country_coverage_v0.csv,complete
|
||||||
|
2,scale trajectories,release v0 election-year panel,validation/figures/party_trajectories.pdf,complete
|
||||||
|
3,landmark party space,release v0 election-year panel,validation/figures/party_landmarks.pdf,complete
|
||||||
|
4,research workflow,documented source and release workflow,validation/figures/research_workflow.pdf,complete
|
||||||
|
5,party-blocked validation,guarded blocked fit with 82 parties and completed post-estimation extraction,validation/outputs/blocked_validation_summary.csv,complete_with_convergence_limitation
|
||||||
|
6,predictive coverage,production posterior run run_2026-06-12_09-34-03,validation/outputs/ppc/posterior_predictive_by_dimension_item.csv,complete
|
||||||
|
7,predictive calibration curve,production posterior run run_2026-06-12_09-34-03,validation/outputs/ppc/posterior_predictive_calibration_curve.csv,complete
|
||||||
|
8,predictive residual patterns,production posterior predictive observations,validation/outputs/ppc/posterior_predictive_residual_patterns.csv,complete
|
||||||
|
9,V-Party sensitivity,completed no-V-Party fit ending 2026-06-05_14-57-17,validation/outputs/vparty_sensitivity_groups.csv,complete
|
||||||
|
10,V-Party subgroup sensitivity,completed no-V-Party fit ending 2026-06-05_14-57-17,validation/outputs/vparty_sensitivity_groups.csv,complete
|
||||||
|
11,pooled source-support balance,release v0 election-year panel,validation/outputs/source_support_pooled.csv,complete
|
||||||
|
12,partially pooled source-support heterogeneity,release v0 election-year panel,validation/outputs/source_support_interactions.csv,complete
|
||||||
|
@@ -0,0 +1,30 @@
|
|||||||
|
# Validation materials
|
||||||
|
|
||||||
|
This directory contains reproducible validation code, party-blocked sensitivity analyses, posterior predictive checks, subgroup balance diagnostics, and figure-generation scripts for the party-position panel.
|
||||||
|
|
||||||
|
All analyses preserve the production release (`v0`, production run `run_2026-06-12_09-34-03`). Existing posterior draws and completed sensitivity fits are reused wherever possible. The party-blocked expert validation tests prediction for parties whose expert evidence is entirely excluded from estimation.
|
||||||
|
|
||||||
|
## Inputs
|
||||||
|
|
||||||
|
- `data/releases/party_2d_election_year_panel_v0.csv.gz`
|
||||||
|
- `data/releases/party_2d_annual_model_output_v0.csv.gz`
|
||||||
|
- production posterior run `run_2026-06-12_09-34-03`
|
||||||
|
- completed no-V-Party sensitivity output `party_positions_2026-06-05_14-57-17.csv`
|
||||||
|
- model-ready input files in `data/`
|
||||||
|
|
||||||
|
Large model runs and their chains remain outside Git. This package contains the scripts, manifests, diagnostics, grouped summaries, plotting data, and figures needed to inspect and reproduce the published validation results. Two large row-level intermediate tables used during post-estimation are intentionally excluded; the retained grouped summaries and scripts document all reported calculations.
|
||||||
|
|
||||||
|
## Computational notes
|
||||||
|
|
||||||
|
The validation fit retains the production warmup length and four-chain design but uses 1,000 retained iterations per chain (4,000 draws total), rather than 2,000 for the production release. The scripts record immutable train/test identifiers and input hashes before fitting and use existing posterior draws for descriptive and posterior-predictive analyses where applicable.
|
||||||
|
|
||||||
|
## Reproduction Entry Points
|
||||||
|
|
||||||
|
- `run_postprocessing.R`: country coverage, scale illustrations, V-Party sensitivity and partially pooled source-support diagnostics.
|
||||||
|
- `validate_uncertainty.jl`: production-chain predictive coverage, calibration and source/item/country/decade breakdowns.
|
||||||
|
- `prepare_blocked_validation.jl`: deterministic party-blocked train/test construction.
|
||||||
|
- `run_blocked_validation.sh`: guarded low-priority fit with disk, duplicate-process, toolchain and logging checks.
|
||||||
|
- `monitor_long_run.sh`: non-invasive status, child-process, disk and recent-log monitoring.
|
||||||
|
- `finalize_blocked_validation.sh`: guarded post-estimation, held-out summaries and compact diagnostic-manifest collection after the fit succeeds.
|
||||||
|
- `summarize_blocked_validation.jl`: held-out prediction records and grouped performance summaries after the fit.
|
||||||
|
- `plot_workflow.R`: data-processing workflow schematic.
|
||||||
Binary file not shown.
Binary file not shown.
Binary file not shown.
Binary file not shown.
Binary file not shown.
Binary file not shown.
Binary file not shown.
Executable
+68
@@ -0,0 +1,68 @@
|
|||||||
|
#!/usr/bin/env bash
|
||||||
|
set -euo pipefail
|
||||||
|
|
||||||
|
repo_root="$(cd "$(dirname "${BASH_SOURCE[0]}")/.." && pwd -P)"
|
||||||
|
cd "$repo_root"
|
||||||
|
|
||||||
|
if [[ "${PARTY2D_APPROVE_POSTESTIMATION:-}" != "YES" ]]; then
|
||||||
|
echo "Refusing to read chains or run post-estimation without explicit approval." >&2
|
||||||
|
echo "After approval, rerun with PARTY2D_APPROVE_POSTESTIMATION=YES." >&2
|
||||||
|
exit 64
|
||||||
|
fi
|
||||||
|
|
||||||
|
run_dir="${1:-$repo_root/_local/validation/blocked_party}"
|
||||||
|
summary_dir="${2:-$repo_root/validation/outputs}"
|
||||||
|
estimation_dir="$run_dir/estimations"
|
||||||
|
|
||||||
|
if [[ ! -f "$run_dir/model_fit.complete" ]]; then
|
||||||
|
echo "Refusing to finalize: the guarded fit is not marked complete." >&2
|
||||||
|
exit 1
|
||||||
|
fi
|
||||||
|
if [[ ! -f "$run_dir/model_fit.exit_code" ]] || [[ "$(tr -dc '0-9' < "$run_dir/model_fit.exit_code")" != "0" ]]; then
|
||||||
|
echo "Refusing to finalize: the guarded fit has no successful exit code." >&2
|
||||||
|
exit 1
|
||||||
|
fi
|
||||||
|
|
||||||
|
mapfile -t model_runs < <(find "$run_dir/model_run/latest" -mindepth 1 -maxdepth 1 -type d -name 'run_*' 2>/dev/null | sort)
|
||||||
|
if [[ "${#model_runs[@]}" -eq 0 ]]; then
|
||||||
|
# The production saver currently writes its run under the repository-level
|
||||||
|
# output directory even when --data-dir points at a validation workspace.
|
||||||
|
# Match the completed fit by timestamp instead of relying on "latest" alone.
|
||||||
|
fit_started="$(tr -d '\r\n' < "$run_dir/model_fit.started")"
|
||||||
|
fit_completed="$(tr -d '\r\n' < "$run_dir/model_fit.complete")"
|
||||||
|
mapfile -t model_runs < <(find "$repo_root/outputs/model_outputs/latest" -mindepth 1 -maxdepth 1 -type d -name 'run_*' -newermt "$fit_started" ! -newermt "$fit_completed" | sort)
|
||||||
|
fi
|
||||||
|
if [[ "${#model_runs[@]}" -ne 1 ]]; then
|
||||||
|
echo "Expected exactly one blocked model run; found ${#model_runs[@]}." >&2
|
||||||
|
printf '%s\n' "${model_runs[@]}" >&2
|
||||||
|
exit 1
|
||||||
|
fi
|
||||||
|
model_run="${model_runs[0]}"
|
||||||
|
|
||||||
|
mkdir -p "$estimation_dir" "$summary_dir/blocked_fit_diagnostics"
|
||||||
|
nice -n 10 julia --project=. src/julia/02_post_estimation.jl \
|
||||||
|
--run-dir "$model_run" --output-dir "$estimation_dir" \
|
||||||
|
2>&1 | tee "$run_dir/post_estimation.log"
|
||||||
|
|
||||||
|
mapfile -t position_files < <(find "$estimation_dir" -maxdepth 1 -type f -name 'party_positions_*.csv' | sort)
|
||||||
|
if [[ "${#position_files[@]}" -ne 1 ]]; then
|
||||||
|
echo "Expected exactly one blocked position file; found ${#position_files[@]}." >&2
|
||||||
|
exit 1
|
||||||
|
fi
|
||||||
|
|
||||||
|
julia --project=. validation/summarize_blocked_validation.jl \
|
||||||
|
"${position_files[0]}" "$run_dir" "$summary_dir" \
|
||||||
|
2>&1 | tee "$run_dir/blocked_summary.log"
|
||||||
|
|
||||||
|
cp "$model_run/metadata.json" "$summary_dir/blocked_fit_diagnostics/metadata.json"
|
||||||
|
cp "$model_run/diagnostics/run_metrics.json" "$summary_dir/blocked_fit_diagnostics/run_metrics.json"
|
||||||
|
cp "$run_dir/blocked_validation_manifest.csv" "$summary_dir/blocked_fit_diagnostics/blocked_validation_manifest.csv"
|
||||||
|
cp "$run_dir/blocked_validation_strata.csv" "$summary_dir/blocked_fit_diagnostics/blocked_validation_strata.csv"
|
||||||
|
cp "$run_dir/blocked_parties.csv" "$summary_dir/blocked_fit_diagnostics/blocked_parties.csv"
|
||||||
|
cp "$run_dir/input_sha256sums.txt" "$summary_dir/blocked_fit_diagnostics/input_sha256sums.txt"
|
||||||
|
cp "$run_dir/model_fit.environment" "$summary_dir/blocked_fit_diagnostics/model_fit.environment"
|
||||||
|
cp "$run_dir/stanc_version.txt" "$summary_dir/blocked_fit_diagnostics/stanc_version.txt"
|
||||||
|
|
||||||
|
echo "Blocked validation finalized from: $model_run"
|
||||||
|
echo "Position file: ${position_files[0]}"
|
||||||
|
echo "Summary directory: $summary_dir"
|
||||||
@@ -0,0 +1,146 @@
|
|||||||
|
============================================================
|
||||||
|
UNCERTAINTY VALIDATION: Posterior Predictive Coverage
|
||||||
|
============================================================
|
||||||
|
Following Claassen (2019) validation framework
|
||||||
|
Posterior predictive intervals account for both position
|
||||||
|
uncertainty AND observation-level measurement noise.
|
||||||
|
|
||||||
|
Using specified run directory: /projects/party4d/archive/party2d_replication/outputs/model_outputs/latest/run_2026-06-12_09-34-03
|
||||||
|
Loaded expert_dim.csv: 22994 observations
|
||||||
|
Unique rr values: 4261
|
||||||
|
Item indices (var_exp_dim): [1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12]
|
||||||
|
Dimensions (dim_idx_exp): [1, 2]
|
||||||
|
|
||||||
|
Loading 4 chain files (selective columns)...
|
||||||
|
Need 8547 columns (4261 rr × 2 dims + 24 item params + 1 phi)
|
||||||
|
Found 8547/8547 columns in chains
|
||||||
|
Loading chain 1: chain_1.csv... 2000 samples, 39.3s
|
||||||
|
Loading chain 2: chain_2.csv... 2000 samples, 22.4s
|
||||||
|
Loading chain 3: chain_3.csv... 2000 samples, 24.5s
|
||||||
|
Loading chain 4: chain_4.csv... 2000 samples, 16.7s
|
||||||
|
Combined: 8000 total posterior draws
|
||||||
|
|
||||||
|
V5 detected: using Beta(phi*K*mu, phi*K*(1-mu)) with per-observation K
|
||||||
|
Computing posterior predictive coverage (95% level)
|
||||||
|
Expert observations: 22994
|
||||||
|
Posterior draws: 8000
|
||||||
|
|
||||||
|
Progress: 5.0% (1149 / 22994)
|
||||||
|
Progress: 10.0% (2298 / 22994)
|
||||||
|
Progress: 15.0% (3447 / 22994)
|
||||||
|
Progress: 20.0% (4596 / 22994)
|
||||||
|
Progress: 25.0% (5745 / 22994)
|
||||||
|
Progress: 30.0% (6894 / 22994)
|
||||||
|
Progress: 35.0% (8043 / 22994)
|
||||||
|
Progress: 40.0% (9192 / 22994)
|
||||||
|
Progress: 45.0% (10341 / 22994)
|
||||||
|
Progress: 50.0% (11490 / 22994)
|
||||||
|
Progress: 55.0% (12639 / 22994)
|
||||||
|
Progress: 60.0% (13788 / 22994)
|
||||||
|
Progress: 65.0% (14937 / 22994)
|
||||||
|
Progress: 70.0% (16086 / 22994)
|
||||||
|
Progress: 75.0% (17235 / 22994)
|
||||||
|
Progress: 80.0% (18384 / 22994)
|
||||||
|
Progress: 84.9% (19533 / 22994)
|
||||||
|
Progress: 89.9% (20682 / 22994)
|
||||||
|
Progress: 94.9% (21831 / 22994)
|
||||||
|
Progress: 99.9% (22980 / 22994)
|
||||||
|
Progress: 100.0% (22994 / 22994)
|
||||||
|
|
||||||
|
============================================================
|
||||||
|
POSTERIOR PREDICTIVE COVERAGE (95%)
|
||||||
|
============================================================
|
||||||
|
|
||||||
|
economic left-right : 89.8% [89.1%, 90.5%] (6698/7455)
|
||||||
|
|
||||||
|
By survey source:
|
||||||
|
Project N PPC
|
||||||
|
V-Party 5645 87.5%
|
||||||
|
CHES 1177 97.3%
|
||||||
|
POPPA 384 98.7%
|
||||||
|
GPS 249 94.0%
|
||||||
|
|
||||||
|
By decade:
|
||||||
|
Decade N PPC
|
||||||
|
1970 658 88.1%
|
||||||
|
1980 712 88.9%
|
||||||
|
1990 1454 88.0%
|
||||||
|
2000 1758 89.4%
|
||||||
|
2010 2407 90.2%
|
||||||
|
2020 466 99.1%
|
||||||
|
|
||||||
|
cultural cosmopolitan--traditionalist : 84.8% [84.2%, 85.3%] (13170/15539)
|
||||||
|
|
||||||
|
By survey source:
|
||||||
|
Project N PPC
|
||||||
|
V-Party 14114 83.8%
|
||||||
|
CHES 1176 94.3%
|
||||||
|
GPS 249 91.2%
|
||||||
|
|
||||||
|
By decade:
|
||||||
|
Decade N PPC
|
||||||
|
1970 1633 86.2%
|
||||||
|
1980 1772 87.9%
|
||||||
|
1990 3485 85.5%
|
||||||
|
2000 3965 84.3%
|
||||||
|
2010 4426 82.1%
|
||||||
|
2020 258 97.7%
|
||||||
|
Saved detailed posterior-predictive results to: revision/outputs/ppc
|
||||||
|
|
||||||
|
============================================================
|
||||||
|
RECOMPUTING WITH 80% LEVEL (Claassen comparison)
|
||||||
|
============================================================
|
||||||
|
|
||||||
|
V5 detected: using Beta(phi*K*mu, phi*K*(1-mu)) with per-observation K
|
||||||
|
Computing posterior predictive coverage (80% level)
|
||||||
|
Expert observations: 22994
|
||||||
|
Posterior draws: 8000
|
||||||
|
|
||||||
|
Progress: 5.0% (1149 / 22994)
|
||||||
|
Progress: 10.0% (2298 / 22994)
|
||||||
|
Progress: 15.0% (3447 / 22994)
|
||||||
|
Progress: 20.0% (4596 / 22994)
|
||||||
|
Progress: 25.0% (5745 / 22994)
|
||||||
|
Progress: 30.0% (6894 / 22994)
|
||||||
|
Progress: 35.0% (8043 / 22994)
|
||||||
|
Progress: 40.0% (9192 / 22994)
|
||||||
|
Progress: 45.0% (10341 / 22994)
|
||||||
|
Progress: 50.0% (11490 / 22994)
|
||||||
|
Progress: 55.0% (12639 / 22994)
|
||||||
|
Progress: 60.0% (13788 / 22994)
|
||||||
|
Progress: 65.0% (14937 / 22994)
|
||||||
|
Progress: 70.0% (16086 / 22994)
|
||||||
|
Progress: 75.0% (17235 / 22994)
|
||||||
|
Progress: 80.0% (18384 / 22994)
|
||||||
|
Progress: 84.9% (19533 / 22994)
|
||||||
|
Progress: 89.9% (20682 / 22994)
|
||||||
|
Progress: 94.9% (21831 / 22994)
|
||||||
|
Progress: 99.9% (22980 / 22994)
|
||||||
|
Progress: 100.0% (22994 / 22994)
|
||||||
|
|
||||||
|
============================================================
|
||||||
|
POSTERIOR PREDICTIVE COVERAGE (80%)
|
||||||
|
============================================================
|
||||||
|
|
||||||
|
economic left-right : 75.9% [74.9%, 76.8%] (5655/7455)
|
||||||
|
|
||||||
|
By survey source:
|
||||||
|
Project N PPC
|
||||||
|
V-Party 5645 71.5%
|
||||||
|
CHES 1177 90.1%
|
||||||
|
POPPA 384 92.2%
|
||||||
|
GPS 249 82.7%
|
||||||
|
|
||||||
|
By decade:
|
||||||
|
Decade N PPC
|
||||||
|
1970 658 69.3%
|
||||||
|
1980 712 73.5%
|
||||||
|
1990 1454 73.8%
|
||||||
|
2000 1758 74.1%
|
||||||
|
2010 2407 77.0%
|
||||||
|
2020 466 95.7%
|
||||||
|
|
||||||
|
cultural cosmopolitan--traditionalist : 67.9% [67.1%, 68.6%] (10544/15539)
|
||||||
|
|
||||||
|
By survey source:
|
||||||
|
Project N PPC
|
||||||
Executable
+50
@@ -0,0 +1,50 @@
|
|||||||
|
#!/usr/bin/env bash
|
||||||
|
set -euo pipefail
|
||||||
|
|
||||||
|
run_dir="${1:-_local/validation/blocked_party}"
|
||||||
|
pid_file="$run_dir/model_fit.pid"
|
||||||
|
log_file="$run_dir/model_fit.log"
|
||||||
|
|
||||||
|
if [[ ! -f "$pid_file" ]]; then
|
||||||
|
echo "No PID file at $pid_file"
|
||||||
|
exit 1
|
||||||
|
fi
|
||||||
|
|
||||||
|
pid="$(tr -dc '0-9' < "$pid_file")"
|
||||||
|
echo "Run directory: $run_dir"
|
||||||
|
echo "PID: $pid"
|
||||||
|
date --iso-8601=seconds
|
||||||
|
|
||||||
|
if ps -p "$pid" >/dev/null 2>&1; then
|
||||||
|
echo "Status: running"
|
||||||
|
ps -o pid,ppid,etimes,%cpu,%mem,rss,vsz,stat,cmd -p "$pid"
|
||||||
|
echo "Model processes:"
|
||||||
|
julia_pid="$(pgrep -P "$pid" -f 'julia.*01_run_model[.]jl' | head -1 || true)"
|
||||||
|
if [[ -n "$julia_pid" ]]; then
|
||||||
|
process_ids="$pid,$julia_pid"
|
||||||
|
while IFS= read -r child_pid; do
|
||||||
|
[[ -n "$child_pid" ]] && process_ids="$process_ids,$child_pid"
|
||||||
|
done < <(pgrep -P "$julia_pid" || true)
|
||||||
|
ps -o pid,ppid,etimes,%cpu,%mem,rss,ni,stat,cmd -p "$process_ids"
|
||||||
|
else
|
||||||
|
ps -eo pid,ppid,etimes,%cpu,%mem,rss,ni,stat,cmd | awk -v p="$pid" '$2 == p || $1 == p'
|
||||||
|
fi
|
||||||
|
else
|
||||||
|
echo "Status: not running"
|
||||||
|
fi
|
||||||
|
|
||||||
|
echo "Latest sampler progress:"
|
||||||
|
grep 'Iteration:' "$log_file" 2>/dev/null | tail -12 || true
|
||||||
|
|
||||||
|
echo "Recorded sampler events:"
|
||||||
|
printf ' rejected-proposal exceptions: '
|
||||||
|
grep -c '^Exception:' "$log_file" 2>/dev/null || true
|
||||||
|
printf ' divergence messages: '
|
||||||
|
grep -ci 'divergent transition' "$log_file" 2>/dev/null || true
|
||||||
|
|
||||||
|
echo "Disk use:"
|
||||||
|
du -sh "$run_dir" _local/tmp/blocked_validation 2>/dev/null || true
|
||||||
|
df -h "$run_dir" | tail -1
|
||||||
|
|
||||||
|
echo "Recent log:"
|
||||||
|
tail -40 "$log_file" 2>/dev/null || true
|
||||||
@@ -0,0 +1,83 @@
|
|||||||
|
party_id,country,region,first_expert_year,last_expert_year,expert_period,text_years,expert_rows,lr_rows,stratum,selected
|
||||||
|
42,HU,Europe,2010,2024,2010s,3,33,6,Europe / 2010s,true
|
||||||
|
96,SI,Europe,1992,2019,1990s,7,31,4,Europe / 1990s,true
|
||||||
|
172,NL,Europe,1999,1999,1990s,5,2,1,Europe / 1990s,true
|
||||||
|
216,MX,Latin America,1991,2020,1990s,8,74,1,Latin America / 1990s,true
|
||||||
|
232,CA,North America,1972,2000,pre-1990,18,63,0,North America / pre-1990,true
|
||||||
|
237,LT,Europe,2004,2019,2000s,4,36,4,Europe / 2000s,true
|
||||||
|
281,BE,Europe,1971,1974,pre-1990,6,14,2,Europe / pre-1990,true
|
||||||
|
284,PT,Europe,1987,2024,pre-1990,9,88,9,Europe / pre-1990,true
|
||||||
|
292,BA,Europe,2002,2019,2000s,6,37,0,Europe / 2000s,true
|
||||||
|
298,NL,Europe,2006,2024,2000s,5,42,7,Europe / 2000s,true
|
||||||
|
306,TR,Europe,2002,2024,2000s,5,32,1,Europe / 2000s,true
|
||||||
|
363,IS,Europe,1971,2024,pre-1990,23,103,9,Europe / pre-1990,true
|
||||||
|
441,ES,Europe,1977,2024,pre-1990,15,116,9,Europe / pre-1990,true
|
||||||
|
447,IL,Other,2021,2022,2020s,4,4,2,Other / 2020s,true
|
||||||
|
466,CZ,Europe,1992,2024,1990s,8,72,8,Europe / 1990s,true
|
||||||
|
472,SI,Europe,1990,2024,1990s,9,72,8,Europe / 1990s,true
|
||||||
|
487,SE,Europe,1970,2024,pre-1990,24,123,18,Europe / pre-1990,true
|
||||||
|
500,BE,Europe,1978,2024,pre-1990,12,102,9,Europe / pre-1990,true
|
||||||
|
537,BA,Europe,1998,2019,1990s,8,51,0,Europe / 1990s,true
|
||||||
|
557,IL,Other,1999,2003,1990s,3,14,0,Other / 1990s,true
|
||||||
|
633,BE,Europe,1971,2024,pre-1990,16,114,9,Europe / pre-1990,true
|
||||||
|
705,NO,Europe,1973,2024,pre-1990,19,90,11,Europe / pre-1990,true
|
||||||
|
714,NL,Europe,2014,2023,2010s,3,8,4,Europe / 2010s,true
|
||||||
|
848,ES,Europe,1999,2024,1990s,15,56,8,Europe / 1990s,true
|
||||||
|
910,HU,Europe,1990,2010,1990s,5,41,3,Europe / 1990s,true
|
||||||
|
934,IT,Europe,1972,1992,pre-1990,14,42,7,Europe / pre-1990,true
|
||||||
|
946,ES,Europe,1999,1999,1990s,6,2,1,Europe / 1990s,true
|
||||||
|
950,EC,Latin America,1979,2019,pre-1990,4,100,0,Latin America / pre-1990,true
|
||||||
|
953,IL,Other,2003,2022,2000s,4,18,2,Other / 2000s,true
|
||||||
|
986,GB,Europe,1999,2024,1990s,7,25,9,Europe / 1990s,true
|
||||||
|
1004,CA,North America,2004,2023,2000s,7,46,1,North America / 2000s,true
|
||||||
|
1036,IL,Other,1973,2022,pre-1990,17,104,2,Other / pre-1990,true
|
||||||
|
1055,IE,Europe,1973,2024,pre-1990,20,102,9,Europe / pre-1990,true
|
||||||
|
1060,TR,Europe,1973,2024,pre-1990,15,67,1,Europe / pre-1990,true
|
||||||
|
1072,NO,Europe,1973,2024,pre-1990,19,88,11,Europe / pre-1990,true
|
||||||
|
1096,FI,Europe,1970,1987,pre-1990,13,42,9,Europe / pre-1990,true
|
||||||
|
1123,CH,Europe,2023,2024,2020s,13,3,2,Europe / 2020s,true
|
||||||
|
1126,IT,Europe,1972,2006,pre-1990,10,11,7,Europe / pre-1990,true
|
||||||
|
1138,LU,Europe,1984,2023,pre-1990,7,48,3,Europe / pre-1990,true
|
||||||
|
1157,NL,Europe,1977,2024,pre-1990,14,109,9,Europe / pre-1990,true
|
||||||
|
1166,BA,Europe,1996,2014,1990s,10,49,0,Europe / 1990s,true
|
||||||
|
1219,ZA,Other,1994,2019,1990s,6,44,0,Other / 1990s,true
|
||||||
|
1241,MX,Latin America,2019,2020,2010s,7,4,1,Latin America / 2010s,true
|
||||||
|
1242,SK,Europe,1994,1998,1990s,4,14,0,Europe / 1990s,true
|
||||||
|
1249,IS,Europe,1971,1995,pre-1990,15,50,8,Europe / pre-1990,true
|
||||||
|
1274,SE,Europe,1970,2024,pre-1990,24,114,18,Europe / pre-1990,true
|
||||||
|
1331,MX,Latin America,2006,2020,2000s,5,25,1,Latin America / 2000s,true
|
||||||
|
1359,PT,Europe,1975,2024,pre-1990,18,114,9,Europe / pre-1990,true
|
||||||
|
1386,SK,Europe,2010,2024,2010s,3,33,6,Europe / 2010s,true
|
||||||
|
1388,GB,Europe,1999,2024,1990s,9,32,9,Europe / 1990s,true
|
||||||
|
1415,CH,Europe,2011,2011,2010s,3,7,0,Europe / 2010s,true
|
||||||
|
1428,CA,North America,1993,2023,1990s,10,60,1,North America / 1990s,true
|
||||||
|
1431,HR,Europe,1992,2024,1990s,9,59,5,Europe / 1990s,true
|
||||||
|
1450,EC,Latin America,1984,2009,pre-1990,3,84,0,Latin America / pre-1990,true
|
||||||
|
1454,BA,Europe,1996,2019,1990s,9,51,0,Europe / 1990s,true
|
||||||
|
1459,NL,Europe,2002,2024,2000s,7,16,8,Europe / 2000s,true
|
||||||
|
1467,NL,Europe,2010,2024,2010s,5,12,6,Europe / 2010s,true
|
||||||
|
1508,MK,Europe,1998,2019,1990s,9,44,0,Europe / 1990s,true
|
||||||
|
1540,AU,Asia-Pacific,1972,1972,pre-1990,10,7,0,Asia-Pacific / pre-1990,true
|
||||||
|
1567,GB,Europe,1970,2024,pre-1990,22,109,9,Europe / pre-1990,true
|
||||||
|
1665,BG,Europe,1991,2024,1990s,8,75,6,Europe / 1990s,true
|
||||||
|
1673,BA,Europe,2000,2019,2000s,5,16,0,Europe / 2000s,true
|
||||||
|
1691,HU,Europe,1990,2024,1990s,9,70,8,Europe / 1990s,true
|
||||||
|
1715,RO,Europe,1990,2014,1990s,7,43,4,Europe / 1990s,true
|
||||||
|
1759,CH,Europe,2011,2024,2010s,4,18,3,Europe / 2010s,true
|
||||||
|
1804,JP,Asia-Pacific,1996,2014,1990s,7,49,0,Asia-Pacific / 1990s,true
|
||||||
|
1808,CH,Europe,1971,2024,pre-1990,19,95,1,Europe / pre-1990,true
|
||||||
|
1824,NZ,Asia-Pacific,1972,2019,pre-1990,26,114,0,Asia-Pacific / pre-1990,true
|
||||||
|
2159,GE,Europe,1999,1999,1990s,3,7,0,Europe / 1990s,true
|
||||||
|
2168,GE,Europe,1995,2003,1990s,3,21,0,Europe / 1990s,true
|
||||||
|
2203,RS,Europe,2000,2000,2000s,6,7,0,Europe / 2000s,true
|
||||||
|
2211,UA,Europe,1998,2006,1990s,5,21,0,Europe / 1990s,true
|
||||||
|
2228,UA,Europe,2002,2012,2000s,3,28,0,Europe / 2000s,true
|
||||||
|
2235,RU,Europe,1993,1993,1990s,3,7,0,Europe / 1990s,true
|
||||||
|
2256,RU,Europe,2003,2019,2000s,3,30,0,Europe / 2000s,true
|
||||||
|
2307,KR,Asia-Pacific,2000,2012,2000s,5,28,0,Asia-Pacific / 2000s,true
|
||||||
|
3162,ME,Europe,1998,2019,1990s,8,51,0,Europe / 1990s,true
|
||||||
|
3171,BA,Europe,2010,2019,2010s,3,23,0,Europe / 2010s,true
|
||||||
|
3185,ME,Europe,1998,2016,1990s,5,49,0,Europe / 1990s,true
|
||||||
|
3916,CO,Latin America,2020,2020,2020s,4,2,1,Latin America / 2020s,true
|
||||||
|
3955,IL,Other,2019,2019,2010s,4,7,0,Other / 2010s,true
|
||||||
|
5453,AU,Asia-Pacific,2019,2019,2010s,3,2,0,Asia-Pacific / 2010s,true
|
||||||
|
@@ -0,0 +1,15 @@
|
|||||||
|
field,value
|
||||||
|
created_at,2026-08-12T17:24:30.781
|
||||||
|
seed,20260812
|
||||||
|
holdout_share,0.2
|
||||||
|
eligible_parties,388
|
||||||
|
blocked_parties,82
|
||||||
|
text_rows_train,38202
|
||||||
|
expert_rows_train,21162
|
||||||
|
expert_rows_test,3916
|
||||||
|
lr_rows_train,1897
|
||||||
|
lr_rows_test,310
|
||||||
|
text_sha256,8ba73dc9d378037333cc4ef1629d59350237f911b4fcd3dfd319cefb4769e23a
|
||||||
|
expert_full_sha256,30b53244055c18ae480197cfde4cdd86f30e0c9bd5af60d371caa5647b2a362a
|
||||||
|
lr_full_sha256,66912cc39d63b4bf63506220268e2eee2d5d94fc9c8b861eb0e25d7e79f492d2
|
||||||
|
union_mapping_sha256,3861d65e279458e339a220df6040c98f0f5237476065d01c855e38ad2164fe5c
|
||||||
|
@@ -0,0 +1,23 @@
|
|||||||
|
stratum,eligible_parties,blocked_parties
|
||||||
|
Asia-Pacific / pre-1990,12,2
|
||||||
|
Europe / 1990s,121,24
|
||||||
|
Europe / pre-1990,103,21
|
||||||
|
Europe / 2010s,34,7
|
||||||
|
Europe / 2000s,47,9
|
||||||
|
North America / pre-1990,6,1
|
||||||
|
Latin America / pre-1990,10,2
|
||||||
|
Latin America / 1990s,2,1
|
||||||
|
Latin America / 2000s,7,1
|
||||||
|
Other / 2000s,6,1
|
||||||
|
Asia-Pacific / 2010s,4,1
|
||||||
|
Other / 2020s,2,1
|
||||||
|
Other / 1990s,9,2
|
||||||
|
Asia-Pacific / 1990s,5,1
|
||||||
|
Other / pre-1990,3,1
|
||||||
|
Europe / 2020s,3,1
|
||||||
|
North America / 2000s,2,1
|
||||||
|
Asia-Pacific / 2000s,4,1
|
||||||
|
Other / 2010s,5,1
|
||||||
|
Latin America / 2010s,1,1
|
||||||
|
North America / 1990s,1,1
|
||||||
|
Latin America / 2020s,1,1
|
||||||
|
@@ -0,0 +1,5 @@
|
|||||||
|
8ba73dc9d378037333cc4ef1629d59350237f911b4fcd3dfd319cefb4769e23a _local/revision/blocked_party/text_data.csv
|
||||||
|
a1b5031df7754ba33234e8e2ec40bb0f0d10b0aed12b277a8ea1f7876fb0f658 _local/revision/blocked_party/expert.csv
|
||||||
|
8129e85cd13877c7a5c8698430be279265d1bb9b9dd16438f55b7159597b5af7 _local/revision/blocked_party/lr_data.csv
|
||||||
|
ed05b52f16139ebce88d3d281e19f4b8140fa1b0f0791fa8fb00ac2ccf347c1f _local/revision/blocked_party/expert_test.csv
|
||||||
|
fcf131031e9bac5231fdd829b4d2f0b6ac8d9a2de748175a5fb897be048a7abe _local/revision/blocked_party/lr_data_test.csv
|
||||||
@@ -0,0 +1,39 @@
|
|||||||
|
{
|
||||||
|
"year0": 1943,
|
||||||
|
"mean_ess": 3900.832212330333,
|
||||||
|
"num_samples": 1000,
|
||||||
|
"max_depth": 15,
|
||||||
|
"num_chains": 4,
|
||||||
|
"run_id": "run_2026-08-13_00-02-50",
|
||||||
|
"files": {
|
||||||
|
"chain_size_gb": 2.28,
|
||||||
|
"data": [
|
||||||
|
"expert_dim.csv",
|
||||||
|
"expert_lr.csv",
|
||||||
|
"segment_info.csv",
|
||||||
|
"segment_year_map.csv",
|
||||||
|
"stan_data.json",
|
||||||
|
"text_data.csv"
|
||||||
|
],
|
||||||
|
"total_size_gb": 9.12,
|
||||||
|
"chains": [
|
||||||
|
"chains/chain_1.csv",
|
||||||
|
"chains/chain_2.csv",
|
||||||
|
"chains/chain_3.csv",
|
||||||
|
"chains/chain_4.csv"
|
||||||
|
]
|
||||||
|
},
|
||||||
|
"max_rhat": 1.02529,
|
||||||
|
"model_file": "models/stan_model_2dim_v6.stan",
|
||||||
|
"dimensions": [
|
||||||
|
"economic_lr",
|
||||||
|
"galtan"
|
||||||
|
],
|
||||||
|
"convergence_status": "excellent",
|
||||||
|
"mean_rhat": 1.001168748932864,
|
||||||
|
"model_version": "2dim",
|
||||||
|
"min_ess": 136.47,
|
||||||
|
"num_warmup": 1000,
|
||||||
|
"timestamp": "2026-08-13_00-02-50",
|
||||||
|
"adapt_delta": 0.95
|
||||||
|
}
|
||||||
@@ -0,0 +1,6 @@
|
|||||||
|
CMDSTAN_HOME=/opt/agent-tools/cmdstan-2.39.0
|
||||||
|
JULIA_NUM_THREADS=4
|
||||||
|
STAN_NUM_THREADS=1
|
||||||
|
PARTY2D_NUM_CHAINS=4
|
||||||
|
PARTY2D_NUM_WARMUP=1000
|
||||||
|
PARTY2D_NUM_SAMPLES=1000
|
||||||
File diff suppressed because one or more lines are too long
@@ -0,0 +1 @@
|
|||||||
|
stanc3 v2.39.0 (Unix)
|
||||||
@@ -0,0 +1,48 @@
|
|||||||
|
dimension,group_type,group,n,parties,pearson_r,mae,rmse,bias_expert_minus_model,latent_interval_overlap_95,latent_interval_overlap_ci_lower,latent_interval_overlap_ci_upper,mean_interval_width,mean_nearest_text_distance
|
||||||
|
economic_lr,overall,All matched held-out ratings,1095,76,0.5475349609862943,0.15659217513079526,0.20393539783060147,-0.05532600480562651,0.5168949771689497,0.4872890521846218,0.5463827709425695,0.26430020129022835,0.16712328767123288
|
||||||
|
economic_lr,decade,1970s,118,23,0.6073768641760265,0.14287929777127256,0.19854739676618133,-0.10205433290640958,0.6101694915254238,0.5200255656951949,0.6933662479382441,0.25169848501186437,0.0
|
||||||
|
economic_lr,decade,2010s,354,57,0.5992779154544512,0.1501300890725984,0.1885321056791222,0.021169532897549623,0.5254237288135594,0.47341101610110464,0.57689056985201,0.26575748477902544,0.2740112994350282
|
||||||
|
economic_lr,decade,2000s,295,55,0.5565144172854626,0.15858892901808844,0.20126848038834502,-0.05233537589313869,0.4847457627118644,0.4282780233542677,0.5416056876150218,0.26056229448652546,0.17288135593220338
|
||||||
|
economic_lr,decade,1990s,208,50,0.6126480905939616,0.16338137608018816,0.21895575578982457,-0.125792230775093,0.5048076923076923,0.43739172918853736,0.5720492871179859,0.27198704433737986,0.15384615384615385
|
||||||
|
economic_lr,decade,1980s,108,20,0.4965786440080896,0.17854227265283956,0.23865365933045032,-0.1349662304081571,0.4722222222222222,0.38064769707967805,0.56570500173722,0.24913035199722225,0.0
|
||||||
|
economic_lr,decade,2020s,12,7,0.7503511986609598,0.11774978066574382,0.1393618688955034,0.0122078457365813,0.75,0.46768966087934005,0.9110599603710386,0.4404074548520834,0.25
|
||||||
|
economic_lr,region,Asia-Pacific,58,5,0.423328443816105,0.17092882695033917,0.20862210989480942,-0.0891987769436264,0.39655172413793105,0.28088568950111137,0.5250701719255011,0.24057815327112078,0.017241379310344827
|
||||||
|
economic_lr,region,Europe,893,59,0.5432249309980023,0.1511392922038456,0.20048220769183805,-0.04801670847392163,0.5397536394176932,0.5069623693775478,0.5722043418900828,0.2582198076309352,0.17245240761478164
|
||||||
|
economic_lr,region,North America,48,3,0.2266003540841267,0.19345335195896976,0.23606823520460213,-0.0729662252491865,0.3333333333333333,0.2167660137089678,0.47460153667527943,0.24118537241145843,0.0
|
||||||
|
economic_lr,region,Latin America,44,4,0.5475305121725508,0.233854863071928,0.26544986189092146,-0.22241677738442417,0.3409090909090909,0.21875632628639946,0.4886113202806049,0.3793346702613636,0.6363636363636364
|
||||||
|
economic_lr,region,Other,52,5,0.7564368993112075,0.13484224995906788,0.16103703074927944,0.014599836243402173,0.5769230769230769,0.4419408497943149,0.7013215209113954,0.3191783834884615,0.0
|
||||||
|
economic_lr,text_distance_class,direct text,971,74,0.5470435275586387,0.15948847158646431,0.20840482347744635,-0.06628643977796411,0.505664263645726,0.47425628054953844,0.537027603929745,0.2590549584609939,0.0
|
||||||
|
economic_lr,text_distance_class,nearby text (1--3 years),124,41,0.6274910136969545,0.1339123053045472,0.16479794566578806,0.030501272276146137,0.6048387096774194,0.5168822918241138,0.686494386815703,0.3053738366707661,1.4758064516129032
|
||||||
|
economic_lr,project,V-Party,914,70,0.5307923250620764,0.1655176264187938,0.21441603701361558,-0.07700198607575016,0.4890590809628009,0.45676497600150245,0.5214447717371051,0.2606530705033918,0.0437636761487965
|
||||||
|
economic_lr,project,GPS,26,26,0.6972897756799338,0.13173814981538515,0.16398882896345623,0.04775018790861701,0.5,0.32060306315002074,0.6793969368499793,0.28959384095,0.8076923076923077
|
||||||
|
economic_lr,project,CHES,129,31,0.8041829585092515,0.10700941979056187,0.13157526424371663,0.044641159665681336,0.6821705426356589,0.5975442490042815,0.7562605832169504,0.2815578695228682,0.7829457364341085
|
||||||
|
economic_lr,project,POPPA,26,21,0.8377220415965767,0.11368900666387213,0.15035280874886425,0.10760098186837214,0.6923076923076923,0.5001138848576876,0.8349887906020732,0.2815926515211539,0.8076923076923077
|
||||||
|
economic_lr,item,lrecon_vparty,457,70,0.6526133931659629,0.12080138892236934,0.1567424568814474,-0.02190935296123517,0.6148796498905909,0.5694819726873812,0.6583620413946809,0.2606530705033917,0.0437636761487965
|
||||||
|
economic_lr,item,welf_vparty,457,70,0.4769148611627237,0.21023386391521848,0.2595771100617617,-0.13209461919026505,0.36323851203501095,0.32045365106651325,0.40830347502626996,0.2606530705033917,0.0437636761487965
|
||||||
|
economic_lr,item,lrecon_gps,26,26,0.6972897756799338,0.13173814981538515,0.16398882896345623,0.04775018790861701,0.5,0.32060306315002074,0.6793969368499793,0.28959384095,0.8076923076923077
|
||||||
|
economic_lr,item,lrecon_ches,129,31,0.8041829585092515,0.10700941979056187,0.13157526424371663,0.044641159665681336,0.6821705426356589,0.5975442490042815,0.7562605832169504,0.2815578695228682,0.7829457364341085
|
||||||
|
economic_lr,item,lrecon_poppa,26,21,0.8377220415965767,0.11368900666387213,0.15035280874886425,0.10760098186837214,0.6923076923076923,0.5001138848576876,0.8349887906020732,0.2815926515211539,0.8076923076923077
|
||||||
|
galtan,overall,All matched held-out ratings,2428,76,0.3879064541660041,0.18914006264912334,0.23816373704876254,-0.013710351768187365,0.4126853377265239,0.39325537210953215,0.4323911667110884,0.258665849890383,0.0914332784184514
|
||||||
|
galtan,decade,1970s,287,23,0.2402564326635899,0.21665443860032066,0.25724972613838026,0.0744907055227787,0.3344947735191638,0.2824119799338199,0.3909497399852015,0.26474203940810126,0.0
|
||||||
|
galtan,decade,2010s,693,57,0.4606647139522965,0.1935411541181932,0.24590414523794452,-0.050660741407812654,0.3838383838383838,0.3483644977339694,0.4205930386702865,0.24824147265028856,0.12265512265512266
|
||||||
|
galtan,decade,2000s,676,55,0.3870265819854018,0.18629380051270433,0.23690167706252432,-0.0468645901306479,0.4275147928994083,0.39073352449867466,0.4651152496863708,0.258417938360392,0.11538461538461539
|
||||||
|
galtan,decade,1990s,499,50,0.41212741006828396,0.17126387793509984,0.2212178226996366,-0.019602997434269034,0.48897795591182364,0.4453697785277533,0.5327545453149837,0.26606359520380773,0.11823647294589178
|
||||||
|
galtan,decade,1980s,266,20,0.3476988123327735,0.18921982830486436,0.2308346130602219,0.08154082475673356,0.37969924812030076,0.3234807111340447,0.4393431084697526,0.2599239549642858,0.0
|
||||||
|
galtan,decade,2020s,7,5,0.6324675234649824,0.17149569493926073,0.20386312323528286,0.030403782877832106,0.8571428571428571,0.48686549668097007,0.9743210440510253,0.49033504546428575,0.0
|
||||||
|
galtan,region,Asia-Pacific,142,5,-0.06990999699587813,0.2028535844559974,0.2421448847066388,-0.04764709776122232,0.43661971830985913,0.3577770958258817,0.5188013289856945,0.31393695714260567,0.007042253521126761
|
||||||
|
galtan,region,Europe,1938,59,0.4006536844413476,0.18821986618869885,0.2391185533726696,-0.007048457123688323,0.4107327141382869,0.3890267037664939,0.4327919244882515,0.24984948892156866,0.07791537667698659
|
||||||
|
galtan,region,North America,117,3,0.15158098770839884,0.2020264962636105,0.24484307272242728,0.04760720430768876,0.2905982905982906,0.21602760266916315,0.37848289702630594,0.23609932028311967,0.0
|
||||||
|
galtan,region,Latin America,110,4,0.510877747351673,0.15390455215418858,0.19797883572628092,-0.07917475272712499,0.6272727272727273,0.5340503169531756,0.7119054672241376,0.3662500658409092,0.6363636363636364
|
||||||
|
galtan,region,Other,121,5,0.3370413394210674,0.20735670781668103,0.24493355112293227,-0.08036162321796078,0.33884297520661155,0.2606254540535898,0.4269786779024248,0.25902643284276866,0.0
|
||||||
|
galtan,text_distance_class,direct text,2293,74,0.3817373914639877,0.1904499145514716,0.23993647489228756,-0.014524078018235874,0.404709986916703,0.38479506099767413,0.42494366891735347,0.25436052910535323,0.0
|
||||||
|
galtan,text_distance_class,nearby text (1--3 years),135,40,0.4399561874640988,0.16689198552257195,0.20573340547684066,0.00011093927893245328,0.5481481481481482,0.46402181812792126,0.6296100616545071,0.3317925207057409,1.6444444444444444
|
||||||
|
galtan,project,V-Party,2273,70,0.3740693791970041,0.1911074338064693,0.24060334153635682,-0.018396438536216624,0.4047514298284206,0.3847495239974894,0.4250747518765987,0.2566477345279916,0.04399472063352398
|
||||||
|
galtan,project,GPS,26,26,0.5443693144157825,0.17671657291477977,0.21780593561007713,0.012007051328791764,0.46153846153846156,0.28755582695234905,0.6454236379556988,0.2777555471798077,0.8076923076923077
|
||||||
|
galtan,project,CHES,129,31,0.5869373239223193,0.15697863700916778,0.19496790336712624,0.06367587104738649,0.5426356589147286,0.456675986048851,0.6261294002156924,0.2903778195740311,0.7829457364341085
|
||||||
|
galtan,item,culsup_vparty,457,70,0.5366188718286774,0.16311644869435024,0.21359963974416982,0.011168223905568687,0.4638949671772429,0.418663126356892,0.5097287549316027,0.25691081806515326,0.0437636761487965
|
||||||
|
galtan,item,gender_vparty,445,70,0.32459509946925547,0.2164337137525867,0.2608755427368426,0.12306040878157333,0.34606741573033706,0.3033546905138159,0.39141513474302303,0.25556702282926974,0.0449438202247191
|
||||||
|
galtan,item,immig_vparty,457,70,0.5004863614572931,0.13311846254318913,0.16732357853899973,-0.011445561652418225,0.5776805251641138,0.5319316389137728,0.6221343134655263,0.25691081806515326,0.0437636761487965
|
||||||
|
galtan,item,lgbt_vparty,457,70,0.46354086621523605,0.15471230984590562,0.18981764808744647,0.04310285191432138,0.474835886214442,0.42945203114163966,0.5206392800594324,0.25691081806515326,0.0437636761487965
|
||||||
|
galtan,item,relig_vparty,457,70,0.4185372751933906,0.288821256864484,0.33467597534992094,-0.25415371263710096,0.15973741794310722,0.12900420700742332,0.19614352271142144,0.25691081806515326,0.0437636761487965
|
||||||
|
galtan,item,libcon_gps,26,26,0.5443693144157825,0.17671657291477977,0.21780593561007713,0.012007051328791764,0.46153846153846156,0.28755582695234905,0.6454236379556988,0.2777555471798077,0.8076923076923077
|
||||||
|
galtan,item,galtan_ches,129,31,0.5869373239223193,0.15697863700916778,0.19496790336712624,0.06367587104738649,0.5426356589147286,0.456675986048851,0.6261294002156924,0.2903778195740311,0.7829457364341085
|
||||||
|
@@ -0,0 +1,18 @@
|
|||||||
|
"party_id","target_year","label","status","actual_year","party_name","country","economic_lr","economic_lower","economic_upper","galtan","galtan_lower","galtan_upper","source_support_class","era"
|
||||||
|
383,1972,"SPD 1972","exact",1972,"Social Democratic Party of Germany","DE",0.271573270375,0.162936225,0.38761415,0.370199619125,0.2680677,0.473299475,"both_direct_or_nearby","Historical & Cold War Era (1970–1999)"
|
||||||
|
383,2021,"SPD 2021","exact",2021,"Social Democratic Party of Germany","DE",0.2363366825,0.148430575,0.330595375,0.36824540875,0.27656695,0.461600275,"both_direct_or_nearby","Contemporary Era (2000–2022)"
|
||||||
|
1375,1983,"CDU 1983","exact",1983,"Christian Democratic Union","DE",0.704186118875,0.559885375,0.832835125,0.6167079735,0.5088347,0.716463575,"text_only_direct_or_nearby","Historical & Cold War Era (1970–1999)"
|
||||||
|
1375,2021,"CDU 2021","exact",2021,"Christian Democratic Union","DE",0.57714137175,0.45681705,0.69505505,0.536185121875,0.433751725,0.636189175,"text_only_direct_or_nearby","Contemporary Era (2000–2022)"
|
||||||
|
1516,1983,"Labour 1983","exact",1983,"Labour Party","GB",0.259080600525,0.1453436,0.3814767,0.34731692875,0.243851525,0.45443645,"both_direct_or_nearby","Historical & Cold War Era (1970–1999)"
|
||||||
|
1516,1997,"Labour 1997","exact",1997,"Labour Party","GB",0.540093913375,0.45312245,0.63004375,0.50597976425,0.44681665,0.566031675,"both_direct_or_nearby","Historical & Cold War Era (1970–1999)"
|
||||||
|
1567,1979,"Conservatives 1979","exact",1979,"Conservative Party","GB",0.863733304625,0.78370345,0.932120225,0.58421918175,0.47261405,0.691922375,"both_direct_or_nearby","Historical & Cold War Era (1970–1999)"
|
||||||
|
1567,2019,"Conservatives 2019","exact",2019,"Conservative Party","GB",0.734806347125,0.6674518,0.805338225,0.604817389125,0.5601045,0.6518521,"both_direct_or_nearby","Contemporary Era (2000–2022)"
|
||||||
|
487,1994,"Swedish SAP 1994","exact",1994,"Social Democratic Labour Party","SE",0.58329747225,0.45880465,0.706358175,0.323307833,0.22678185,0.42106565,"both_direct_or_nearby","Historical & Cold War Era (1970–1999)"
|
||||||
|
487,2022,"Swedish SAP 2022","exact",2022,"Social Democratic Labour Party","SE",0.271161611625,0.18001335,0.364595125,0.431600183875,0.333020475,0.530597425,"both_direct_or_nearby","Contemporary Era (2000–2022)"
|
||||||
|
409,2010,"Sweden Democrats 2010","exact",2010,"Sweden Democrats","SE",0.57402493025,0.47088145,0.681297275,0.721373298875,0.653960175,0.78768305,"both_direct_or_nearby","Contemporary Era (2000–2022)"
|
||||||
|
409,2022,"Sweden Democrats 2022","exact",2022,"Sweden Democrats","SE",0.609638368125,0.490017925,0.7238046,0.753425659,0.653108875,0.84672335,"both_direct_or_nearby","Contemporary Era (2000–2022)"
|
||||||
|
433,1988,"French FN 1988","exact",1988,"National Front","FR",0.78687783275,0.6924592,0.8758855,0.797982186375,0.7315981,0.86105335,"both_direct_or_nearby","Historical & Cold War Era (1970–1999)"
|
||||||
|
433,2022,"French FN 2022","exact",2022,"National Front","FR",0.524320955125,0.398589175,0.645052275,0.829604491,0.73907015,0.911430725,"both_direct_or_nearby","Contemporary Era (2000–2022)"
|
||||||
|
1545,2021,"The Left 2021","exact",2021,"The Left","DE",0.05463340125125,0.0205853625,0.104672875,0.2266517688,0.13204575,0.336095325,"both_direct_or_nearby","Contemporary Era (2000–2022)"
|
||||||
|
432,2020,"US Democrats 2020","exact",2020,"Democratic Party","US",0.299892256625,0.1963003,0.410606025,0.3345769065,0.263694925,0.405684575,"both_direct_or_nearby","Contemporary Era (2000–2022)"
|
||||||
|
809,2020,"US Republicans 2020","exact",2020,"Republican Party","US",0.8507495345,0.77742365,0.917357075,0.721311676375,0.65728435,0.78442835,"both_direct_or_nearby","Contemporary Era (2000–2022)"
|
||||||
|
@@ -0,0 +1,147 @@
|
|||||||
|
"party_id","party_label","country","year","source_support_class","dimension","estimate","lower","upper"
|
||||||
|
1567,"United Kingdom: Conservatives","GB",1945,"text_only_direct_or_nearby","Economic",0.7897200205,0.68672145,0.882517725
|
||||||
|
1567,"United Kingdom: Conservatives","GB",1950,"text_only_direct_or_nearby","Economic",0.696324671375,0.58645,0.7999215
|
||||||
|
1567,"United Kingdom: Conservatives","GB",1951,"text_only_direct_or_nearby","Economic",0.648801438125,0.509441425,0.7756712
|
||||||
|
1567,"United Kingdom: Conservatives","GB",1955,"text_only_direct_or_nearby","Economic",0.559596760875,0.4111588,0.696031575
|
||||||
|
1567,"United Kingdom: Conservatives","GB",1959,"text_only_direct_or_nearby","Economic",0.56006642225,0.388645375,0.718084975
|
||||||
|
1567,"United Kingdom: Conservatives","GB",1964,"text_only_direct_or_nearby","Economic",0.713843190625,0.5883803,0.829548525
|
||||||
|
1567,"United Kingdom: Conservatives","GB",1966,"text_only_direct_or_nearby","Economic",0.79834168225,0.6939884,0.890500425
|
||||||
|
1567,"United Kingdom: Conservatives","GB",1970,"both_direct_or_nearby","Economic",0.67897010575,0.55999,0.79320825
|
||||||
|
1567,"United Kingdom: Conservatives","GB",1974,"both_direct_or_nearby","Economic",0.68530368975,0.6022573,0.771763125
|
||||||
|
1567,"United Kingdom: Conservatives","GB",1979,"both_direct_or_nearby","Economic",0.863733304625,0.78370345,0.932120225
|
||||||
|
1567,"United Kingdom: Conservatives","GB",1983,"both_direct_or_nearby","Economic",0.897143095,0.823332125,0.953360675
|
||||||
|
1567,"United Kingdom: Conservatives","GB",1987,"both_direct_or_nearby","Economic",0.864987285375,0.782492075,0.9328753
|
||||||
|
1567,"United Kingdom: Conservatives","GB",1992,"both_direct_or_nearby","Economic",0.80683245025,0.731234925,0.88153825
|
||||||
|
1567,"United Kingdom: Conservatives","GB",1997,"both_direct_or_nearby","Economic",0.815790791,0.741322225,0.886000175
|
||||||
|
1567,"United Kingdom: Conservatives","GB",2001,"both_direct_or_nearby","Economic",0.735943766375,0.64358755,0.825612475
|
||||||
|
1567,"United Kingdom: Conservatives","GB",2005,"both_direct_or_nearby","Economic",0.7468415025,0.657667875,0.8347423
|
||||||
|
1567,"United Kingdom: Conservatives","GB",2010,"both_direct_or_nearby","Economic",0.807639387875,0.737347625,0.877409025
|
||||||
|
1567,"United Kingdom: Conservatives","GB",2015,"both_direct_or_nearby","Economic",0.650294524875,0.582851025,0.72313155
|
||||||
|
1567,"United Kingdom: Conservatives","GB",2017,"both_direct_or_nearby","Economic",0.680683625125,0.607886575,0.758350075
|
||||||
|
1567,"United Kingdom: Conservatives","GB",2019,"both_direct_or_nearby","Economic",0.734806347125,0.6674518,0.805338225
|
||||||
|
1567,"United Kingdom: Conservatives","GB",2024,"both_direct_or_nearby","Economic",0.696441988625,0.611536275,0.7845477
|
||||||
|
383,"Germany: SPD","DE",1949,"text_only_direct_or_nearby","Economic",0.2594422284125,0.124083075,0.416025325
|
||||||
|
383,"Germany: SPD","DE",1953,"text_only_direct_or_nearby","Economic",0.3044620151625,0.172417475,0.44553015
|
||||||
|
383,"Germany: SPD","DE",1957,"text_only_direct_or_nearby","Economic",0.364714905375,0.217271,0.518439625
|
||||||
|
383,"Germany: SPD","DE",1961,"text_only_direct_or_nearby","Economic",0.45071135275,0.3107251,0.58867
|
||||||
|
383,"Germany: SPD","DE",1965,"text_only_direct_or_nearby","Economic",0.40004889475,0.253314775,0.5547562
|
||||||
|
383,"Germany: SPD","DE",1969,"text_only_direct_or_nearby","Economic",0.36430581075,0.234999825,0.496653
|
||||||
|
383,"Germany: SPD","DE",1972,"both_direct_or_nearby","Economic",0.271573270375,0.162936225,0.38761415
|
||||||
|
383,"Germany: SPD","DE",1976,"both_direct_or_nearby","Economic",0.22760841445,0.1408578,0.317522325
|
||||||
|
383,"Germany: SPD","DE",1980,"both_direct_or_nearby","Economic",0.2854944402625,0.176353725,0.402044875
|
||||||
|
383,"Germany: SPD","DE",1983,"both_direct_or_nearby","Economic",0.3201057485,0.198932275,0.449359525
|
||||||
|
383,"Germany: SPD","DE",1987,"both_direct_or_nearby","Economic",0.3082503073875,0.1890146,0.43586885
|
||||||
|
383,"Germany: SPD","DE",1990,"both_direct_or_nearby","Economic",0.2997481731625,0.1826493,0.42186305
|
||||||
|
383,"Germany: SPD","DE",1994,"both_direct_or_nearby","Economic",0.344258320125,0.247450525,0.440361825
|
||||||
|
383,"Germany: SPD","DE",1998,"both_direct_or_nearby","Economic",0.42037938775,0.335025575,0.504438225
|
||||||
|
383,"Germany: SPD","DE",2002,"both_direct_or_nearby","Economic",0.354348980875,0.285402475,0.419218875
|
||||||
|
383,"Germany: SPD","DE",2005,"both_direct_or_nearby","Economic",0.408510935875,0.326714325,0.4912897
|
||||||
|
383,"Germany: SPD","DE",2009,"both_direct_or_nearby","Economic",0.2258038923125,0.154613,0.2986832
|
||||||
|
383,"Germany: SPD","DE",2013,"both_direct_or_nearby","Economic",0.2547526295,0.177729475,0.3305311
|
||||||
|
383,"Germany: SPD","DE",2017,"both_direct_or_nearby","Economic",0.2187940564125,0.150431,0.288139
|
||||||
|
383,"Germany: SPD","DE",2021,"both_direct_or_nearby","Economic",0.2363366825,0.148430575,0.330595375
|
||||||
|
383,"Germany: SPD","DE",2025,"both_direct_or_nearby","Economic",0.24988987125,0.155485225,0.353246025
|
||||||
|
379,"Denmark: Social Democrats","DK",1945,"text_only_direct_or_nearby","Economic",0.39289630125,0.306313775,0.4744366
|
||||||
|
379,"Denmark: Social Democrats","DK",1947,"text_only_direct_or_nearby","Economic",0.4133295765,0.326647125,0.497578325
|
||||||
|
379,"Denmark: Social Democrats","DK",1950,"text_only_direct_or_nearby","Economic",0.402314539625,0.313989,0.48990055
|
||||||
|
379,"Denmark: Social Democrats","DK",1953,"text_only_direct_or_nearby","Economic",0.404059824625,0.31188405,0.492347325
|
||||||
|
379,"Denmark: Social Democrats","DK",1957,"text_only_direct_or_nearby","Economic",0.386770333875,0.2915608,0.4740622
|
||||||
|
379,"Denmark: Social Democrats","DK",1960,"text_only_direct_or_nearby","Economic",0.39128767025,0.30583805,0.46915175
|
||||||
|
379,"Denmark: Social Democrats","DK",1964,"text_only_direct_or_nearby","Economic",0.41323511225,0.3319368,0.490628525
|
||||||
|
379,"Denmark: Social Democrats","DK",1966,"text_only_direct_or_nearby","Economic",0.424408032125,0.348613025,0.502216325
|
||||||
|
379,"Denmark: Social Democrats","DK",1968,"text_only_direct_or_nearby","Economic",0.426020581125,0.346608825,0.502008425
|
||||||
|
379,"Denmark: Social Democrats","DK",1971,"both_direct_or_nearby","Economic",0.3838693815,0.31048555,0.4534922
|
||||||
|
379,"Denmark: Social Democrats","DK",1973,"both_direct_or_nearby","Economic",0.36916859375,0.292048475,0.4383149
|
||||||
|
379,"Denmark: Social Democrats","DK",1975,"both_direct_or_nearby","Economic",0.2551262250375,0.1655264,0.348229075
|
||||||
|
379,"Denmark: Social Democrats","DK",1977,"both_direct_or_nearby","Economic",0.2030203874375,0.11497425,0.301540125
|
||||||
|
379,"Denmark: Social Democrats","DK",1979,"both_direct_or_nearby","Economic",0.1862332029375,0.10144465,0.2859892
|
||||||
|
379,"Denmark: Social Democrats","DK",1981,"both_direct_or_nearby","Economic",0.1919036092625,0.1020977,0.2975764
|
||||||
|
379,"Denmark: Social Democrats","DK",1984,"both_direct_or_nearby","Economic",0.1734391782625,0.08982039,0.2725962
|
||||||
|
379,"Denmark: Social Democrats","DK",1987,"both_direct_or_nearby","Economic",0.1872400464625,0.10188485,0.281978775
|
||||||
|
379,"Denmark: Social Democrats","DK",1988,"both_direct_or_nearby","Economic",0.1844144026,0.10014465,0.278666375
|
||||||
|
379,"Denmark: Social Democrats","DK",1990,"both_direct_or_nearby","Economic",0.1593657011125,0.0820198525,0.250674325
|
||||||
|
379,"Denmark: Social Democrats","DK",1994,"both_direct_or_nearby","Economic",0.18027195025,0.0911998025,0.286319075
|
||||||
|
379,"Denmark: Social Democrats","DK",1998,"both_direct_or_nearby","Economic",0.2332968959625,0.14133775,0.3324363
|
||||||
|
379,"Denmark: Social Democrats","DK",2001,"both_direct_or_nearby","Economic",0.25665459875,0.167550175,0.3497182
|
||||||
|
379,"Denmark: Social Democrats","DK",2005,"both_direct_or_nearby","Economic",0.243111738,0.15668705,0.33332095
|
||||||
|
379,"Denmark: Social Democrats","DK",2007,"both_direct_or_nearby","Economic",0.2504832553,0.1616851,0.347202325
|
||||||
|
379,"Denmark: Social Democrats","DK",2011,"both_direct_or_nearby","Economic",0.277551306,0.187687975,0.3698644
|
||||||
|
379,"Denmark: Social Democrats","DK",2015,"both_direct_or_nearby","Economic",0.2492670365,0.160125675,0.34358355
|
||||||
|
379,"Denmark: Social Democrats","DK",2019,"both_direct_or_nearby","Economic",0.254542113875,0.180740975,0.32727345
|
||||||
|
409,"Sweden: Sweden Democrats","SE",2010,"both_direct_or_nearby","Economic",0.57402493025,0.47088145,0.681297275
|
||||||
|
409,"Sweden: Sweden Democrats","SE",2014,"both_direct_or_nearby","Economic",0.514751535125,0.4326713,0.602047575
|
||||||
|
409,"Sweden: Sweden Democrats","SE",2018,"both_direct_or_nearby","Economic",0.499110805875,0.397517425,0.59893585
|
||||||
|
409,"Sweden: Sweden Democrats","SE",2022,"both_direct_or_nearby","Economic",0.609638368125,0.490017925,0.7238046
|
||||||
|
1567,"United Kingdom: Conservatives","GB",1945,"text_only_direct_or_nearby","Cultural",0.42283955775,0.30617415,0.539308425
|
||||||
|
1567,"United Kingdom: Conservatives","GB",1950,"text_only_direct_or_nearby","Cultural",0.597652829625,0.493804425,0.695553875
|
||||||
|
1567,"United Kingdom: Conservatives","GB",1951,"text_only_direct_or_nearby","Cultural",0.5963407815,0.4953524,0.694196075
|
||||||
|
1567,"United Kingdom: Conservatives","GB",1955,"text_only_direct_or_nearby","Cultural",0.4481433145,0.349619975,0.546246075
|
||||||
|
1567,"United Kingdom: Conservatives","GB",1959,"text_only_direct_or_nearby","Cultural",0.379823589625,0.24070795,0.5271287
|
||||||
|
1567,"United Kingdom: Conservatives","GB",1964,"text_only_direct_or_nearby","Cultural",0.4359372675,0.309309225,0.5597016
|
||||||
|
1567,"United Kingdom: Conservatives","GB",1966,"text_only_direct_or_nearby","Cultural",0.433826746,0.305561125,0.564284875
|
||||||
|
1567,"United Kingdom: Conservatives","GB",1970,"both_direct_or_nearby","Cultural",0.586682156125,0.487144225,0.681696075
|
||||||
|
1567,"United Kingdom: Conservatives","GB",1974,"both_direct_or_nearby","Cultural",0.582726903625,0.512885,0.65400205
|
||||||
|
1567,"United Kingdom: Conservatives","GB",1979,"both_direct_or_nearby","Cultural",0.58421918175,0.47261405,0.691922375
|
||||||
|
1567,"United Kingdom: Conservatives","GB",1983,"both_direct_or_nearby","Cultural",0.54862467925,0.4436958,0.651204175
|
||||||
|
1567,"United Kingdom: Conservatives","GB",1987,"both_direct_or_nearby","Cultural",0.5044723465,0.398577575,0.60606875
|
||||||
|
1567,"United Kingdom: Conservatives","GB",1992,"both_direct_or_nearby","Cultural",0.608807442,0.53829935,0.67952055
|
||||||
|
1567,"United Kingdom: Conservatives","GB",1997,"both_direct_or_nearby","Cultural",0.653249609625,0.5903779,0.719252725
|
||||||
|
1567,"United Kingdom: Conservatives","GB",2001,"both_direct_or_nearby","Cultural",0.62022832225,0.5519112,0.68966615
|
||||||
|
1567,"United Kingdom: Conservatives","GB",2005,"both_direct_or_nearby","Cultural",0.63051460775,0.562743875,0.6984813
|
||||||
|
1567,"United Kingdom: Conservatives","GB",2010,"both_direct_or_nearby","Cultural",0.515056946125,0.464064225,0.5684148
|
||||||
|
1567,"United Kingdom: Conservatives","GB",2015,"both_direct_or_nearby","Cultural",0.53108352025,0.47741145,0.585770875
|
||||||
|
1567,"United Kingdom: Conservatives","GB",2017,"both_direct_or_nearby","Cultural",0.550617655875,0.503622825,0.597269
|
||||||
|
1567,"United Kingdom: Conservatives","GB",2019,"both_direct_or_nearby","Cultural",0.604817389125,0.5601045,0.6518521
|
||||||
|
1567,"United Kingdom: Conservatives","GB",2024,"both_direct_or_nearby","Cultural",0.64147235475,0.568830525,0.717038025
|
||||||
|
383,"Germany: SPD","DE",1949,"text_only_direct_or_nearby","Cultural",0.4562653665,0.29105445,0.6148909
|
||||||
|
383,"Germany: SPD","DE",1953,"text_only_direct_or_nearby","Cultural",0.41201424225,0.239960775,0.59720855
|
||||||
|
383,"Germany: SPD","DE",1957,"text_only_direct_or_nearby","Cultural",0.361933815225,0.2020638,0.530546125
|
||||||
|
383,"Germany: SPD","DE",1961,"text_only_direct_or_nearby","Cultural",0.3403875836625,0.189634975,0.502483725
|
||||||
|
383,"Germany: SPD","DE",1965,"text_only_direct_or_nearby","Cultural",0.349500525125,0.197610625,0.519782425
|
||||||
|
383,"Germany: SPD","DE",1969,"text_only_direct_or_nearby","Cultural",0.367453271375,0.2312429,0.51644905
|
||||||
|
383,"Germany: SPD","DE",1972,"both_direct_or_nearby","Cultural",0.370199619125,0.2680677,0.473299475
|
||||||
|
383,"Germany: SPD","DE",1976,"both_direct_or_nearby","Cultural",0.28787436475,0.21332925,0.364130925
|
||||||
|
383,"Germany: SPD","DE",1980,"both_direct_or_nearby","Cultural",0.346500680875,0.245533525,0.448428875
|
||||||
|
383,"Germany: SPD","DE",1983,"both_direct_or_nearby","Cultural",0.3643932085,0.2585273,0.474535775
|
||||||
|
383,"Germany: SPD","DE",1987,"both_direct_or_nearby","Cultural",0.373242374375,0.266124675,0.479610175
|
||||||
|
383,"Germany: SPD","DE",1990,"both_direct_or_nearby","Cultural",0.354967389625,0.25200245,0.460687325
|
||||||
|
383,"Germany: SPD","DE",1994,"both_direct_or_nearby","Cultural",0.410000861625,0.3313879,0.488224525
|
||||||
|
383,"Germany: SPD","DE",1998,"both_direct_or_nearby","Cultural",0.458140040875,0.389836225,0.525399375
|
||||||
|
383,"Germany: SPD","DE",2002,"both_direct_or_nearby","Cultural",0.442940016,0.3876151,0.4964075
|
||||||
|
383,"Germany: SPD","DE",2005,"both_direct_or_nearby","Cultural",0.456470195875,0.39176415,0.5189629
|
||||||
|
383,"Germany: SPD","DE",2009,"both_direct_or_nearby","Cultural",0.389663805625,0.32557465,0.451253625
|
||||||
|
383,"Germany: SPD","DE",2013,"both_direct_or_nearby","Cultural",0.2109102855,0.15503785,0.268407625
|
||||||
|
383,"Germany: SPD","DE",2017,"both_direct_or_nearby","Cultural",0.40809383225,0.35346185,0.45866175
|
||||||
|
383,"Germany: SPD","DE",2021,"both_direct_or_nearby","Cultural",0.36824540875,0.27656695,0.461600275
|
||||||
|
383,"Germany: SPD","DE",2025,"both_direct_or_nearby","Cultural",0.3699090585,0.27155195,0.46999925
|
||||||
|
379,"Denmark: Social Democrats","DK",1945,"text_only_direct_or_nearby","Cultural",0.367253073125,0.222935775,0.513292475
|
||||||
|
379,"Denmark: Social Democrats","DK",1947,"text_only_direct_or_nearby","Cultural",0.38745918675,0.234832075,0.545806525
|
||||||
|
379,"Denmark: Social Democrats","DK",1950,"text_only_direct_or_nearby","Cultural",0.3968814625,0.2531828,0.544707225
|
||||||
|
379,"Denmark: Social Democrats","DK",1953,"text_only_direct_or_nearby","Cultural",0.42331649125,0.283525625,0.5604742
|
||||||
|
379,"Denmark: Social Democrats","DK",1957,"text_only_direct_or_nearby","Cultural",0.42351804925,0.268476675,0.589941825
|
||||||
|
379,"Denmark: Social Democrats","DK",1960,"text_only_direct_or_nearby","Cultural",0.384011194875,0.24432495,0.52920335
|
||||||
|
379,"Denmark: Social Democrats","DK",1964,"text_only_direct_or_nearby","Cultural",0.362245514375,0.224588175,0.502448675
|
||||||
|
379,"Denmark: Social Democrats","DK",1966,"text_only_direct_or_nearby","Cultural",0.3679157005,0.230511325,0.5031338
|
||||||
|
379,"Denmark: Social Democrats","DK",1968,"text_only_direct_or_nearby","Cultural",0.382175081625,0.25659165,0.507958075
|
||||||
|
379,"Denmark: Social Democrats","DK",1971,"both_direct_or_nearby","Cultural",0.470225764375,0.36279785,0.577130125
|
||||||
|
379,"Denmark: Social Democrats","DK",1973,"both_direct_or_nearby","Cultural",0.501466689875,0.3986878,0.60298055
|
||||||
|
379,"Denmark: Social Democrats","DK",1975,"both_direct_or_nearby","Cultural",0.51072920025,0.41364105,0.608167125
|
||||||
|
379,"Denmark: Social Democrats","DK",1977,"both_direct_or_nearby","Cultural",0.48888551625,0.3865325,0.588461525
|
||||||
|
379,"Denmark: Social Democrats","DK",1979,"both_direct_or_nearby","Cultural",0.469331547625,0.36461055,0.5720459
|
||||||
|
379,"Denmark: Social Democrats","DK",1981,"both_direct_or_nearby","Cultural",0.440505821125,0.334911475,0.544872125
|
||||||
|
379,"Denmark: Social Democrats","DK",1984,"both_direct_or_nearby","Cultural",0.427927746125,0.3199225,0.535149325
|
||||||
|
379,"Denmark: Social Democrats","DK",1987,"both_direct_or_nearby","Cultural",0.42510295375,0.332117075,0.518948075
|
||||||
|
379,"Denmark: Social Democrats","DK",1988,"both_direct_or_nearby","Cultural",0.4189467795,0.3290549,0.51075845
|
||||||
|
379,"Denmark: Social Democrats","DK",1990,"both_direct_or_nearby","Cultural",0.394384283125,0.289929725,0.498693825
|
||||||
|
379,"Denmark: Social Democrats","DK",1994,"both_direct_or_nearby","Cultural",0.365051991,0.259654975,0.469250275
|
||||||
|
379,"Denmark: Social Democrats","DK",1998,"both_direct_or_nearby","Cultural",0.4057949125,0.318039475,0.493913625
|
||||||
|
379,"Denmark: Social Democrats","DK",2001,"both_direct_or_nearby","Cultural",0.377267655375,0.2986888,0.456555875
|
||||||
|
379,"Denmark: Social Democrats","DK",2005,"both_direct_or_nearby","Cultural",0.3840174395,0.296652025,0.472492425
|
||||||
|
379,"Denmark: Social Democrats","DK",2007,"both_direct_or_nearby","Cultural",0.377073825875,0.288222975,0.46642125
|
||||||
|
379,"Denmark: Social Democrats","DK",2011,"both_direct_or_nearby","Cultural",0.44217709925,0.351102525,0.5332076
|
||||||
|
379,"Denmark: Social Democrats","DK",2015,"both_direct_or_nearby","Cultural",0.5274942445,0.44580275,0.608968575
|
||||||
|
379,"Denmark: Social Democrats","DK",2019,"both_direct_or_nearby","Cultural",0.440971318625,0.38293125,0.4998911
|
||||||
|
409,"Sweden: Sweden Democrats","SE",2010,"both_direct_or_nearby","Cultural",0.721373298875,0.653960175,0.78768305
|
||||||
|
409,"Sweden: Sweden Democrats","SE",2014,"both_direct_or_nearby","Cultural",0.750375772125,0.690054075,0.81067825
|
||||||
|
409,"Sweden: Sweden Democrats","SE",2018,"both_direct_or_nearby","Cultural",0.671777804625,0.5936524,0.747156725
|
||||||
|
409,"Sweden: Sweden Democrats","SE",2022,"both_direct_or_nearby","Cultural",0.753425659,0.653108875,0.84672335
|
||||||
|
Some files were not shown because too many files have changed in this diff Show More
Reference in New Issue
Block a user